From c08efdbe2b9aaf2a81d0348314ca0a535f3012c4 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 3 Sep 2026 17:55:52 +0700 Subject: [PATCH 01/78] feat(gemini): persist and replay thoughtSignature with session namespace - Add open-sse/services/thoughtSignatureStore.js managing LRU Map (2k) + SQLite kv table - Store thoughtSignature with sessionId namespace and toolCallId fallback - Replay cached signature by sessionId:tool_call_id to prevent multi-process collisions - Normalize Antigravity sessionId to numeric int64 format Co-Authored-By: Claude Code --- open-sse/executors/antigravity.js | 40 +++-- open-sse/handlers/chatCore.js | 2 +- .../handlers/chatCore/streamingHandler.js | 10 +- open-sse/services/thoughtSignatureStore.js | 170 ++++++++++++++++++ .../translator/request/openai-to-gemini.js | 40 +++-- .../translator/response/gemini-to-openai.js | 30 +++- open-sse/utils/stream.js | 10 +- 7 files changed, 263 insertions(+), 39 deletions(-) create mode 100644 open-sse/services/thoughtSignatureStore.js diff --git a/open-sse/executors/antigravity.js b/open-sse/executors/antigravity.js index 35ec1006..f2ad630a 100644 --- a/open-sse/executors/antigravity.js +++ b/open-sse/executors/antigravity.js @@ -3,10 +3,11 @@ import { BaseExecutor } from "./base.js"; import { PROVIDERS } from "../config/providers.js"; import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js"; import { HTTP_STATUS } from "../config/runtimeConfig.js"; -import { resolveSessionId } from "../utils/sessionManager.js"; +import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js"; import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js"; +import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js"; // Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63} function sanitizeFunctionName(name) { @@ -187,6 +188,9 @@ export class AntigravityExecutor extends BaseExecutor { }; } + const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }); + const sessionId = toNumericSessionId(rawSessionId) || rawSessionId; + // ─── Standard (non-image) request ─── // Fix contents for Claude models via Antigravity const contents = body.request?.contents?.map(c => { @@ -202,17 +206,31 @@ export class AntigravityExecutor extends BaseExecutor { return true; }); // Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE) - // don't persist thoughtSignature in their history, so backfill the default signature on any - // functionCall part that arrives without one. - const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false; - if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) { + // don't persist thoughtSignature in their history, so backfill from cache or default signature. + // In parallel function calls, only the first call needs a signature; siblings stay unsigned. + let firstFunctionCallSeen = false; + const modifiedParts = parts?.map(p => { + if (!p.functionCall) return p; + const callId = p.functionCall.id; + const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null; + const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined); + firstFunctionCallSeen = true; + if (callSig) { + return { ...p, thoughtSignature: callSig }; + } + if (p.thoughtSignature && !cachedSig) { + // Unsigned sibling call + const { thoughtSignature: _, ...rest } = p; + return rest; + } + return p; + }); + + const partsChanged = parts?.length !== c.parts?.length || modifiedParts?.some((p, idx) => p !== c.parts[idx]); + if (role !== c.role || partsChanged) { return { ...c, role, - parts: needsBackfill - ? parts.map(p => (p.functionCall && !p.thoughtSignature) - ? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE } - : p) - : parts, + parts: modifiedParts || parts, }; } return c; @@ -267,7 +285,7 @@ export class AntigravityExecutor extends BaseExecutor { generationConfig, ...(contents && { contents }), ...(tools && { tools }), - sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }), + sessionId, safetySettings: undefined, ...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } }) }; diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index 67e377ea..3c083978 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -469,7 +469,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred // Streaming response const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx }); - return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId }); + return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials }); } export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) { diff --git a/open-sse/handlers/chatCore/streamingHandler.js b/open-sse/handlers/chatCore/streamingHandler.js index be008e7d..7b7634e2 100644 --- a/open-sse/handlers/chatCore/streamingHandler.js +++ b/open-sse/handlers/chatCore/streamingHandler.js @@ -22,7 +22,7 @@ const CODEX_SOURCE_TO_TARGET = { /** * Determine which SSE transform stream to use based on provider/format. */ -function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) { +function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) { const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli"); // Responses-API providers (e.g. codex) emit Responses SSE → translate into client format const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES; @@ -30,11 +30,11 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, if (needsCodexTranslation) { const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI; - return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames); + return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials); } if (needsTranslation(targetFormat, sourceFormat)) { - return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames); + return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials); } return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey); @@ -43,7 +43,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, /** * Handle streaming response — pipe provider SSE through transform stream to client. */ -export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) { +export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) { if (onRequestSuccess) { Promise.resolve() .then(onRequestSuccess) @@ -79,7 +79,7 @@ export async function handleStreamingResponse({ providerResponse, provider, mode }; } - const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }); + const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }); // Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES; diff --git a/open-sse/services/thoughtSignatureStore.js b/open-sse/services/thoughtSignatureStore.js new file mode 100644 index 00000000..eb8a60e4 --- /dev/null +++ b/open-sse/services/thoughtSignatureStore.js @@ -0,0 +1,170 @@ +import { makeKv } from "../../src/lib/db/helpers/kvStore.js"; + +const MAX_SIGNATURES = 2000; +const MAX_PERSISTED_SIGNATURES = 10_000; +const MEMORY_TTL_MS = 1000 * 60 * 60; // 1 hour +const PERSISTED_TTL_MS = 1000 * 60 * 60 * 24 * 7; // 7 days +const SCOPE = "gemini_thought_signatures"; + +const signatureKv = makeKv(SCOPE); +const memorySignatures = new Map(); +let pruneCounter = 0; + +function pruneMemoryExpired() { + const now = Date.now(); + for (const [key, value] of memorySignatures.entries()) { + if (value.expiresAt <= now) { + memorySignatures.delete(key); + } + } + + while (memorySignatures.size > MAX_SIGNATURES) { + const oldestKey = memorySignatures.keys().next().value; + if (!oldestKey) break; + memorySignatures.delete(oldestKey); + } +} + +async function maybePrunePersisted() { + pruneCounter++; + if (pruneCounter % 100 !== 0) return; + + try { + const all = await signatureKv.getAll(); + const keys = Object.keys(all); + const now = Date.now(); + const expiredKeys = []; + const valid = []; + + for (const k of keys) { + const entry = all[k]; + if (!entry || typeof entry.signature !== "string" || (entry.expiresAt && entry.expiresAt <= now)) { + expiredKeys.push(k); + } else { + valid.push({ key: k, createdAt: entry.createdAt || 0 }); + } + } + + for (const k of expiredKeys) { + await signatureKv.remove(k).catch(() => {}); + } + + if (valid.length > MAX_PERSISTED_SIGNATURES) { + valid.sort((a, b) => b.createdAt - a.createdAt); + const toRemove = valid.slice(MAX_PERSISTED_SIGNATURES); + for (const item of toRemove) { + await signatureKv.remove(item.key).catch(() => {}); + } + } + } catch { + // Fail-open + } +} + +/** + * Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async) + */ +export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) { + if (typeof toolCallId !== "string" || !toolCallId) return; + if (typeof signature !== "string" || !signature) return; + + const now = Date.now(); + pruneMemoryExpired(); + + const keys = []; + if (sessionId && typeof sessionId === "string") { + keys.push(`${sessionId}:${toolCallId}`); + } + keys.push(toolCallId); + + for (const k of keys) { + memorySignatures.set(k, { + signature, + expiresAt: now + MEMORY_TTL_MS, + }); + + // Async persist to SQLite kv table without blocking + signatureKv.set(k, { + signature, + createdAt: now, + expiresAt: now + PERSISTED_TTL_MS, + }).catch(() => {}); + } + + maybePrunePersisted().catch(() => {}); +} + +/** + * Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback) + */ +export async function getGeminiThoughtSignature(toolCallId, sessionId = null) { + if (typeof toolCallId !== "string" || !toolCallId) return null; + + pruneMemoryExpired(); + + if (sessionId && typeof sessionId === "string") { + const sessionKey = `${sessionId}:${toolCallId}`; + const sessionEntry = memorySignatures.get(sessionKey); + if (sessionEntry && sessionEntry.expiresAt > Date.now()) { + return sessionEntry.signature; + } + } + + const entry = memorySignatures.get(toolCallId); + if (entry && entry.expiresAt > Date.now()) { + return entry.signature; + } + + try { + if (sessionId && typeof sessionId === "string") { + const sessionKey = `${sessionId}:${toolCallId}`; + const sessionRow = await signatureKv.get(sessionKey); + if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) { + memorySignatures.set(sessionKey, { + signature: sessionRow.signature, + expiresAt: Date.now() + MEMORY_TTL_MS, + }); + return sessionRow.signature; + } + } + + const row = await signatureKv.get(toolCallId); + if (row && typeof row.signature === "string") { + if (row.expiresAt && row.expiresAt <= Date.now()) { + signatureKv.remove(toolCallId).catch(() => {}); + return null; + } + memorySignatures.set(toolCallId, { + signature: row.signature, + expiresAt: Date.now() + MEMORY_TTL_MS, + }); + return row.signature; + } + } catch { + // Fail-open + } + + return null; +} + +/** + * Synchronous get from RAM cache only (for sync translators) + */ +export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) { + if (typeof toolCallId !== "string" || !toolCallId) return null; + pruneMemoryExpired(); + + if (sessionId && typeof sessionId === "string") { + const sessionKey = `${sessionId}:${toolCallId}`; + const sessionEntry = memorySignatures.get(sessionKey); + if (sessionEntry && sessionEntry.expiresAt > Date.now()) { + return sessionEntry.signature; + } + } + + const entry = memorySignatures.get(toolCallId); + if (entry && entry.expiresAt > Date.now()) { + return entry.signature; + } + return null; +} diff --git a/open-sse/translator/request/openai-to-gemini.js b/open-sse/translator/request/openai-to-gemini.js index 9e029e2c..2b05670d 100644 --- a/open-sse/translator/request/openai-to-gemini.js +++ b/open-sse/translator/request/openai-to-gemini.js @@ -2,6 +2,7 @@ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js"; import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js"; +import { getGeminiThoughtSignatureSync } from "../../services/thoughtSignatureStore.js"; function generateUUID() { return crypto.randomUUID(); } @@ -46,7 +47,7 @@ function normalizeGeminiContents(contents) { } // Core: Convert OpenAI request to Gemini format (base for all variants) -function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) { +function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) { const result = { model: model, contents: [], @@ -133,18 +134,27 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG if (msg.tool_calls && Array.isArray(msg.tool_calls)) { const toolCallIds = []; + let firstFunctionCallSeen = false; for (const tc of msg.tool_calls) { if (tc.type !== OPENAI_BLOCK.FUNCTION) continue; const args = tryParseJSON(tc.function?.arguments || "{}"); - parts.push({ - thoughtSignature: signature, + const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null; + // First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig + const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined); + firstFunctionCallSeen = true; + + const part = { functionCall: { id: tc.id, name: sanitizeGeminiFunctionName(tc.function.name), args: args } - }); + }; + if (callSig) { + part.thoughtSignature = callSig; + } + parts.push(part); toolCallIds.push(tc.id); } @@ -232,13 +242,13 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG } // OpenAI -> Gemini (standard API) -export function openaiToGeminiRequest(model, body, stream) { - return openaiToGeminiBase(model, body, stream); +export function openaiToGeminiRequest(model, body, stream, credentials = null) { + return openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_AG_SIGNATURE, credentials?._clientSessionId); } // OpenAI -> Gemini CLI (Cloud Code Assist) -export function openaiToGeminiCLIRequest(model, body, stream) { - const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE); +export function openaiToGeminiCLIRequest(model, body, stream, credentials = null) { + const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE, credentials?._clientSessionId); // Thinking is normalized centrally by applyThinking (thinkingUnified.js) after translation. // Clean schema for tools @@ -335,18 +345,26 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu const parts = []; if (Array.isArray(msg.content)) { + let firstToolUseSeen = false; for (const block of msg.content) { if (block.type === CLAUDE_BLOCK.TEXT) { parts.push({ text: block.text }); } else if (block.type === CLAUDE_BLOCK.TOOL_USE) { - parts.push({ - thoughtSignature: signature, + const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null; + const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined); + firstToolUseSeen = true; + + const part = { functionCall: { id: block.id, name: sanitizeGeminiFunctionName(block.name), args: block.input || {} } - }); + }; + if (callSig) { + part.thoughtSignature = callSig; + } + parts.push(part); } else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) { let content = block.content; if (Array.isArray(content)) { diff --git a/open-sse/translator/response/gemini-to-openai.js b/open-sse/translator/response/gemini-to-openai.js index 8f04ffd3..66bf1022 100644 --- a/open-sse/translator/response/gemini-to-openai.js +++ b/open-sse/translator/response/gemini-to-openai.js @@ -6,6 +6,7 @@ import { toOpenAIUsage } from "../concerns/usage.js"; import { reasoningDelta } from "../concerns/reasoning.js"; import { encodeDataUri } from "../concerns/image.js"; import { toOpenAIFinish } from "../concerns/finishReason.js"; +import { storeGeminiThoughtSignature } from "../../services/thoughtSignatureStore.js"; // Build chunk meta for current gemini state function chunkMeta(state) { @@ -13,14 +14,18 @@ function chunkMeta(state) { } // Build a tool_call chunk from a gemini functionCall part (shared by sig/non-sig branches) -function emitFunctionCall(functionCall, state) { +function emitFunctionCall(functionCall, state, signature = null) { const rawName = functionCall.name; // Restore original tool name from mapping (AG cloaking) const fcName = state.toolNameMap?.get(rawName) || rawName; const fcArgs = functionCall.args || {}; const toolCallIndex = state.functionIndex++; + const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`; + if (signature) { + storeGeminiThoughtSignature(callId, signature, state.sessionId); + } const toolCall = { - id: `${fcName}-${Date.now()}-${toolCallIndex}`, + id: callId, index: toolCallIndex, type: OPENAI_BLOCK.FUNCTION, function: { name: fcName, arguments: JSON.stringify(fcArgs) }, @@ -57,13 +62,21 @@ export function geminiToOpenAIResponse(chunk, state) { if (content?.parts) { for (const part of content.parts) { const hasThoughtSig = part.thoughtSignature || part.thought_signature; + if (hasThoughtSig && typeof hasThoughtSig === "string") { + state.pendingThoughtSignature = hasThoughtSig; + } const isThought = part.thought === true; - + // Handle thought signature (thinking mode) if (hasThoughtSig) { const hasTextContent = part.text !== undefined && part.text !== ""; const hasFunctionCall = !!part.functionCall; - + + // Standalone thoughtSignature part (no text, no functionCall): keep pending for next functionCall + if (!hasTextContent && !hasFunctionCall) { + continue; + } + if (hasTextContent) { results.push(buildChunk( chunkMeta(state), @@ -71,9 +84,10 @@ export function geminiToOpenAIResponse(chunk, state) { null )); } - + if (hasFunctionCall) { - results.push(emitFunctionCall(part.functionCall, state)); + results.push(emitFunctionCall(part.functionCall, state, hasThoughtSig)); + state.pendingThoughtSignature = null; } continue; } @@ -92,7 +106,9 @@ export function geminiToOpenAIResponse(chunk, state) { // Function call if (part.functionCall) { - results.push(emitFunctionCall(part.functionCall, state)); + const sig = state.pendingThoughtSignature || null; + results.push(emitFunctionCall(part.functionCall, state, sig)); + state.pendingThoughtSignature = null; } // Inline data (images) diff --git a/open-sse/utils/stream.js b/open-sse/utils/stream.js index daca8d6a..15ee37d8 100644 --- a/open-sse/utils/stream.js +++ b/open-sse/utils/stream.js @@ -49,7 +49,8 @@ export function createSSEStream(options = {}) { connectionId = null, body = null, onStreamComplete = null, - apiKey = null + apiKey = null, + credentials = null } = options; let buffer = ""; @@ -59,7 +60,7 @@ export function createSSEStream(options = {}) { const decoder = new TextDecoder("utf-8", { fatal: false }); const state = mode === STREAM_MODE.TRANSLATE - ? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model } + ? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null } : null; let totalContentLength = 0; @@ -487,7 +488,7 @@ export function createSSEStream(options = {}) { }); } -export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null) { +export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null, credentials = null) { return createSSEStream({ mode: STREAM_MODE.TRANSLATE, targetFormat, @@ -500,7 +501,8 @@ export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, p connectionId, body, onStreamComplete, - apiKey + apiKey, + credentials }); } From 2ab6a4c949dba6b24ddfd6985b17f64822102957 Mon Sep 17 00:00:00 2001 From: hangyu Date: Thu, 3 Sep 2026 23:01:53 +0700 Subject: [PATCH 02/78] feat(qoder): refresh model catalog, add capability mapping and image pass-through - Registry/constants: drop qmodel_preview/gm51model, add lite, qmodel_38max (Qwen3.8-Max), qfmodel (Qwen3.8-Flash), gmodel (GLM-5.3), gfmodel (GLM-5.3-Flash) - capabilities: add PROVIDER_CAPABILITIES['qoder'] so opaque internal ids resolve to their real models' context windows and limits - executor: preserve image blocks instead of flattening away, convert Claude-style image blocks, and hash images into chat_record_id - tests: cover image preservation, data-URI and Claude-block conversion - build(docker): use CN mirrors for apk and npm --- Dockerfile | 5 +- open-sse/executors/qoder.js | 74 +++++++++++++++++++++++++--- open-sse/providers/capabilities.js | 32 ++++++++++++ open-sse/providers/registry/qoder.js | 7 ++- open-sse/shared/qoder/constants.js | 5 +- tests/unit/qoder.test.js | 64 ++++++++++++++++++++++++ 6 files changed, 176 insertions(+), 11 deletions(-) diff --git a/Dockerfile b/Dockerfile index 9e41ee2f..dfcee598 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,14 +2,15 @@ ARG NODE_IMAGE=node:22-alpine FROM ${NODE_IMAGE} AS base WORKDIR /app +# CN mirror for apk (used by builder and runner stages) +RUN sed -i 's|dl-cdn.alpinelinux.org|mirrors.aliyun.com|g' /etc/apk/repositories FROM base AS builder RUN apk --no-cache upgrade && apk --no-cache add python3 make g++ linux-headers COPY package.json ./ -RUN --mount=type=cache,target=/root/.npm \ - npm install +RUN npm install --registry=https://registry.npmmirror.com COPY . ./ ENV NEXT_TELEMETRY_DISABLED=1 diff --git a/open-sse/executors/qoder.js b/open-sse/executors/qoder.js index e52a00df..86055aba 100644 --- a/open-sse/executors/qoder.js +++ b/open-sse/executors/qoder.js @@ -37,10 +37,13 @@ import { QODER_MODEL_MAP, } from "../shared/qoder/constants.js"; import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js"; +import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js"; +import { encodeDataUri } from "../translator/concerns/image.js"; /** * Hoist role:"system" messages out of the messages array (Qoder rejects - * system in messages) and flatten any multipart content arrays. + * system in messages) and flatten multipart content arrays — EXCEPT image + * blocks, which are preserved (see normalizeContent). */ function normalizeMessages(messages) { if (!Array.isArray(messages) || messages.length === 0) { @@ -50,18 +53,72 @@ function normalizeMessages(messages) { const out = []; for (const msg of messages) { if (!msg || typeof msg !== "object") continue; - const text = extractText(msg.content); if (msg.role === "system") { + const text = extractText(msg.content); if (text) systemParts.push(text); continue; } const cloned = { ...msg }; - cloned.content = text; + cloned.content = normalizeContent(msg.content); out.push(cloned); } return { messages: out, systemText: systemParts.join("\n\n") }; } +/** + * Normalize one message's content for Qoder. + * + * Text-only content is flattened to a plain string (Qoder's historical + * shape). When images are present the content stays an array and image + * blocks are kept as OpenAI-style `image_url` parts — verified against the + * upstream: it accepts both http(s) URLs and inline base64 data: URIs + * directly, no pre-upload to the /image/upload OSS flow required (that is + * a qodercli client-side choice, not a protocol requirement). The legacy + * top-level `image_urls` / `chat_context.imageUrls` slots stay null — + * qodercli leaves them null too. + * + * Claude-style `{type:"image", source:{...}}` blocks are converted to + * `image_url` so claude-format clients also round-trip. + */ +function normalizeContent(content) { + if (typeof content === "string") return content; + if (content == null) return ""; + if (!Array.isArray(content)) return String(content); + + const blocks = []; + const textParts = []; + let hasImage = false; + for (const item of content) { + if (!item || typeof item !== "object") continue; + if (item.type === OPENAI_BLOCK.IMAGE_URL && typeof item.image_url?.url === "string" && item.image_url.url) { + blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: item.image_url.url } }); + hasImage = true; + } else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) { + // Claude base64/url image → OpenAI image_url equivalent. + const src = item.source; + const url = src.type === "base64" && src.data + ? encodeDataUri(src.media_type || "image/png", src.data) + : typeof src.url === "string" && src.url ? src.url : null; + if (url) { + blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } }); + hasImage = true; + } + } else if (typeof item.text === "string" && item.text) { + if (hasImage || blocks.length) { + // Keep ordering faithful once images are in play. + blocks.push({ type: OPENAI_BLOCK.TEXT, text: item.text }); + } else { + textParts.push(item.text); + } + } + } + + if (!hasImage) return textParts.join("\n"); + // Prepend any text collected before the first image block. + if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") }); + return blocks; +} + function extractText(content) { if (typeof content === "string") return content; if (content == null) return ""; @@ -84,9 +141,9 @@ function extractText(content) { function lastUserText(messages) { for (let i = messages.length - 1; i >= 0; i--) { const m = messages[i]; - if (m?.role === "user" && typeof m.content === "string") { - return m.content; - } + if (m?.role !== "user") continue; + if (typeof m.content === "string") return m.content; + if (Array.isArray(m.content)) return extractText(m.content); } return ""; } @@ -110,6 +167,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) { if (m.role) { h.update("\0"); h.update(m.role); } if (typeof m.content === "string" && m.content) { h.update("\0"); h.update(m.content); + } else if (Array.isArray(m.content)) { + // Include image refs so the same prompt with a different image gets + // a distinct chat_record_id. + h.update("\0"); + try { h.update(JSON.stringify(m.content)); } catch {} } } if (tools) { diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 5a0b7271..91c6266b 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -206,6 +206,38 @@ export const PROVIDER_CAPABILITIES = { "deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, "deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 }, }, + // Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the + // registry `name` is display-only and capability lookup matches on the raw + // id, so every qoder model would fall through to DEFAULT_CAPABILITIES + // (200K) without this map. contextWindow follows the real model family's + // spec: the /algo/api/v2/model/list max_input_tokens under-reports some + // windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more). + // max_output_tokens arrives as 0 for every model, so outputs are + // best-guess from the real model family. Vision tags below follow the + // upstream is_vl flag per explicit request, even though the executor + // currently sends image_urls:null (image pass-through over the agent_chat + // SSE protocol is unverified). reasoning:true on all of them — every model can + // reason; the upstream is_reasoning flag only drives model_config selection. + // thinkingFormat keeps the true-model family for documentation/UI, but + // thinkingCanDisable:false everywhere: the executor only forwards + // messages/tools/max_tokens, and thinking is fixed upstream via + // modelConfig.is_reasoning — client thinking intent is dropped, so "none" + // must never be offered as an option. + "qoder": { + "ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5 + "performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5 + "dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro + "dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash + "gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3 + "gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash + "kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3 + "kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code + "mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3 + "qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max + "qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus + "qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash + "qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max + }, // Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output). "poolside": { "laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 }, diff --git a/open-sse/providers/registry/qoder.js b/open-sse/providers/registry/qoder.js index fe76fd72..042c83ae 100644 --- a/open-sse/providers/registry/qoder.js +++ b/open-sse/providers/registry/qoder.js @@ -30,12 +30,15 @@ export default { { id: "auto", name: "Auto" }, { id: "performance", name: "Performance" }, { id: "efficient", name: "Efficient" }, - { id: "qmodel_preview", name: "Qwen3.8-Max-Preview" }, + { id: "lite", name: "Lite" }, + { id: "qmodel_38max", name: "Qwen3.8-Max" }, { id: "qmodel_latest", name: "Qwen3.7-Max" }, { id: "qmodel", name: "Qwen3.7-Plus" }, + { id: "qfmodel", name: "Qwen3.8-Flash" }, { id: "kmodel_latest", name: "Kimi-K3" }, { id: "kmodel", name: "Kimi-K2.7-Code" }, - { id: "gm51model", name: "GLM-5.2" }, + { id: "gmodel", name: "GLM-5.3" }, + { id: "gfmodel", name: "GLM-5.3-Flash" }, { id: "dmodel", name: "DeepSeek-V4-Pro" }, { id: "dfmodel", name: "DeepSeek-V4-Flash" }, { id: "mmodel", name: "MiniMax-M3" }, diff --git a/open-sse/shared/qoder/constants.js b/open-sse/shared/qoder/constants.js index e2635f40..861e8f30 100644 --- a/open-sse/shared/qoder/constants.js +++ b/open-sse/shared/qoder/constants.js @@ -54,10 +54,13 @@ export const QODER_MODEL_MAP = { lite: "lite", // Frontier models qmodel: "qmodel", + qfmodel: "qfmodel", qmodel_latest: "qmodel_latest", + qmodel_38max: "qmodel_38max", dmodel: "dmodel", dfmodel: "dfmodel", - gm51model: "gm51model", + gmodel: "gmodel", + gfmodel: "gfmodel", kmodel: "kmodel", mmodel: "mmodel", }; diff --git a/tests/unit/qoder.test.js b/tests/unit/qoder.test.js index d8fce1ae..04ce1c45 100644 --- a/tests/unit/qoder.test.js +++ b/tests/unit/qoder.test.js @@ -367,6 +367,70 @@ describe("normalizeMessages", () => { expect(result.messages).toEqual([]); expect(result.systemText).toBe(""); }); + + it("preserves image_url blocks (http URL) instead of dropping them", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "describe" }, + { type: "image_url", image_url: { url: "https://example.com/a.png" } }, + ], + }, + ]); + const content = result.messages[0].content; + expect(Array.isArray(content)).toBe(true); + expect(content).toContainEqual({ type: "text", text: "describe" }); + expect(content).toContainEqual({ type: "image_url", image_url: { url: "https://example.com/a.png" } }); + }); + + it("preserves base64 data: URI images (no OSS upload needed)", () => { + const dataUri = "data:image/png;base64,iVBORw0KGgo="; + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "image_url", image_url: { url: dataUri } }, + { type: "text", text: "what color?" }, + ], + }, + ]); + const content = result.messages[0].content; + expect(Array.isArray(content)).toBe(true); + expect(content[0]).toEqual({ type: "image_url", image_url: { url: dataUri } }); + expect(content.some((b) => b.type === "text" && b.text === "what color?")).toBe(true); + }); + + it("converts claude-style base64 image blocks to image_url data URIs", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "see this" }, + { type: "image", source: { type: "base64", media_type: "image/jpeg", data: "AAAA" } }, + ], + }, + ]); + const content = result.messages[0].content; + expect(content).toContainEqual({ + type: "image_url", + image_url: { url: "data:image/jpeg;base64,AAAA" }, + }); + }); + + it("drops image blocks with no usable url but keeps the text", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "hi" }, + { type: "image_url", image_url: {} }, + { type: "image", source: { type: "base64" } }, + ], + }, + ]); + expect(result.messages[0].content).toBe("hi"); + }); }); describe("wrapQoderSSE", () => { From b84681d5a4f12a14db3cd3c8cc3b8c979c7a645f Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 3 Sep 2026 23:27:04 +0700 Subject: [PATCH 03/78] feat(cli-tools): replace copilot mitm with vscode extension setup guide Co-Authored-By: Claude Code --- .../cli-tools/[toolId]/ToolDetailClient.js | 4 +- .../cli-tools/components/ToolSummaryCard.js | 5 +- src/app/api/cli-tools/all-statuses/route.js | 2 - src/shared/constants/cliTools.js | 54 ++++++++++--------- 4 files changed, 32 insertions(+), 33 deletions(-) diff --git a/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js b/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js index 209a6d33..82211e48 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js +++ b/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js @@ -8,7 +8,7 @@ import { getModelsByProviderId, PROVIDER_ID_TO_ALIAS } from "@/shared/constants/ import { ClaudeToolCard, CodexToolCard, DroidToolCard, OpenClawToolCard, HermesToolCard, DefaultToolCard, OpenCodeToolCard, CoworkToolCard, - CopilotToolCard, ClineToolCard, KiloToolCard, DeepSeekTuiToolCard, + ClineToolCard, KiloToolCard, DeepSeekTuiToolCard, JcodeToolCard, GrokBuildToolCard, } from "../components"; @@ -156,8 +156,6 @@ export default function ToolDetailClient({ toolId, machineId }) { return ; case "hermes": return ; - case "copilot": - return ; case "cline": return ; case "kilo": diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js b/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js index 6e944d79..83d6c15b 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js @@ -5,7 +5,8 @@ import Image from "next/image"; import { Card } from "@/shared/components"; // Derive simple connected/configured/not-installed status from API payload -function getStatus(status) { +function getStatus(status, tool) { + if (tool?.configType === "guide") return { label: "Guide", cls: "bg-blue-500/10 text-blue-600 dark:text-blue-400" }; if (!status) return { label: "Unknown", cls: "bg-gray-500/10 text-gray-500" }; if (!status.installed) return { label: "Not installed", cls: "bg-gray-500/10 text-gray-500" }; if (status.has9Router) return { label: "Connected", cls: "bg-green-500/10 text-green-600 dark:text-green-400" }; @@ -13,7 +14,7 @@ function getStatus(status) { } export default function ToolSummaryCard({ toolId, tool, status }) { - const s = getStatus(status); + const s = getStatus(status, tool); return ( diff --git a/src/app/api/cli-tools/all-statuses/route.js b/src/app/api/cli-tools/all-statuses/route.js index 5925e526..b4270163 100644 --- a/src/app/api/cli-tools/all-statuses/route.js +++ b/src/app/api/cli-tools/all-statuses/route.js @@ -8,7 +8,6 @@ import { GET as droidGet } from "../droid-settings/route"; import { GET as openclawGet } from "../openclaw-settings/route"; import { GET as hermesGet } from "../hermes-settings/route"; import { GET as coworkGet } from "../cowork-settings/route"; -import { GET as copilotGet } from "../copilot-settings/route"; import { GET as clineGet } from "../cline-settings/route"; import { GET as kiloGet } from "../kilo-settings/route"; import { GET as deepseekTuiGet } from "../deepseek-tui-settings/route"; @@ -24,7 +23,6 @@ const STATUS_GETTERS = { openclaw: openclawGet, hermes: hermesGet, cowork: coworkGet, - copilot: copilotGet, cline: clineGet, kilo: kiloGet, "deepseek-tui": deepseekTuiGet, diff --git a/src/shared/constants/cliTools.js b/src/shared/constants/cliTools.js index 9a388866..0f7c0236 100644 --- a/src/shared/constants/cliTools.js +++ b/src/shared/constants/cliTools.js @@ -30,32 +30,6 @@ export const MITM_TOOLS = { { id: "gemini-3-flash", name: "Gemini 3 Flash (Command)", alias: "gemini-3-flash" }, ], }, - copilot: { - id: "copilot", - name: "GitHub Copilot", - image: "/providers/copilot.png", - color: "#1F6FEB", - description: "GitHub Copilot IDE with MITM", - configType: "mitm", - mitmDomain: "api.individual.githubcopilot.com", - modelAliases: ["gpt-5-mini", "gpt-5.4-nano", "claude-haiku-4.5", "gpt-4o", "gpt-4.1"], - defaultModels: [ - // Verified via live MITM passthrough capture of the GitHub Copilot CLI: its model - // picker offers "GPT-5 mini" (default → wire id "gpt-5-mini"), "Claude Haiku 4.5" - // ("claude-haiku-4.5") and "Auto". "Auto" is NOT a wire id — Copilot dispatches - // concrete models dynamically (observed "gpt-5.4-nano" for light tasks and - // "claude-haiku-4.5"), so it needs no slot of its own. Without a slot for - // gpt-5-mini / gpt-5.4-nano, getMappedModel returns null and the /chat/completions - // call is passed through to GitHub Copilot instead of the configured provider — - // and gpt-5-mini is the CLI default, so the primary turn leaks (same class as the - // Kiro "auto" misrouting). gpt-4o / gpt-4.1 are kept for the VS Code Copilot Chat picker. - { id: "gpt-5-mini", name: "GPT-5 mini", alias: "gpt-5-mini" }, - { id: "gpt-5.4-nano", name: "GPT-5.4 nano", alias: "gpt-5.4-nano" }, - { id: "claude-haiku-4.5", name: "Claude Haiku 4.5", alias: "claude-haiku-4.5" }, - { id: "gpt-4o", name: "GPT-4o", alias: "gpt-4o" }, - { id: "gpt-4.1", name: "GPT-4.1", alias: "gpt-4.1" }, - ], - }, kiro: { id: "kiro", name: "Kiro", @@ -140,6 +114,34 @@ export const CLI_TOOLS = { description: "OpenAI Codex CLI", configType: "custom", }, + copilot: { + id: "copilot", + name: "GitHub Copilot", + image: "/providers/copilot.png", + color: "#1F6FEB", + description: "GitHub Copilot in VS Code via 9Router extension", + configType: "guide", + docsUrl: "https://marketplace.visualstudio.com/items?itemName=hotrungnhan.9router-for-github-copilot", + guideSteps: [ + { + step: 1, + title: "Install Extension", + desc: "In VS Code, open Extensions (Ctrl+Shift+X or Cmd+Shift+X), search for '9Router for Github Copilot' and click Install.", + }, + { + step: 2, + title: "Configure Server", + desc: "Press Cmd+Shift+P (or Ctrl+Shift+P), run '9Router: Configure Server', then enter your Server URL and API Key:", + value: "{{baseUrl}}", + copyable: true, + }, + { + step: 3, + title: "Select Model in Copilot Chat", + desc: "Open Copilot Chat, click the model picker at the bottom → 'Manage Models...' → check the 9Router models to use.", + }, + ], + }, opencode: { id: "opencode", name: "OpenCode", From f388b5e56ba80cd99825bdd34469d00da39bd44f Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 4 Sep 2026 10:17:26 +0700 Subject: [PATCH 04/78] chore(logger): remove noisy background token refresh logs Co-Authored-By: Claude Code --- src/sse/services/backgroundTokenRefresh.js | 37 +++------------------- 1 file changed, 4 insertions(+), 33 deletions(-) diff --git a/src/sse/services/backgroundTokenRefresh.js b/src/sse/services/backgroundTokenRefresh.js index b10a86ba..5810e049 100644 --- a/src/sse/services/backgroundTokenRefresh.js +++ b/src/sse/services/backgroundTokenRefresh.js @@ -87,10 +87,7 @@ async function refreshOne(connection) { * @param {{ loadConnections?: Function, refreshConnection?: Function }} [deps] */ export async function runBackgroundTokenRefreshTick(deps = {}) { - if (tickRunning) { - log.debug("BG_TOKEN_REFRESH", "Tick already running, skip"); - return; - } + if (tickRunning) return; tickRunning = true; try { const load = deps.loadConnections || loadActiveConnections; @@ -99,26 +96,12 @@ export async function runBackgroundTokenRefreshTick(deps = {}) { const connections = await load(); const due = selectConnectionsNeedingRefresh(connections, Date.now()); - if (due.length === 0) { - log.debug("BG_TOKEN_REFRESH", "No connections due for refresh", { - active: Array.isArray(connections) ? connections.length : 0, - }); - return; - } - - log.info("BG_TOKEN_REFRESH", "Refreshing due OAuth connections", { - due: due.length, - ids: due.map((c) => c.id).filter(Boolean), - }); + if (due.length === 0) return; await Promise.allSettled( due.map(async (conn) => { try { await refresh(conn); - log.info("BG_TOKEN_REFRESH", "Connection refresh finished", { - id: conn.id, - provider: conn.provider, - }); } catch (err) { log.warn("BG_TOKEN_REFRESH", "Connection refresh failed (swallowed)", { id: conn?.id, @@ -144,14 +127,8 @@ export async function runBackgroundTokenRefreshTick(deps = {}) { */ export function startBackgroundTokenRefresh({ intervalMs } = {}) { if (started) return false; - if (isTruthyEnv(process.env.DISABLE_BACKGROUND_TOKEN_REFRESH)) { - log.info("BG_TOKEN_REFRESH", "Disabled via DISABLE_BACKGROUND_TOKEN_REFRESH"); - return false; - } - if (isNonServerRuntime()) { - log.debug("BG_TOKEN_REFRESH", "Skip start outside long-running server runtime"); - return false; - } + if (isTruthyEnv(process.env.DISABLE_BACKGROUND_TOKEN_REFRESH)) return false; + if (isNonServerRuntime()) return false; started = true; const period = Number.isFinite(intervalMs) && intervalMs > 0 ? intervalMs : DEFAULT_INTERVAL_MS; @@ -171,11 +148,6 @@ export function startBackgroundTokenRefresh({ intervalMs } = {}) { intervalHandle = setInterval(safeTick, period); if (intervalHandle.unref) intervalHandle.unref(); - log.info("BG_TOKEN_REFRESH", "Scheduler started", { - intervalMs: period, - initialDelayMs: INITIAL_DELAY_MS, - leadMs: BACKGROUND_REFRESH_LEAD_MS, - }); return true; } @@ -190,6 +162,5 @@ export function stopBackgroundTokenRefresh() { } if (started) { started = false; - log.info("BG_TOKEN_REFRESH", "Scheduler stopped"); } } From ed963931b42831afa7271564eb3dd30c6445d2a1 Mon Sep 17 00:00:00 2001 From: An Nguyen <264959754+anndev-69@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:00:51 +0700 Subject: [PATCH 05/78] feat(codex): add GPT-5.6 Sol, Terra, and Luna image aliases (#3806) --- gitbook/content/en/providers/subscription.md | 21 ++++++ open-sse/providers/registry/codex.js | 3 + tests/unit/codex-image-models.test.js | 76 ++++++++++++++++++++ tests/unit/image-generation.test.js | 6 +- 4 files changed, 103 insertions(+), 3 deletions(-) create mode 100644 tests/unit/codex-image-models.test.js diff --git a/gitbook/content/en/providers/subscription.md b/gitbook/content/en/providers/subscription.md index 830f429e..ffe5fb0d 100644 --- a/gitbook/content/en/providers/subscription.md +++ b/gitbook/content/en/providers/subscription.md @@ -111,6 +111,27 @@ Model: cx/gpt-5.2-codex | `cx/gpt-5.2` | GPT 5.2 | General tasks | | `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding | +### Image Generation + +The Codex image catalog includes `cx/gpt-5.6-sol-image`, +`cx/gpt-5.6-terra-image`, and `cx/gpt-5.6-luna-image`, alongside the existing +GPT 5.5, 5.4, and 5.3 image aliases. Select them under **Image → OpenAI Codex** +in the dashboard, or discover them with `GET /v1/models/image` after connecting +a Codex account. + +```bash +curl http://localhost:20128/v1/images/generations \ + -H "Authorization: Bearer $NINE_ROUTER_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"model":"cx/gpt-5.6-sol-image","prompt":"A blue square","size":"1024x1024"}' +``` + +These are 9Router aliases: the image adapter removes `-image` and sends the +underlying model an `image_generation` tool through the Codex Responses API. +The same endpoint accepts an `image` reference for edits. Image generation +requires an eligible ChatGPT Plus or higher account; availability of each +underlying model and its image tool depends on the connected account. + ### Pro Tips - **5-hour rolling quota** - Fresh quota every 5 hours diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 6fc7501d..427c47a4 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -59,6 +59,9 @@ export default { { id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" }, { id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" }, { id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" }, + { id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, diff --git a/tests/unit/codex-image-models.test.js b/tests/unit/codex-image-models.test.js new file mode 100644 index 00000000..9a4e9c44 --- /dev/null +++ b/tests/unit/codex-image-models.test.js @@ -0,0 +1,76 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { getModelsByProviderId, getModelType, isValidModel } from "../../open-sse/config/providerModels.js"; +import { getModelInfoCore } from "../../open-sse/services/model.js"; +import { handleImageGenerationCore } from "../../open-sse/handlers/imageGenerationCore.js"; + +const models = ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra"]; + +afterEach(() => vi.unstubAllGlobals()); + +describe("Codex GPT-5.6 image models", () => { + it.each(models)("exposes %s-image as an image model while retaining its chat entry", (model) => { + const catalog = getModelsByProviderId("codex"); + expect(catalog.filter((entry) => entry.id === `${model}-image`)).toHaveLength(1); + expect(catalog.find((entry) => entry.id === `${model}-image`)).toMatchObject({ + kind: "image", + capabilities: ["text2img", "edit"], + params: ["size", "quality", "background", "image_detail", "output_format"], + }); + expect(isValidModel("cx", `${model}-image`)).toBe(true); + expect(getModelType("cx", `${model}-image`)).toBe("image"); + expect(catalog.find((entry) => entry.id === model)).toBeDefined(); + expect(getModelType("cx", model)).not.toBe("image"); + }); + + it.each(models)("routes %s-image edits and streams image events", async (model) => { + const events = [ + ["response.image_generation_call.partial_image", { partial_image_b64: "cGFydGlhbA==", partial_image_index: 0 }], + ["response.output_item.done", { item: { type: "image_generation_call", result: "ZmluYWw=" } }], + ]; + const fetchMock = vi.fn().mockResolvedValue(new Response( + events.map(([event, data]) => `event: ${event}\ndata: ${JSON.stringify(data)}\n\n`).join(""), + { headers: { "Content-Type": "text/event-stream" } }, + )); + vi.stubGlobal("fetch", fetchMock); + const onRequestSuccess = vi.fn(); + const modelInfo = await getModelInfoCore(`cx/${model}-image`); + expect(modelInfo).toEqual({ provider: "codex", model: `${model}-image` }); + expect(await getModelInfoCore(`codex/${model}-image`)).toEqual(modelInfo); + + const result = await handleImageGenerationCore({ + modelInfo, + body: { + prompt: "Make the square blue", + image: "data:image/png;base64,cmVmZXJlbmNl", + image_detail: "low", + size: "1024x1024", + quality: "high", + background: "transparent", + output_format: "WEBP", + }, + credentials: { accessToken: "test-token" }, + streamToClient: true, + onRequestSuccess, + }); + + expect(result.success).toBe(true); + expect(result.response.status).toBe(200); + expect(result.response.headers.get("content-type")).toBe("text/event-stream"); + const [url, options] = fetchMock.mock.calls[0]; + expect(url).toBe("https://chatgpt.com/backend-api/codex/responses"); + const upstreamBody = JSON.parse(options.body); + expect(upstreamBody.model).toBe(model); + expect(upstreamBody.tools).toEqual([{ + type: "image_generation", output_format: "webp", size: "1024x1024", + quality: "high", background: "transparent", + }]); + expect(upstreamBody.input[0].content).toContainEqual({ + type: "input_image", image_url: "data:image/png;base64,cmVmZXJlbmNl", detail: "low", + }); + const stream = await result.response.text(); + expect(stream).toContain('event: partial_image\ndata: {"b64_json":"cGFydGlhbA==","index":0}'); + expect(stream).toContain("event: done\n"); + expect(stream).toContain('"data":[{"b64_json":"ZmluYWw="}]'); + expect(onRequestSuccess).toHaveBeenCalledTimes(1); + }); +}); diff --git a/tests/unit/image-generation.test.js b/tests/unit/image-generation.test.js index e5504aec..12dce95d 100644 --- a/tests/unit/image-generation.test.js +++ b/tests/unit/image-generation.test.js @@ -316,7 +316,7 @@ describe("handleImageGenerationCore", () => { expect(responseBody.data[0].b64_json).toBeTruthy(); }); - it("generates image with Codex gpt-5.5-image using current Codex version header", async () => { + it.each(["gpt-5.5", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])("generates image with Codex %s-image using current Codex version header", async (model) => { global.fetch.mockResolvedValueOnce( new Response( [ @@ -335,7 +335,7 @@ describe("handleImageGenerationCore", () => { size: "1024x1024", output_format: "png", }, - modelInfo: { provider: "codex", model: "gpt-5.5-image" }, + modelInfo: { provider: "codex", model: `${model}-image` }, credentials: { accessToken: "codex-token", providerSpecificData: { chatgptAccountId: "account-123" }, @@ -358,7 +358,7 @@ describe("handleImageGenerationCore", () => { const fetchCall = global.fetch.mock.calls[0]; const requestBody = JSON.parse(fetchCall[1].body); - expect(requestBody.model).toBe("gpt-5.5"); + expect(requestBody.model).toBe(model); expect(requestBody.tools).toEqual([ { type: "image_generation", output_format: "png", size: "1024x1024" }, ]); From cec672d9d922786a99883d42a0fadfa598513afb Mon Sep 17 00:00:00 2001 From: zmf Date: Sat, 5 Sep 2026 21:02:16 +0700 Subject: [PATCH 06/78] feat(providers): align codebuddy-cn catalog/capabilities with server config - Sync codebuddy-cn catalog and capabilities with copilot.tencent.com server payload - Fix thinkingCanDisable semantics for glm-5.3 and deepseek-v4 models - Add missing glm-5.2 thinking levels to thinkingLevels.js - Add glm-5-turbo model to glm and glm-cn registries --- open-sse/providers/capabilities.js | 44 +++++++++++---------- open-sse/providers/registry/codebuddy-cn.js | 19 +++++---- open-sse/providers/registry/glm-cn.js | 1 + open-sse/providers/registry/glm.js | 1 + open-sse/providers/thinkingLevels.js | 11 ++++-- 5 files changed, 43 insertions(+), 33 deletions(-) diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 91c6266b..7e1e757b 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -178,33 +178,37 @@ export const PROVIDER_CAPABILITIES = { // CodeBuddy.cn — authoritative per-model metadata from the gateway's model // config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision= // supportsImages). Every model reasons via OpenAI-style reasoning_effort - // (see registry thinkingFormat). `onlyReasoning` models can't turn thinking - // off → thinkingCanDisable:false (clamped to minimal instead of disabled). + // (see registry thinkingFormat). For thinkingCanDisable use the server's + // reasoning.canDisableThinking flag — see the note in the codebuddy-cn block + // below; it is NOT the inverse of onlyReasoning. "codebuddy-cn": { - "glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 }, - "glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, + "glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 }, + "glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, "glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 }, - "glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, - "glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 }, + // maxOutput 64000 per both the plugin-baked fallback and the live server + // table (the old 38000 had no source and truncated output). + "glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 }, "glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 }, - "minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 }, - "minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, + "minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 }, "kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, "kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, - "kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 }, - "hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, - // hy3/hy3-x: 256K official (192K conservative, matches hy3-preview); hy4-preview: 1M official. - // glm-5.3: 1M (GLM-5.x gen); glm-5.3-flash window unverified (200K conservative). + // Per-model values mirror the server's product-config payload (the plugin + // fetches it from copilot.tencent.com; the `models[]` entries carry + // maxInputTokens/maxOutputTokens/supportsImages). contextWindow = + // maxInputTokens, maxOutput = maxOutputTokens. Where the server and the + // plugin-baked fallback disagree, the server table wins. + // ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking — + // it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is + // on by default"; canDisableThinking means "it CAN be turned off". glm-5.3 + // and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so + // their thinking is switchable; the hy* models are forced always-on. "hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, - "hy3-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, "hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 }, - "hy4-preview-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 }, - "glm-5.3": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 }, - "glm-5.3-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, - "kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, - "deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, - "deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, - "deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 }, + "glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 }, + "glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 }, + "kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 }, + "deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 }, + "deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 }, }, // Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the // registry `name` is display-only and capability lookup matches on the raw diff --git a/open-sse/providers/registry/codebuddy-cn.js b/open-sse/providers/registry/codebuddy-cn.js index 33345b9f..fdb53e68 100644 --- a/open-sse/providers/registry/codebuddy-cn.js +++ b/open-sse/providers/registry/codebuddy-cn.js @@ -47,27 +47,26 @@ export default { models: [ { id: "glm-5.2", name: "GLM-5.2" }, { id: "glm-5.1", name: "GLM-5.1" }, - { id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" }, { id: "glm-5v-turbo", name: "GLM-5v-Turbo" }, { id: "minimax-m3", name: "MiniMax-M3" }, - { id: "minimax-m2.7", name: "MiniMax-M2.7" }, { id: "kimi-k2.7", name: "Kimi-K2.7-Code" }, { id: "kimi-k2.6", name: "Kimi-K2.6" }, - { id: "kimi-k2.5", name: "Kimi-K2.5" }, - // "-x" suffix = paid tier of the same model (free id rides the promo quota: - // hy3 free until 2026-08-31, hy4-preview until 2026-09-10). Server model table - // seen in client logs 2026-08-30; glm-5.0 / glm-4.7 removed (API 11102 dead). - { id: "hy3-preview", name: "Hy3 Preview" }, + // Catalog mirrors the server's product-config payload (the plugin fetches + // it from copilot.tencent.com). Models the server no longer publishes are + // removed even when the chat endpoint still answers them — the published + // list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x + // (endpoint returns 11102 "model service info not found"), plus + // glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview / + // deepseek-v3-2-volc (absent from the server list, though still answering + // 200) and hy3-x (paid tier, not used here). + // "-x" suffix = paid tier of the same model (free id rides the promo quota). { id: "hy3", name: "Hy3" }, - { id: "hy3-x", name: "Hy3 (Paid)" }, { id: "hy4-preview", name: "Hy4-Preview" }, - { id: "hy4-preview-x", name: "Hy4-Preview (Paid)" }, { id: "glm-5.3", name: "GLM-5.3" }, { id: "glm-5.3-flash", name: "GLM-5.3-Flash" }, { id: "kimi-k3-1", name: "Kimi-K3" }, { id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" }, { id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" }, - { id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" }, ], oauth: { baseUrl: "https://copilot.tencent.com", diff --git a/open-sse/providers/registry/glm-cn.js b/open-sse/providers/registry/glm-cn.js index 73189464..7424547e 100644 --- a/open-sse/providers/registry/glm-cn.js +++ b/open-sse/providers/registry/glm-cn.js @@ -25,6 +25,7 @@ export default { { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" }, { id: "glm-5.2", name: "GLM 5.2" }, { id: "glm-5.1", name: "GLM 5.1" }, + { id: "glm-5-turbo", name: "GLM 5 Turbo" }, { id: "glm-5", name: "GLM 5" }, { id: "glm-4.7", name: "GLM-4.7" }, { id: "glm-4.6v", name: "GLM 4.6V (Vision)" }, diff --git a/open-sse/providers/registry/glm.js b/open-sse/providers/registry/glm.js index 6c5f0f6e..88f4c563 100644 --- a/open-sse/providers/registry/glm.js +++ b/open-sse/providers/registry/glm.js @@ -49,6 +49,7 @@ export default { { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" }, { id: "glm-5.2", name: "GLM 5.2" }, { id: "glm-5.1", name: "GLM 5.1" }, + { id: "glm-5-turbo", name: "GLM 5 Turbo" }, { id: "glm-5", name: "GLM 5" }, { id: "glm-4.7", name: "GLM 4.7" }, { id: "glm-4.6v", name: "GLM 4.6V (Vision)" }, diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index b56896b1..8b2d6ead 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -39,10 +39,15 @@ const PATTERN_THINKING = [ { provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS }, { pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking - // codebuddy-cn per-model effort sets — read off the client picker (server- - // delivered supportedEfforts), 2026-08-30. Gateway uses thinkingFormat "openai" - // but rejects levels outside each model's set. + // codebuddy-cn per-model effort sets — the server's product-config payload + // publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint + // accepts any level you send (probed none/minimal/low/medium/high/xhigh/max + // → all 200), but values outside a model's supportedEfforts are silently + // clamped, so the declared set stays authoritative for the picker. Models + // that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x / + // kimi-k3-1 / minimax-m3) fall through to the openai format default. { provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] }, + { provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] }, { provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] }, { provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] }, { provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] }, From 81f4f930821805b58a25e36c49b1656903593f3b Mon Sep 17 00:00:00 2001 From: turingcat Date: Sat, 5 Sep 2026 21:07:51 +0700 Subject: [PATCH 07/78] fix(opencode-go): send stable session header (#3800) - add a dedicated OpenCode Go executor that always sends x-opencode-session - preserve a valid caller-provided native OpenCode session header - translate downstream Agent session IDs into opaque, stable, Agent-scoped IDs - forward the original provider session seed and client tool on both initial and credential-refresh requests --- .../2026-09-04-opencode-go-session-header.md | 261 ++++++++++++++++++ ...09-04-opencode-go-session-header-design.md | 114 ++++++++ open-sse/executors/index.js | 3 + open-sse/executors/opencode-go.js | 71 +++++ open-sse/handlers/chatCore.js | 24 +- tests/unit/opencode-go-session.test.js | 166 +++++++++++ 6 files changed, 637 insertions(+), 2 deletions(-) create mode 100644 docs/superpowers/plans/2026-09-04-opencode-go-session-header.md create mode 100644 docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md create mode 100644 open-sse/executors/opencode-go.js create mode 100644 tests/unit/opencode-go-session.test.js diff --git a/docs/superpowers/plans/2026-09-04-opencode-go-session-header.md b/docs/superpowers/plans/2026-09-04-opencode-go-session-header.md new file mode 100644 index 00000000..546de027 --- /dev/null +++ b/docs/superpowers/plans/2026-09-04-opencode-go-session-header.md @@ -0,0 +1,261 @@ +# OpenCode Go Session Header Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Send a stable, conversation-scoped `x-opencode-session` header on every OpenCode Go request and install the patched CLI locally. + +**Architecture:** Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. `chatCore` passes the provider-scoped session resolved from the original request plus the detected client tool; the executor derives a request-local upstream session and delegates all existing transport, authentication, retry, and proxy behavior to `DefaultExecutor`. + +**Tech Stack:** Node.js ESM, Vitest, Next.js, npm CLI packaging, GitHub CLI. + +## Global Constraints + +- Apply the header to OpenCode Go chat completions, Claude Messages, and OpenAI Responses transports. +- Preserve a valid native `x-opencode-session`; hash all translated non-OpenCode identities to `ses_<32 lowercase hex>`. +- Namespace translated identities by detected client tool, using `generic` when unknown. +- Do not keep mutable per-request session state on the executor singleton or mutate the caller's credentials object. +- Do not change OpenCode Go models, routing, reasoning, tool behavior, dependencies, or unrelated providers. +- Reuse upstream issue #3759 instead of creating a duplicate issue. + +--- + +### Task 1: Add Failing OpenCode Go Session Tests + +**Files:** +- Create: `tests/unit/opencode-go-session.test.js` + +**Interfaces:** +- Consumes: `getExecutor(provider)` and `DefaultExecutor.buildHeaders(credentials, stream, url, model)`. +- Produces: the required public behavior for `OpenCodeGoExecutor.prepareRequestCredentials({ body, credentials, providerSessionId, clientTool })` and `OpenCodeGoExecutor.execute(args)`. + +- [ ] **Step 1: Write the failing tests** + +Create a Vitest suite that mocks `proxyAwareFetch`, obtains `getExecutor("opencode-go")`, and asserts: + +```js +const prepared = executor.prepareRequestCredentials({ + body: { messages: [{ role: "user", content: "hello" }] }, + credentials: { apiKey: "test-key", connectionId: "conn-a", rawHeaders: {} }, + providerSessionId: "conversation-a", + clientTool: "claude", +}); + +expect(prepared).not.toBe(credentials); +expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/); +expect(credentials).not.toHaveProperty("_opencodeGoSession"); +``` + +Cover native header preservation, stable values across all three runtime transports, different conversation IDs, different client tools using the same ID, connection fallback, no singleton state, no header on `DefaultExecutor("openai")`, and the final fetch headers returned by `execute()`. + +- [ ] **Step 2: Run the focused test and verify RED** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js +``` + +Expected: FAIL because `getExecutor("opencode-go")` still returns `DefaultExecutor` and `prepareRequestCredentials` does not exist. + +- [ ] **Step 3: Commit the failing test** + +```bash +git add tests/unit/opencode-go-session.test.js +git commit -m "test: cover OpenCode Go session headers" +``` + +### Task 2: Implement the Dedicated Executor + +**Files:** +- Create: `open-sse/executors/opencode-go.js` +- Modify: `open-sse/executors/index.js` + +**Interfaces:** +- Consumes: `DefaultExecutor`, `resolveSessionId()`, request `credentials.rawHeaders`, `providerSessionId`, and `clientTool`. +- Produces: `OpenCodeGoExecutor`, `prepareRequestCredentials()`, and an `execute()` override that delegates with cloned credentials. + +- [ ] **Step 1: Add the minimal executor implementation** + +Implement these rules: + +```js +function translatedSessionId(sessionId, clientTool) { + const digest = crypto + .createHash("sha256") + .update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`) + .digest("hex") + .slice(0, 32); + return `ses_${digest}`; +} +``` + +`prepareRequestCredentials()` must read a case-insensitive native +`x-opencode-session` with the same non-empty, 256-character cap used by the +session manager. Otherwise it uses `providerSessionId` or calls +`resolveSessionId({ headers, body, connectionId, scope: "opencode-go" })`, then +returns `{ ...credentials, _opencodeGoSession: value }`. + +`execute(args)` must call `prepareRequestCredentials(args)` and delegate using +`super.execute({ ...args, credentials: prepared })`. `buildHeaders()` must call +`super.buildHeaders()` and add the prepared session, with a connection-scoped +fallback for direct callers. + +Register `new OpenCodeGoExecutor()` under `"opencode-go"` and export the class. + +- [ ] **Step 2: Run the focused test and verify partial GREEN** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js +``` + +Expected: executor-level tests pass; any chatCore-context assertion remains failing until Task 3. + +- [ ] **Step 3: Commit the executor** + +```bash +git add open-sse/executors/opencode-go.js open-sse/executors/index.js tests/unit/opencode-go-session.test.js +git commit -m "fix(opencode-go): add stable session header executor" +``` + +### Task 3: Pass Original Request Session Context + +**Files:** +- Modify: `open-sse/handlers/chatCore.js` +- Modify: `tests/unit/opencode-go-session.test.js` + +**Interfaces:** +- Consumes: existing `sessionSeed` and `clientTool` variables in `handleChatCore()`. +- Produces: `providerSessionId` and `clientTool` fields on both initial and refreshed-credential calls to `executor.execute()`. + +- [ ] **Step 1: Add or enable the failing integration assertion** + +Use a mocked executor or source request containing a body-only `session_id` and +assert the executor receives the provider-scoped session resolved before +translation. + +- [ ] **Step 2: Run the focused test and verify RED** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js +``` + +Expected: FAIL because `handleChatCore()` does not pass `providerSessionId` or +`clientTool` to `executor.execute()`. + +- [ ] **Step 3: Pass the request context** + +Add the same fields to both executor calls: + +```js +executor.execute({ + model, + body: translatedBody, + stream, + credentials, + providerSessionId: sessionSeed, + clientTool, + signal: streamController.signal, + log, + proxyOptions, +}); +``` + +- [ ] **Step 4: Run focused and neighboring tests** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js \ + tests/unit/opencode-go-session.test.js \ + tests/unit/opencode-go-models.test.js \ + tests/unit/session-manager.test.js \ + tests/unit/executor-const-guard.test.js +``` + +Expected: PASS with zero failed tests. + +- [ ] **Step 5: Commit the context wiring** + +```bash +git add open-sse/handlers/chatCore.js tests/unit/opencode-go-session.test.js +git commit -m "fix(chat): forward provider session context" +``` + +### Task 4: Verify and Install the Local CLI Package + +**Files:** +- Generated: `9router-0.5.65.tgz` +- Packaged output: `cli/app/server.js` + +**Interfaces:** +- Consumes: completed source changes and existing CLI build scripts. +- Produces: a globally installed patched `9router@0.5.65`. + +- [ ] **Step 1: Run source verification** + +```bash +git diff --check origin/master...HEAD +npx vitest run --config tests/vitest.config.js tests/unit/ +npm run build +``` + +Expected: every command exits zero. Record any pre-existing full-suite failures +separately rather than hiding them. + +- [ ] **Step 2: Build and package the CLI** + +```bash +npm --prefix cli run build +npm --prefix cli pack -- --pack-destination .. +``` + +Expected: `9router-0.5.65.tgz` exists and contains the patched bundled server. + +- [ ] **Step 3: Replace the global npm installation** + +```bash +npm install -g ./9router-0.5.65.tgz +``` + +Expected: `/opt/homebrew/lib/node_modules/9router/package.json` reports `0.5.65` +and the installed bundle contains `x-opencode-session` plus the new executor. + +- [ ] **Step 4: Commit any required package-source adjustment** + +Do not commit generated tarballs or CLI build artifacts unless the repository +already tracks and requires them. + +### Task 5: Publish the Upstream Pull Request + +**Files:** +- No additional source files unless verification finds a required correction. + +**Interfaces:** +- Consumes: verified branch commits and GitHub issue #3759. +- Produces: a fork branch and a PR against `decolua/9router:master`. + +- [ ] **Step 1: Create or repair the GitHub fork remote** + +Use `gh repo fork decolua/9router --remote` if the current `fork` remote remains +missing, then push `fix/opencode-go-session-header`. + +- [ ] **Step 2: Create the PR** + +Use title: + +```text +fix(opencode-go): send stable session header +``` + +The body must include the root cause, downstream-session translation policy, +three covered transports, concurrency behavior, verification evidence, +`Fixes #3759`, and a note that this PR is intentionally narrower than #3780. + +- [ ] **Step 3: Verify the published PR** + +Run `gh pr view --json number,title,state,url,headRefName,baseRefName` and report +the issue and PR URLs. diff --git a/docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md b/docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md new file mode 100644 index 00000000..94daf644 --- /dev/null +++ b/docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md @@ -0,0 +1,114 @@ +# OpenCode Go Session Header Design + +## Problem + +OpenCode Go will begin rejecting some requests without an +`x-opencode-session` header on September 6, 2026. In 9Router v0.5.65, +`opencode-go` uses `DefaultExecutor`, whose generic header builder does not add +that header. The specialized OpenCode Free executor already sends it, but that +logic does not apply to the paid OpenCode Go provider or its three transports. + +## Goals + +- Add `x-opencode-session` to every OpenCode Go chat, Claude Messages, and + OpenAI Responses request. +- Translate a downstream conversation identity into a stable upstream identity. +- Keep identities isolated across different downstream agents and conversations. +- Avoid exposing non-OpenCode downstream session identifiers to OpenCode Go. +- Avoid mutable session state on the shared executor singleton. +- Leave OpenCode Free and all unrelated providers unchanged. + +## Non-Goals + +- Inferring an exact conversation boundary when a downstream client provides no + session or conversation identifier. +- Adding or changing OpenCode Go models, routing, reasoning, or tool behavior. +- Changing the general session-resolution policy for other providers. + +## Architecture + +Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. The executor +keeps the existing generic URL, authentication, translation, retry, and proxy +behavior, and overrides only the OpenCode Go session-header concern. + +`handleChatCore` already resolves a provider-scoped session from the original +request before translation. It will pass that value and the detected client +tool to `executor.execute()` as request context. `OpenCodeGoExecutor.execute()` +will create a shallow request-local credentials object containing the resolved +OpenCode Go session. It will then delegate to `DefaultExecutor.execute()`. +This avoids storing request state on the executor singleton or mutating shared +provider credentials. + +## Session Resolution + +The original downstream request remains the source of truth. Existing +`resolveSessionId()` behavior recognizes Claude Code, Antigravity, generic +session headers, and common body fields before request translation can discard +them. + +Resolution rules: + +1. If the downstream request supplies `x-opencode-session`, treat it as an + authoritative OpenCode identity after trimming and length validation. +2. Otherwise use the provider-scoped session resolved from the original request. +3. Namespace the resolved value with the detected downstream agent, falling back + to `generic` when the agent is unknown. +4. Convert the namespaced value to an opaque deterministic identifier: + `ses_` plus the first 32 hexadecimal characters of SHA-256. +5. If no explicit downstream identity exists, the existing provider connection + fallback guarantees that a header is still sent. It is stable but cannot + distinguish multiple conversations sharing that connection. + +The same input conversation produces the same upstream identifier for all three +OpenCode Go transports. Different agents using the same raw session value +produce different identifiers. + +## Header Injection + +`OpenCodeGoExecutor.buildHeaders()` delegates to +`DefaultExecutor.buildHeaders()` and adds only: + +```text +x-opencode-session: +``` + +The implementation applies to: + +- `https://opencode.ai/zen/go/v1/chat/completions` +- `https://opencode.ai/zen/go/v1/messages` +- `https://opencode.ai/zen/go/v1/responses` + +## Error Handling + +Session derivation must not make requests fail. Invalid or oversized native +header values are ignored and the normal resolved-session fallback is used. +Hashing uses Node's built-in `crypto` module and requires no new dependency. + +## Testing + +Add a focused unit suite that proves: + +- all three OpenCode Go transports receive the header; +- the same conversation remains stable across requests and transports; +- different conversations produce different values; +- different agents using the same raw ID remain isolated; +- non-OpenCode session IDs are represented as opaque `ses_<32 hex>` values; +- a valid native `x-opencode-session` remains stable; +- headerless requests still receive a stable fallback; +- OpenCode Free behavior is unchanged; +- unrelated `DefaultExecutor` providers do not receive the header; +- no request state is retained on the shared executor instance. + +Run the focused unit tests first, then the neighboring executor/session tests, +the full offline test suite, the application build, and the CLI package build. + +## Delivery + +Build the CLI with `npm --prefix cli run build`, create a package with +`npm --prefix cli pack`, and install the generated tarball globally to replace +the current npm-installed `9router@0.5.65`. Verify the installed package version +and packaged source contains the new executor. + +Upstream issue #3759 already tracks the problem, so no duplicate issue will be +created. The pull request will be narrowly scoped to this fix, reference +`Fixes #3759`, and explain how it differs from the broader open PR #3780. diff --git a/open-sse/executors/index.js b/open-sse/executors/index.js index bd96ab49..8dd03421 100644 --- a/open-sse/executors/index.js +++ b/open-sse/executors/index.js @@ -10,6 +10,7 @@ import { CodexExecutor } from "./codex.js"; import { CursorExecutor } from "./cursor.js"; import { VertexExecutor } from "./vertex.js"; import { OpenCodeExecutor } from "./opencode.js"; +import { OpenCodeGoExecutor } from "./opencode-go.js"; import { GrokWebExecutor } from "./grok-web.js"; import { GrokCliExecutor } from "./grok-cli.js"; import { PerplexityWebExecutor } from "./perplexity-web.js"; @@ -40,6 +41,7 @@ const executors = { vertex: new VertexExecutor("vertex"), "vertex-partner": new VertexExecutor("vertex-partner"), opencode: new OpenCodeExecutor(), + "opencode-go": new OpenCodeGoExecutor(), "grok-web": new GrokWebExecutor(), "grok-cli": new GrokCliExecutor(), gcli: new GrokCliExecutor(), // Alias @@ -84,6 +86,7 @@ export { CursorExecutor } from "./cursor.js"; export { VertexExecutor } from "./vertex.js"; export { DefaultExecutor } from "./default.js"; export { OpenCodeExecutor } from "./opencode.js"; +export { OpenCodeGoExecutor } from "./opencode-go.js"; export { GrokWebExecutor } from "./grok-web.js"; export { GrokCliExecutor } from "./grok-cli.js"; export { PerplexityWebExecutor } from "./perplexity-web.js"; diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js new file mode 100644 index 00000000..a4dc4bfa --- /dev/null +++ b/open-sse/executors/opencode-go.js @@ -0,0 +1,71 @@ +import crypto from "node:crypto"; +import { DefaultExecutor } from "./default.js"; +import { resolveSessionId } from "../utils/sessionManager.js"; + +const SESSION_HEADER = "x-opencode-session"; +const SESSION_FIELD = "_opencodeGoSession"; +const MAX_SESSION_LENGTH = 256; + +function normalizeSession(value) { + if (typeof value !== "string") return null; + const normalized = value.trim(); + if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null; + return normalized; +} + +function nativeSession(headers) { + if (!headers || typeof headers !== "object") return null; + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value); + } + return null; +} + +function translatedSession(sessionId, clientTool) { + const digest = crypto + .createHash("sha256") + .update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`) + .digest("hex") + .slice(0, 32); + return `ses_${digest}`; +} + +export class OpenCodeGoExecutor extends DefaultExecutor { + constructor() { + super("opencode-go"); + } + + prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) { + const sourceCredentials = credentials || {}; + const native = nativeSession(sourceCredentials.rawHeaders); + const resolved = normalizeSession(providerSessionId) || resolveSessionId({ + headers: sourceCredentials.rawHeaders, + body, + connectionId: sourceCredentials.connectionId, + scope: "opencode-go", + }); + + return { + ...sourceCredentials, + [SESSION_FIELD]: native || translatedSession(resolved, clientTool), + }; + } + + async execute(args) { + const credentials = this.prepareRequestCredentials(args); + return super.execute({ ...args, credentials }); + } + + buildHeaders(credentials, stream = true, url, model) { + const headers = super.buildHeaders(credentials || {}, stream, url, model); + const prepared = credentials?.[SESSION_FIELD]; + if (prepared) { + headers[SESSION_HEADER] = prepared; + return headers; + } + + const fallback = this.prepareRequestCredentials({ credentials }); + headers[SESSION_HEADER] = fallback[SESSION_FIELD]; + return headers; + } +} diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index 3c083978..d5dcf2d8 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -356,7 +356,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred // exception: it is decoded by the executor into OpenAI-compatible output. let providerResponseFormat = targetFormat; try { - const result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions }); + const result = await executor.execute({ + model, + body: translatedBody, + stream, + credentials, + providerSessionId: sessionSeed, + clientTool, + signal: streamController.signal, + log, + proxyOptions, + }); providerResponse = result.response; providerUrl = result.url; providerHeaders = result.headers; @@ -410,7 +420,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); } } try { - const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions }); + const retryResult = await executor.execute({ + model, + body: translatedBody, + stream, + credentials, + providerSessionId: sessionSeed, + clientTool, + signal: streamController.signal, + log, + proxyOptions, + }); if (retryResult.response.ok) { providerResponse = retryResult.response; providerUrl = retryResult.url; diff --git a/tests/unit/opencode-go-session.test.js b/tests/unit/opencode-go-session.test.js new file mode 100644 index 00000000..2d0ec560 --- /dev/null +++ b/tests/unit/opencode-go-session.test.js @@ -0,0 +1,166 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; + +const { fetchMock } = vi.hoisted(() => ({ + fetchMock: vi.fn(), +})); + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: fetchMock, +})); + +import { DefaultExecutor } from "../../open-sse/executors/default.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; + +const TRANSPORTS = [ + { format: "openai", baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, + { format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } }, + { format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, +]; + +function makeCredentials(overrides = {}) { + return { + apiKey: "test-key", + connectionId: "connection-a", + rawHeaders: {}, + runtimeTransport: TRANSPORTS[0], + ...overrides, + }; +} + +function prepare(executor, overrides = {}) { + const credentials = overrides.credentials || makeCredentials(); + const prepared = executor.prepareRequestCredentials({ + body: overrides.body || { messages: [{ role: "user", content: "hello" }] }, + credentials, + providerSessionId: overrides.providerSessionId ?? "conversation-a", + clientTool: overrides.clientTool ?? "claude", + }); + return { credentials, prepared }; +} + +beforeEach(() => { + fetchMock.mockReset(); + fetchMock.mockResolvedValue(new Response("{}", { + status: 200, + headers: { "content-type": "application/json" }, + })); +}); + +describe("OpenCode Go x-opencode-session", () => { + it("uses a dedicated executor with request-local session credentials", () => { + const executor = getExecutor("opencode-go"); + const { credentials, prepared } = prepare(executor); + + expect(executor.constructor.name).toBe("OpenCodeGoExecutor"); + expect(prepared).not.toBe(credentials); + expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/); + expect(credentials).not.toHaveProperty("_opencodeGoSession"); + expect(executor).not.toHaveProperty("_currentSessionId"); + expect(executor).not.toHaveProperty("_opencodeGoSession"); + }); + + it("preserves a valid native session header case-insensitively", () => { + const executor = getExecutor("opencode-go"); + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "X-OpenCode-Session": " native-session-a " } }), + }); + + expect(prepared._opencodeGoSession).toBe("native-session-a"); + }); + + it("ignores an oversized native session and uses the translated identity", () => { + const executor = getExecutor("opencode-go"); + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "x-opencode-session": "x".repeat(257) } }), + }); + + expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/); + }); + + it("keeps the same translated conversation stable across all transports", () => { + const executor = getExecutor("opencode-go"); + const values = TRANSPORTS.map((runtimeTransport) => { + const { prepared } = prepare(executor, { + credentials: makeCredentials({ runtimeTransport }), + }); + return executor.buildHeaders(prepared, true)["x-opencode-session"]; + }); + + expect(new Set(values).size).toBe(1); + expect(values[0]).toMatch(/^ses_[0-9a-f]{32}$/); + expect(values[0]).not.toContain("conversation-a"); + }); + + it("isolates different conversations", () => { + const executor = getExecutor("opencode-go"); + const a = prepare(executor, { providerSessionId: "conversation-a" }).prepared._opencodeGoSession; + const b = prepare(executor, { providerSessionId: "conversation-b" }).prepared._opencodeGoSession; + + expect(a).not.toBe(b); + }); + + it("isolates different downstream agents that reuse the same raw id", () => { + const executor = getExecutor("opencode-go"); + const claude = prepare(executor, { clientTool: "claude" }).prepared._opencodeGoSession; + const codex = prepare(executor, { clientTool: "codex" }).prepared._opencodeGoSession; + + expect(claude).not.toBe(codex); + }); + + it("uses a stable opaque connection fallback when no session is supplied", () => { + const executor = getExecutor("opencode-go"); + const options = { + credentials: makeCredentials({ connectionId: "fallback-connection" }), + providerSessionId: null, + clientTool: null, + body: { messages: [{ role: "user", content: "headerless" }] }, + }; + const first = prepare(executor, options).prepared._opencodeGoSession; + const second = prepare(executor, options).prepared._opencodeGoSession; + + expect(first).toBe(second); + expect(first).toMatch(/^ses_[0-9a-f]{32}$/); + expect(first).not.toContain("fallback-connection"); + }); + + it("adds the prepared session to the actual fetch headers", async () => { + const executor = getExecutor("opencode-go"); + const credentials = makeCredentials(); + const result = await executor.execute({ + model: "glm-5.2", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: false, + credentials, + providerSessionId: "conversation-fetch", + clientTool: "codex", + }); + + expect(result.headers["x-opencode-session"]).toMatch(/^ses_[0-9a-f]{32}$/); + expect(fetchMock).toHaveBeenCalledOnce(); + expect(fetchMock.mock.calls[0][1].headers["x-opencode-session"]).toBe(result.headers["x-opencode-session"]); + expect(credentials).not.toHaveProperty("_opencodeGoSession"); + }); + + it("does not add the header to unrelated default executors", () => { + const headers = new DefaultExecutor("openai").buildHeaders({ apiKey: "test-key" }, false); + expect(headers["x-opencode-session"]).toBeUndefined(); + }); +}); + +describe("chatCore provider session forwarding", () => { + it("passes the original provider session and client tool on initial and retry execution", () => { + const source = readFileSync( + fileURLToPath(new URL("../../open-sse/handlers/chatCore.js", import.meta.url)), + "utf8", + ); + const calls = [...source.matchAll(/executor\.execute\(\{([\s\S]*?)\}\)/g)].map((match) => match[1]); + + expect(calls).toHaveLength(2); + for (const call of calls) { + expect(call).toMatch(/providerSessionId:\s*sessionSeed/); + expect(call).toMatch(/\bclientTool\b/); + } + }); +}); From 0da803eef42a837838d089f9775a5129457960c0 Mon Sep 17 00:00:00 2001 From: JOJO Date: Sat, 5 Sep 2026 21:07:58 +0700 Subject: [PATCH 08/78] fix(usage): track OpenCode Go quota (#3791) OpenCode Go API-key connections now appear in the Quota Tracker and report rolling, weekly, and monthly subscription usage. --- open-sse/providers/registry/opencode-go.js | 7 ++ open-sse/services/usage.js | 2 + open-sse/services/usage/opencode-go.js | 107 +++++++++++++++++ tests/unit/opencode-go-usage.test.js | 129 +++++++++++++++++++++ tests/unit/usage-dispatch.test.js | 2 +- 5 files changed, 246 insertions(+), 1 deletion(-) create mode 100644 open-sse/services/usage/opencode-go.js create mode 100644 tests/unit/opencode-go-usage.test.js diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index 0dad175f..06ae7ca1 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -21,6 +21,9 @@ export default { transport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", headers: {}, + usage: { + url: "https://opencode.ai/zen/go/v1/usage", + }, }, // Multi-endpoint: pick the transport matching the client sourceFormat to skip // translation. Guarded per-model by `supportedFormats` (see chatCore) because @@ -48,4 +51,8 @@ export default { { id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, ], + features: { + usage: true, + usageApikey: true, + }, }; diff --git a/open-sse/services/usage.js b/open-sse/services/usage.js index 51f00898..eeb46c3c 100644 --- a/open-sse/services/usage.js +++ b/open-sse/services/usage.js @@ -14,6 +14,7 @@ import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn import { getGrokCliUsage } from "./usage/grok-cli.js"; import { getKimiUsage } from "./usage/kimi.js"; import { getDeepseekUsage } from "./usage/deepseek.js"; +import { getOpenCodeGoUsage } from "./usage/opencode-go.js"; import { getGroqUsage } from "./usage/groq.js"; import { getZedUsage } from "./usage/zed.js"; import { resolveQoderCredentials } from "./qoderModels.js"; @@ -55,6 +56,7 @@ const USAGE_HANDLERS = { "codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions), "grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData), + "opencode-go": (c) => getOpenCodeGoUsage(c.apiKey, c.proxyOptions), deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions), groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions), zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), diff --git a/open-sse/services/usage/opencode-go.js b/open-sse/services/usage/opencode-go.js new file mode 100644 index 00000000..2d352498 --- /dev/null +++ b/open-sse/services/usage/opencode-go.js @@ -0,0 +1,107 @@ +/** + * OpenCode Go usage — GET https://opencode.ai/zen/go/v1/usage + * Auth: Bearer + */ + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { parseResetTime, toFiniteNumber, U } from "./shared.js"; + +const USAGE_URL = U("opencode-go").url; +const QUOTA_NAMES = { + rolling: "Rolling", + weekly: "Weekly", + monthly: "Monthly", +}; + +function parsePercent(value) { + if (typeof value === "number" && Number.isFinite(value)) return value; + if (typeof value === "string" && value.trim()) { + const parsed = Number(value); + if (Number.isFinite(parsed)) return parsed; + } + return null; +} + +export async function getOpenCodeGoUsage(apiKey = null, proxyOptions = null) { + if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) { + return { + message: "OpenCode Go API key not available. Add a key to view usage.", + }; + } + + try { + const response = await proxyAwareFetch( + USAGE_URL, + { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey.trim()}`, + Accept: "application/json", + }, + }, + proxyOptions, + ); + + if (response.status === 401) { + return { + plan: "OpenCode Go", + message: "OpenCode Go authentication failed. Check the API key.", + }; + } + + if (response.status === 403) { + const error = await response.json().catch(() => null); + const subscriptionRequired = error?.error?.type === "EntitlementError"; + return { + plan: "OpenCode Go", + message: subscriptionRequired + ? "OpenCode Go subscription required for this API key." + : "OpenCode Go access forbidden for this API key.", + }; + } + + if (!response.ok) { + return { + plan: "OpenCode Go", + message: `OpenCode Go usage API error (${response.status}).`, + }; + } + + const data = await response.json().catch(() => null); + if (!data?.usage || typeof data.usage !== "object") { + return { + plan: "OpenCode Go", + message: "OpenCode Go usage response did not contain quota data.", + }; + } + + const quotas = {}; + for (const [period, name] of Object.entries(QUOTA_NAMES)) { + const quota = data.usage[period]; + if (!quota || typeof quota !== "object") continue; + const percent = parsePercent(quota.percent); + if (percent === null) continue; + const used = Math.max(0, Math.min(100, toFiniteNumber(percent, 0))); + quotas[name] = { + used, + total: 100, + remaining: 100 - used, + remainingPercentage: 100 - used, + resetAt: parseResetTime(quota.resetsAt), + unlimited: false, + }; + } + + + if (Object.keys(quotas).length === 0) { + return { + plan: "OpenCode Go", + message: "OpenCode Go usage response did not contain valid quota data.", + }; + } + + return { plan: "OpenCode Go", quotas }; + } catch (error) { + return { message: `OpenCode Go error: ${error.message}` }; + } +} diff --git a/tests/unit/opencode-go-usage.test.js b/tests/unit/opencode-go-usage.test.js new file mode 100644 index 00000000..349554a4 --- /dev/null +++ b/tests/unit/opencode-go-usage.test.js @@ -0,0 +1,129 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import { getUsageForProvider } from "../../open-sse/services/usage.js"; +import { + USAGE_APIKEY_PROVIDERS, + USAGE_SUPPORTED_PROVIDERS, +} from "../../src/shared/constants/providers.js"; + +const USAGE_URL = "https://opencode.ai/zen/go/v1/usage"; + +function jsonResponse(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +describe("OpenCode Go registry usage flags", () => { + it("is listed for the API key quota dashboard", () => { + expect(USAGE_SUPPORTED_PROVIDERS).toContain("opencode-go"); + expect(USAGE_APIKEY_PROVIDERS).toContain("opencode-go"); + }); +}); + +describe("getUsageForProvider(opencode-go)", () => { + beforeEach(() => vi.clearAllMocks()); + + it("fetches and normalizes subscription usage", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ + usage: { + rolling: { status: "ok", percent: 13, resetsAt: "2026-09-04T14:28:02.617Z" }, + weekly: { status: "ok", percent: 5, resetsAt: "2026-09-07T00:00:00.617Z" }, + monthly: { status: "ok", percent: 2, resetsAt: "2026-10-02T12:14:24.617Z" }, + }, + }), + ); + + const usage = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + + expect(proxyAwareFetch).toHaveBeenCalledWith( + USAGE_URL, + expect.objectContaining({ + method: "GET", + headers: expect.objectContaining({ Authorization: "Bearer sk-go-test" }), + }), + null, + ); + expect(usage).toEqual({ + plan: "OpenCode Go", + quotas: { + Rolling: { + used: 13, + total: 100, + remaining: 87, + remainingPercentage: 87, + resetAt: "2026-09-04T14:28:02.617Z", + unlimited: false, + }, + Weekly: expect.objectContaining({ used: 5, remainingPercentage: 95 }), + Monthly: expect.objectContaining({ used: 2, remainingPercentage: 98 }), + }, + }); + }); + + it("reports missing and rejected credentials", async () => { + const missing = await getUsageForProvider({ provider: "opencode-go" }); + expect(missing.message).toMatch(/api key/i); + expect(proxyAwareFetch).not.toHaveBeenCalled(); + + proxyAwareFetch.mockResolvedValueOnce(jsonResponse({ error: "unauthorized" }, 401)); + const rejected = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "bad", + }); + expect(rejected.message).toMatch(/authentication failed/i); + }); + + it("distinguishes a missing subscription from invalid credentials", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ error: { type: "EntitlementError" } }, 403), + ); + + const usage = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-without-go", + }); + + expect(usage.message).toMatch(/subscription required/i); + }); + + it("rejects responses without a valid quota percentage", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ usage: { rolling: { status: "ok" }, future: { percent: 10 } } }), + ); + + const usage = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + + expect(usage.quotas).toBeUndefined(); + expect(usage.message).toMatch(/valid quota data/i); + }); + + it("reports upstream and network failures", async () => { + proxyAwareFetch.mockResolvedValueOnce(jsonResponse({ error: "unavailable" }, 500)); + const upstream = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + expect(upstream.message).toContain("500"); + + proxyAwareFetch.mockRejectedValueOnce(new Error("socket closed")); + const network = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + expect(network.message).toContain("socket closed"); + }); +}); diff --git a/tests/unit/usage-dispatch.test.js b/tests/unit/usage-dispatch.test.js index 5b86ee9e..84a5f469 100644 --- a/tests/unit/usage-dispatch.test.js +++ b/tests/unit/usage-dispatch.test.js @@ -16,7 +16,7 @@ const SUPPORTED = [ "github", "gemini-cli", "antigravity", "claude", "codex", "kiro", "qoder", "iflow", "ollama", "glm", "glm-cn", "minimax", "minimax-cn", "vercel-ai-gateway", "grok-cli", "kimi", - "deepseek", "zed", + "deepseek", "opencode-go", "zed", ]; describe("usage dispatch", () => { From 97f3ab97b14b55aca61ff45718e0624f3186a74d Mon Sep 17 00:00:00 2001 From: soroush5 Date: Sat, 5 Sep 2026 21:10:15 +0700 Subject: [PATCH 09/78] fix(security): guard cowork-mcp-tools probe against SSRF (#3783) --- .../api/cli-tools/cowork-mcp-tools/route.js | 10 +++ tests/unit/cowork-mcp-ssrf-guard.test.js | 61 +++++++++++++++++++ 2 files changed, 71 insertions(+) create mode 100644 tests/unit/cowork-mcp-ssrf-guard.test.js diff --git a/src/app/api/cli-tools/cowork-mcp-tools/route.js b/src/app/api/cli-tools/cowork-mcp-tools/route.js index 5cc3d3a1..539dcd70 100644 --- a/src/app/api/cli-tools/cowork-mcp-tools/route.js +++ b/src/app/api/cli-tools/cowork-mcp-tools/route.js @@ -1,6 +1,8 @@ "use server"; import { NextResponse } from "next/server"; +import { assertPublicUrl } from "@/shared/utils/ssrfGuard.js"; +import { isLocalRequest } from "@/dashboardGuard"; const TIMEOUT_MS = 8000; @@ -87,6 +89,14 @@ export async function POST(request) { if (!url || typeof url !== "string") { return NextResponse.json({ error: "url required" }, { status: 400 }); } + // SSRF guard for remote callers; local host keeps self-hosted MCP servers. + if (!isLocalRequest(request)) { + try { + assertPublicUrl(url); + } catch { + return NextResponse.json({ error: "URL not allowed" }, { status: 400 }); + } + } const result = await probeMcp(url); return NextResponse.json(result); } catch (e) { diff --git a/tests/unit/cowork-mcp-ssrf-guard.test.js b/tests/unit/cowork-mcp-ssrf-guard.test.js new file mode 100644 index 00000000..663c13b8 --- /dev/null +++ b/tests/unit/cowork-mcp-ssrf-guard.test.js @@ -0,0 +1,61 @@ +/** + * SSRF guard on POST /api/cli-tools/cowork-mcp-tools (#3782). + * + * Remote callers must not be able to force server-side fetches to + * internal URLs; local-host use (self-hosted MCP servers) keeps working. + */ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("next/server", () => ({ + NextResponse: { + json: (body, init) => + new Response(JSON.stringify(body), { + status: init?.status ?? 200, + headers: { "content-type": "application/json" }, + }), + }, +})); + +const { POST } = await import( + "../../src/app/api/cli-tools/cowork-mcp-tools/route.js" +); + +function remoteRequest(url) { + return new Request("http://gateway.example.com/api/cli-tools/cowork-mcp-tools", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ url }), + }); +} + +describe("cowork-mcp-tools SSRF guard", () => { + beforeEach(() => { + vi.restoreAllMocks(); + }); + + it("rejects loopback URLs from remote callers without fetching", async () => { + const fetchSpy = vi.spyOn(globalThis, "fetch"); + const res = await POST(remoteRequest("http://127.0.0.1:18731/internal-admin")); + expect(res.status).toBe(400); + expect(await res.json()).toEqual({ error: "URL not allowed" }); + expect(fetchSpy).not.toHaveBeenCalled(); + }); + + it("rejects private-network URLs from remote callers", async () => { + for (const url of ["http://10.0.0.5/mcp", "http://192.168.1.1/mcp", "http://localhost:3000/mcp"]) { + const res = await POST(remoteRequest(url)); + expect(res.status, `should reject ${url}`).toBe(400); + } + }); + + it("still requires a url", async () => { + const res = await POST( + new Request("http://gateway.example.com/api/cli-tools/cowork-mcp-tools", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({}), + }) + ); + expect(res.status).toBe(400); + }); +}); From 1a3d44683159acfd420855b728d5b1f49d0de29d Mon Sep 17 00:00:00 2001 From: Raisal P Wardana <68648097+IEatCodeDaily@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:15:26 +0700 Subject: [PATCH 10/78] fix(codex): format reset credit API errors (#3778) --- open-sse/services/usage/codex.js | 9 ++++++++- tests/unit/codex-reset-credits.test.js | 11 +++++++++++ 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/open-sse/services/usage/codex.js b/open-sse/services/usage/codex.js index 64d3cbbc..67d91c4b 100644 --- a/open-sse/services/usage/codex.js +++ b/open-sse/services/usage/codex.js @@ -21,6 +21,13 @@ function toIsoDate(value) { return Number.isFinite(time) ? date.toISOString() : null; } +function errorMessage(value, fallback) { + if (!value) return fallback; + if (typeof value === "string") return value; + if (typeof value.message === "string") return value.message; + return JSON.stringify(value); +} + function getCodexAccountId(providerSpecificData) { return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null; } @@ -162,7 +169,7 @@ export async function getCodexRateLimitResetCredits(accessToken, proxyOptions = } if (!response.ok) { - const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`; + const message = errorMessage(data?.message || data?.error || data?.detail, `Codex reset credits API unavailable (${response.status}).`); throw new Error(message); } diff --git a/tests/unit/codex-reset-credits.test.js b/tests/unit/codex-reset-credits.test.js index c8b4c6fd..43846248 100644 --- a/tests/unit/codex-reset-credits.test.js +++ b/tests/unit/codex-reset-credits.test.js @@ -91,6 +91,17 @@ describe("Codex reset credits", () => { }); }); + it("surfaces structured upstream errors as readable messages", async () => { + mocks.proxyAwareFetch.mockResolvedValue({ + ok: false, + status: 403, + json: async () => ({ error: { message: "Reset credits are unavailable for this account" } }), + }); + + const { getCodexRateLimitResetCredits } = await import("../../open-sse/services/usage/codex.js"); + await expect(getCodexRateLimitResetCredits("token")).rejects.toThrow("Reset credits are unavailable for this account"); + }); + it("GET refreshes OAuth credentials before returning reset credit details", async () => { const connection = { id: "conn_1", From 28cfd9facf3edd528a9d102a45ccc900604410f3 Mon Sep 17 00:00:00 2001 From: vianhanif Date: Sat, 5 Sep 2026 21:18:48 +0700 Subject: [PATCH 11/78] fix(dashboard): dynamic mode label for local/remote detection (#3801) --- src/app/(dashboard)/dashboard/profile/page.js | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/src/app/(dashboard)/dashboard/profile/page.js b/src/app/(dashboard)/dashboard/profile/page.js index 2a70e19c..685d053a 100644 --- a/src/app/(dashboard)/dashboard/profile/page.js +++ b/src/app/(dashboard)/dashboard/profile/page.js @@ -81,6 +81,12 @@ export default function ProfilePage() { const [proxyLoading, setProxyLoading] = useState(false); const [proxyTestLoading, setProxyTestLoading] = useState(false); + const [isRemoteHost, setIsRemoteHost] = useState(false); + useEffect(() => { + if (typeof window !== "undefined") + setIsRemoteHost(!["localhost", "127.0.0.1", "::1"].includes(window.location.hostname)); + }, []); + useEffect(() => { fetch("/api/settings") .then((res) => res.json()) @@ -1639,7 +1645,7 @@ export default function ProfilePage() { {/* App Info */}

{APP_CONFIG.name} v{APP_CONFIG.version}

-

Local Mode - All data stored on your machine

+

{isRemoteHost ? "Remote Mode" : "Local Mode - All data stored on your machine"}

From fb9fab0206aa753e104ae4620e6ddcfb7a2ea5b2 Mon Sep 17 00:00:00 2001 From: Federico Liva Date: Sat, 5 Sep 2026 21:25:28 +0700 Subject: [PATCH 12/78] fix(anthropic-compatible): send Claude beta flags to nodes fronting Anthropic (#3797) --- open-sse/executors/default.js | 13 ++++++- tests/unit/claude-header-forwarding.test.js | 40 +++++++++++++++++++++ 2 files changed, 52 insertions(+), 1 deletion(-) diff --git a/open-sse/executors/default.js b/open-sse/executors/default.js index 92c78e92..00f3e3de 100644 --- a/open-sse/executors/default.js +++ b/open-sse/executors/default.js @@ -154,7 +154,18 @@ export class DefaultExecutor extends BaseExecutor { for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials); applyAuth(headers, desc, credentials); - if (this.provider === "claude" && model) { + // anthropic-compatible-* nodes serving a real Claude model sit in front of + // Anthropic itself (a rotating multi-account proxy, a corporate gateway), + // so the request needs the same beta flags the `claude` provider sends: + // without `context-management-2025-06-27` upstream rejects the + // `context_management` block Claude Code puts in every request with + // "context_management: Extra inputs are not permitted" (HTTP 400), and the + // combo silently falls through to the next model. The model id gates this: + // a node fronting Kimi or GLM answers on its own ids and never matches, so + // gateways that would choke on unknown beta flags are left untouched. + const isClaudeModel = typeof model === "string" && /^claude-/.test(model); + if (model && (this.provider === "claude" + || (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) { headers["Anthropic-Beta"] = selectAnthropicBeta(model); } diff --git a/tests/unit/claude-header-forwarding.test.js b/tests/unit/claude-header-forwarding.test.js index 840a8559..d813f356 100644 --- a/tests/unit/claude-header-forwarding.test.js +++ b/tests/unit/claude-header-forwarding.test.js @@ -187,6 +187,46 @@ describe("DefaultExecutor.buildHeaders() — anthropic-compatible stripping", () headers["Anthropic-Version"] || headers["anthropic-version"]; expect(hasVersion).toBeDefined(); }); + + // A node fronting Anthropic (rotating multi-account proxy, corporate gateway) + // needs the same beta flags the `claude` provider sends. Without + // context-management-2025-06-27 upstream answers HTTP 400 + // "context_management: Extra inputs are not permitted" and the combo falls + // through to the next model without anyone noticing. + it("sends context-management beta for a Claude model on a custom host", () => { + const executor = new DefaultExecutor("anthropic-compatible-custom"); + const headers = executor.buildHeaders( + { + apiKey: "key", + providerSpecificData: { baseUrl: "https://myproxy.example.com/v1" }, + }, + true, + undefined, + "claude-opus-5" + ); + + const betaFlags = (headers["Anthropic-Beta"] || headers["anthropic-beta"] || "") + .split(",").map(s => s.trim()); + expect(betaFlags).toContain("context-management-2025-06-27"); + // The first-party identity flag is still stripped for a non-Anthropic host. + expect(betaFlags).not.toContain("claude-code-20250219"); + }); + + it("gates the beta flags on the model id, not the provider prefix", () => { + const executor = new DefaultExecutor("anthropic-compatible-custom"); + const headers = executor.buildHeaders( + { + apiKey: "key", + providerSpecificData: { baseUrl: "https://myproxy.example.com/v1" }, + }, + true, + undefined, + "kimi-k3" + ); + + const betaVal = headers["Anthropic-Beta"] || headers["anthropic-beta"] || ""; + expect(betaVal).not.toContain("context-management-2025-06-27"); + }); }); // ─── proxyFetch anthropicFetch routing ──────────────────────────────────────── From 1442cc73ce12d4d30f8a8e6a50a00fbdacb90f18 Mon Sep 17 00:00:00 2001 From: Hifzi Date: Sat, 5 Sep 2026 21:27:21 +0700 Subject: [PATCH 13/78] fix(antigravity): prevent Google anti-abuse rate limits on multi-account refresh (#3813) --- open-sse/services/projectId.js | 13 ++++--- src/sse/services/backgroundTokenRefresh.js | 44 +++++++++++++++------- src/sse/services/tokenRefresh.js | 31 ++++++++------- 3 files changed, 57 insertions(+), 31 deletions(-) diff --git a/open-sse/services/projectId.js b/open-sse/services/projectId.js index 84ab5a2b..f582c4af 100644 --- a/open-sse/services/projectId.js +++ b/open-sse/services/projectId.js @@ -203,7 +203,8 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA }; const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS; - const MAX_ATTEMPTS = 5; + const MAX_ATTEMPTS = Number(process.env.ONBOARD_MAX_ATTEMPTS) || 2; + const BASE_RETRY_DELAY_MS = Number(process.env.ONBOARD_RETRY_DELAY_MS) || 12_000; for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) { // Bail out immediately if the connection was removed @@ -241,9 +242,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi throw new Error("onboardUser done but no project_id in response"); } - // Server not done yet – wait and retry + // Server not done yet – wait and retry with jitter + const jitter = Math.floor(Math.random() * 5000); console.log(`[ProjectId] Onboard attempt ${attempt}/${MAX_ATTEMPTS}: not done yet, waiting...`); - await new Promise(resolve => setTimeout(resolve, 2000)); + await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter)); } catch (error) { clearTimeout(timeoutId); @@ -256,9 +258,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi console.warn(`[ProjectId] onboardUser failed after ${MAX_ATTEMPTS} attempts: ${error.message}`); return null; } - // Continue to next attempt instead of throwing (which would skip remaining retries) + // Wait with jitter before retrying + const jitter = Math.floor(Math.random() * 5000); console.warn(`[ProjectId] onboardUser attempt ${attempt} failed: ${error.message}, retrying...`); - await new Promise(resolve => setTimeout(resolve, 2000)); + await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter)); } finally { clearTimeout(timeoutId); externalSignal?.removeEventListener("abort", forwardAbort); diff --git a/src/sse/services/backgroundTokenRefresh.js b/src/sse/services/backgroundTokenRefresh.js index 5810e049..b83bf335 100644 --- a/src/sse/services/backgroundTokenRefresh.js +++ b/src/sse/services/backgroundTokenRefresh.js @@ -9,6 +9,7 @@ import { getCredentialExpiryMs } from "open-sse/services/oauthCredentialManager. export const BACKGROUND_REFRESH_LEAD_MS = 30 * 60 * 1000; const DEFAULT_INTERVAL_MS = 5 * 60 * 1000; const INITIAL_DELAY_MS = 10 * 1000; +const SENSITIVE_PROVIDERS = new Set(["antigravity", "gemini-cli"]); let started = false; let intervalHandle = null; @@ -92,25 +93,42 @@ export async function runBackgroundTokenRefreshTick(deps = {}) { try { const load = deps.loadConnections || loadActiveConnections; const refresh = deps.refreshConnection || refreshOne; + const sleep = deps.sleep || ((ms) => new Promise((res) => setTimeout(res, ms))); const connections = await load(); const due = selectConnectionsNeedingRefresh(connections, Date.now()); if (due.length === 0) return; - await Promise.allSettled( - due.map(async (conn) => { - try { - await refresh(conn); - } catch (err) { - log.warn("BG_TOKEN_REFRESH", "Connection refresh failed (swallowed)", { - id: conn?.id, - provider: conn?.provider, - error: err?.message ?? String(err), - }); - } - }) - ); + const baseSensitiveDelay = Number(process.env.BG_REFRESH_GOOGLE_DELAY_MS) || 12_000; + const baseNormalDelay = Number(process.env.BG_REFRESH_DELAY_MS) || 1_500; + + for (let i = 0; i < due.length; i++) { + const conn = due[i]; + try { + await refresh(conn); + log.info("BG_TOKEN_REFRESH", "Connection refresh finished", { + id: conn.id, + email: conn.email || conn.name || conn.id, + provider: conn.provider, + }); + } catch (err) { + log.warn("BG_TOKEN_REFRESH", "Connection refresh failed (swallowed)", { + id: conn?.id, + email: conn?.email || conn?.name || conn?.id, + provider: conn?.provider, + error: err?.message ?? String(err), + }); + } + + // Sequential delay between accounts to prevent bursting upstream providers (especially Google Cloud) + if (i < due.length - 1) { + const isSensitive = SENSITIVE_PROVIDERS.has(conn.provider); + const baseDelay = isSensitive ? baseSensitiveDelay : baseNormalDelay; + const jitter = isSensitive ? Math.floor(Math.random() * 4000) : 200; + await sleep(baseDelay + jitter); + } + } } catch (err) { log.warn("BG_TOKEN_REFRESH", "Tick failed (swallowed)", { error: err?.message ?? String(err), diff --git a/src/sse/services/tokenRefresh.js b/src/sse/services/tokenRefresh.js index 6c808f07..58a6f870 100644 --- a/src/sse/services/tokenRefresh.js +++ b/src/sse/services/tokenRefresh.js @@ -124,25 +124,30 @@ function needsProjectId(provider) { function _refreshProjectId(provider, connectionId, accessToken) { if (!needsProjectId(provider) || !connectionId || !accessToken) return; - // Evict the stale cached entry so getProjectIdForConnection does a real fetch + // Invalidate the stale cached entry so getProjectIdForConnection does a real fetch invalidateProjectId(connectionId); - getProjectIdForConnection(connectionId, accessToken) - .then((projectId) => { - if (!projectId) return; - updateProviderCredentials(connectionId, { projectId }).catch((err) => { - log.debug("TOKEN_REFRESH", "Failed to persist refreshed projectId", { + // Lazy resolution: Do not eagerly trigger onboardUser during background token refresh. + // Eagerly fetching projectId across multiple accounts simultaneously triggers Google Cloud anti-abuse / rate limits. + // Runtime handlers (e.g. chat handler) will lazily call getProjectIdForConnection() on demand. + if (process.env.EAGER_PROJECT_ID_REFRESH === "true") { + getProjectIdForConnection(connectionId, accessToken, provider) + .then((projectId) => { + if (!projectId) return; + updateProviderCredentials(connectionId, { projectId }).catch((err) => { + log.debug("TOKEN_REFRESH", "Failed to persist refreshed projectId", { + connectionId, + error: err?.message ?? err, + }); + }); + }) + .catch((err) => { + log.debug("TOKEN_REFRESH", "Failed to fetch projectId after token refresh", { connectionId, error: err?.message ?? err, }); }); - }) - .catch((err) => { - log.debug("TOKEN_REFRESH", "Failed to fetch projectId after token refresh", { - connectionId, - error: err?.message ?? err, - }); - }); + } } // ─── Local-specific: persist credentials to localDb ────────────────────────── From 77e6a227fe1ec0688b3bab4ff4bd1de037e7a46d Mon Sep 17 00:00:00 2001 From: Sutarto Jordan Chrisfivo Date: Sat, 5 Sep 2026 21:37:32 +0700 Subject: [PATCH 14/78] fix(claude): normalize adaptive auto effort (#3792) Claude adaptive requests without an explicit effort are normalized to output_config.effort: "high" instead of forwarding the unsupported literal value "auto" which Anthropic rejects with HTTP 400. --- open-sse/translator/concerns/thinkingUnified.js | 2 +- tests/translator/thinking-unified.test.js | 10 ++++++++++ 2 files changed, 11 insertions(+), 1 deletion(-) diff --git a/open-sse/translator/concerns/thinkingUnified.js b/open-sse/translator/concerns/thinkingUnified.js index a3035823..e86dbef7 100644 --- a/open-sse/translator/concerns/thinkingUnified.js +++ b/open-sse/translator/concerns/thinkingUnified.js @@ -246,7 +246,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) { if (canDisable) body.thinking = { type: "adaptive" }; else delete body.thinking; const level = toLevel(eff); - body.output_config = { effort: level === "xhigh" ? "high" : level }; + body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level }; break; } case "claude-budget": { diff --git a/tests/translator/thinking-unified.test.js b/tests/translator/thinking-unified.test.js index 93d8b683..fd8f9d28 100644 --- a/tests/translator/thinking-unified.test.js +++ b/tests/translator/thinking-unified.test.js @@ -82,6 +82,16 @@ describe("applyThinking per provider format", () => { // Sonnet 5). Both fields together are the documented adaptive shape. expect(out.thinking).toEqual({ type: "adaptive" }); }); + it("claude adaptive thinking maps auto effort to a supported level", () => { + const out = apply("claude", "claude-opus-4.7", { thinking: { type: "adaptive" } }, "claude"); + expect(out.output_config).toEqual({ effort: "high" }); + expect(out.thinking).toEqual({ type: "adaptive" }); + }); + it("permanently adaptive Claude maps auto effort without adding a thinking switch", () => { + const out = apply("claude", "claude-fable-5-1", { thinking: { type: "adaptive" } }, "claude"); + expect(out.output_config).toEqual({ effort: "high" }); + expect(out.thinking).toBeUndefined(); + }); it("Fable 5.1 → effort without a redundant thinking switch", () => { const out = apply("claude", "claude-fable-5-1", { reasoning_effort: "high" }, "claude"); expect(out.output_config).toEqual({ effort: "high" }); From e74db4d0a64ab75e736f129d0f6cbd18fd157791 Mon Sep 17 00:00:00 2001 From: Sina Sadeghi Date: Sat, 5 Sep 2026 21:49:00 +0700 Subject: [PATCH 15/78] feat(opencode-go): add muse-spark-1.3-contributor and fix parallel tool calls on Responses paths (#3819) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Add muse-spark-1.3-contributor as responses-only model on OpenCode Go with dedicated executor - Key Responses→chat streaming tool calls by item_id to prevent parallel tool calls merging into index 0 - Standardize tool coercions and call_id clamping in Responses API translation --- open-sse/config/providerModels.js | 2 +- open-sse/executors/opencode-go.js | 108 ++++++++++++ open-sse/providers/registry/opencode-go.js | 3 + open-sse/translator/formats/responsesApi.js | 40 +++++ .../translator/request/openai-responses.js | 55 ++++-- .../translator/response/openai-responses.js | 51 +++++- tests/unit/opencode-go-models.test.js | 1 + .../opencode-go-muse-spark-responses.test.js | 165 ++++++++++++++++++ .../responses-parallel-tool-calls.test.js | 140 +++++++++++++++ 9 files changed, 541 insertions(+), 24 deletions(-) create mode 100644 tests/unit/opencode-go-muse-spark-responses.test.js create mode 100644 tests/unit/responses-parallel-tool-calls.test.js diff --git a/open-sse/config/providerModels.js b/open-sse/config/providerModels.js index 90a6b0b2..065def24 100644 --- a/open-sse/config/providerModels.js +++ b/open-sse/config/providerModels.js @@ -50,7 +50,7 @@ export function findModelName(aliasOrId, modelId) { } export function getModelTargetFormat(aliasOrId, modelId) { - if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode") && isMuseSparkModel(modelId)) { + if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) { return FORMATS.OPENAI_RESPONSES; } const models = PROVIDER_MODELS[aliasOrId]; diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js index a4dc4bfa..e171be62 100644 --- a/open-sse/executors/opencode-go.js +++ b/open-sse/executors/opencode-go.js @@ -1,11 +1,21 @@ import crypto from "node:crypto"; import { DefaultExecutor } from "./default.js"; import { resolveSessionId } from "../utils/sessionManager.js"; +import { isMuseSparkModel } from "../providers/models/helpers.js"; +import { + normalizeResponsesInput, + clampResponsesCallId, + coerceResponsesArguments, + coerceResponsesOutput, +} from "../translator/formats/responsesApi.js"; const SESSION_HEADER = "x-opencode-session"; const SESSION_FIELD = "_opencodeGoSession"; const MAX_SESSION_LENGTH = 256; +const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses"; +const MAX_TOOL_NAME_LEN = 128; + function normalizeSession(value) { if (typeof value !== "string") return null; const normalized = value.trim(); @@ -30,11 +40,80 @@ function translatedSession(sessionId, clientTool) { return `ses_${digest}`; } +// Strip the thinking suffix "model(level)" so checks hit the base id. +function baseModelId(model) { + return String(model || "").replace(/\([^()]+\)\s*$/, "").trim(); +} + +function isResponsesModel(model) { + return isMuseSparkModel(baseModelId(model)); +} + +// Flatten Chat Completions tool declarations into the Responses flat shape and +// drop hosted/nameless tools the /responses endpoint rejects. +function normalizeResponsesTools(body) { + if (!Array.isArray(body.tools)) return; + const validNames = new Set(); + body.tools = body.tools.filter((tool) => { + if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false; + const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null; + const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : ""); + const name = rawName.trim(); + if (!name) return false; + const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : ""); + const parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)) + ? tool.parameters + : (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} }); + for (const k of Object.keys(tool)) delete tool[k]; + tool.type = "function"; + tool.name = name.slice(0, MAX_TOOL_NAME_LEN); + if (description) tool.description = description; + tool.parameters = parameters; + validNames.add(tool.name); + return true; + }); + if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) { + if (body.tool_choice.type === "function") { + const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : ""; + if (!n || !validNames.has(n)) delete body.tool_choice; + } + } +} + +// Last line of defense for native Responses clients (sourceFormat === targetFormat +// skips translation): coerce items in place so malformed tool payloads 400 here +// with a clear shape instead of upstream as InputValidationError. +function sanitizeResponsesItems(body) { + if (!Array.isArray(body.input)) return; + body.input = body.input.filter((item) => { + if (!item || typeof item !== "object" || Array.isArray(item)) return true; + if (item.type === "function_call") { + if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false; + item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN); + item.call_id = clampResponsesCallId(item.call_id); + item.arguments = coerceResponsesArguments(item.arguments); + return true; + } + if (item.type === "function_call_output") { + item.call_id = clampResponsesCallId(item.call_id); + item.output = coerceResponsesOutput(item.output); + return true; + } + return true; + }); +} + export class OpenCodeGoExecutor extends DefaultExecutor { constructor() { super("opencode-go"); } + buildUrl(model, stream, urlIndex = 0, credentials = null) { + // Muse Spark lives on /responses even when a stale runtimeTransport leaks in. + if (isResponsesModel(model)) return RESPONSES_BASE_URL; + return super.buildUrl(model, stream, urlIndex, credentials); + } + prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) { const sourceCredentials = credentials || {}; const native = nativeSession(sourceCredentials.rawHeaders); @@ -68,4 +147,33 @@ export class OpenCodeGoExecutor extends DefaultExecutor { headers[SESSION_HEADER] = fallback[SESSION_FIELD]; return headers; } + + transformRequest(model, body, stream, credentials) { + const out = super.transformRequest(model, body); + if (!isResponsesModel(model || body?.model)) return out; + const normalized = normalizeResponsesInput(out.input); + if (normalized) out.input = normalized; + if (!Array.isArray(out.input) || out.input.length === 0) { + out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }]; + } + // Responses names the output cap max_output_tokens, not max_tokens. + if (out.max_output_tokens === undefined) { + if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens; + else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens; + } + delete out.max_tokens; + delete out.max_completion_tokens; + if (out.reasoning_effort !== undefined && out.reasoning === undefined) { + out.reasoning = { effort: out.reasoning_effort, summary: "auto" }; + } + if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) { + if (!out.reasoning.summary) out.reasoning.summary = "auto"; + } + delete out.reasoning_effort; + out.stream = true; + out.store = false; + normalizeResponsesTools(out); + sanitizeResponsesItems(out); + return out; + } } diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index 06ae7ca1..ac78da97 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -50,6 +50,9 @@ export default { { id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, + // Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces + // chatCore past the sourceFormat-matched transports into translation (see chatCore guard). + { id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, ], features: { usage: true, diff --git a/open-sse/translator/formats/responsesApi.js b/open-sse/translator/formats/responsesApi.js index c41ee470..26b737f1 100644 --- a/open-sse/translator/formats/responsesApi.js +++ b/open-sse/translator/formats/responsesApi.js @@ -23,6 +23,46 @@ export function normalizeResponsesInput(input) { return null; } +// Strict Responses upstreams reject overlong call_ids with InputValidationError (#393). +export const MAX_RESPONSES_CALL_ID_LEN = 64; + +export function clampResponsesCallId(id) { + if (typeof id !== "string" || !id) return `call_${Date.now()}`; + return id.length > MAX_RESPONSES_CALL_ID_LEN ? id.substring(0, MAX_RESPONSES_CALL_ID_LEN) : id; +} + +// Single-stringify: objects → JSON once; valid JSON strings pass through untouched; +// anything else (partial fragments, empty) falls back to "{}" instead of +// double-encoding and tripping upstream InputValidationError. +export function coerceResponsesArguments(value) { + if (value === undefined || value === null || value === "") return "{}"; + if (typeof value !== "string") { + try { + return JSON.stringify(value); + } catch { + return "{}"; + } + } + try { + JSON.parse(value); + return value; + } catch { + return "{}"; + } +} + +// function_call_output.output must be a string — never null/object. +export function coerceResponsesOutput(value) { + if (typeof value === "string") return value; + if (value === undefined || value === null) return ""; + if (Array.isArray(value)) return value.map((c) => c?.text ?? JSON.stringify(c)).join(""); + try { + return JSON.stringify(value); + } catch { + return String(value); + } +} + /** * Convert OpenAI Responses API format to standard chat completions format * Responses API uses: { input: [...], instructions: "..." } diff --git a/open-sse/translator/request/openai-responses.js b/open-sse/translator/request/openai-responses.js index cf5bc529..5c719fd2 100644 --- a/open-sse/translator/request/openai-responses.js +++ b/open-sse/translator/request/openai-responses.js @@ -6,12 +6,15 @@ */ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; -import { normalizeResponsesInput } from "../formats/responsesApi.js"; +import { + normalizeResponsesInput, + clampResponsesCallId, + coerceResponsesArguments, + coerceResponsesOutput, +} from "../formats/responsesApi.js"; import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM } from "../schema/index.js"; -// Responses API enforces max 64 chars on call_id (#393) -const MAX_CALL_ID_LEN = 64; -const clampCallId = (id) => (typeof id === "string" && id.length > MAX_CALL_ID_LEN ? id.substring(0, MAX_CALL_ID_LEN) : id); +const MAX_TOOL_NAME_LEN = 128; /** * Convert OpenAI Responses API request to OpenAI Chat Completions format @@ -249,6 +252,23 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials) return result; } +/** + * Extract plain text from a system/developer message for Responses instructions. + * Array content (text parts) is joined; anything else falls back to "" rather + * than leaking "[object Object]" upstream. + */ +function extractInstructionsText(content) { + if (typeof content === "string") return content; + if (Array.isArray(content)) { + return content.map((c) => { + if (typeof c?.text === "string") return c.text; + if (typeof c?.content === "string") return c.content; + return ""; + }).filter(Boolean).join("\n"); + } + return ""; +} + /** * Ensure object schema always has properties field (required by Codex Responses API) */ @@ -327,7 +347,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) // Use the first instruction-bearing message as instructions. // OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt. if (!hasSystemMessage) { - result.instructions = typeof msg.content === "string" ? msg.content : ""; + result.instructions = extractInstructionsText(msg.content); hasSystemMessage = true; } continue; // Skip instruction messages in input @@ -378,26 +398,24 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) // Convert tool calls if (msg.role === ROLE.ASSISTANT && msg.tool_calls) { for (const tc of msg.tool_calls) { + // Skip nameless calls — strict Responses upstreams reject them (#444) + const name = typeof tc.function?.name === "string" ? tc.function.name.trim() : ""; + if (!name) continue; result.input.push({ type: RESPONSES_ITEM.FUNCTION_CALL, - call_id: clampCallId(tc.id), - name: tc.function?.name || "_unknown", - arguments: tc.function?.arguments || "{}" + call_id: clampResponsesCallId(tc.id), + name: name.slice(0, MAX_TOOL_NAME_LEN), + arguments: coerceResponsesArguments(tc.function?.arguments) }); } } // Convert tool results - output must be a string for Responses API if (msg.role === ROLE.TOOL) { - const output = typeof msg.content === "string" - ? msg.content - : Array.isArray(msg.content) - ? msg.content.map(c => c.text || JSON.stringify(c)).join("") - : JSON.stringify(msg.content); result.input.push({ type: RESPONSES_ITEM.FUNCTION_CALL_OUTPUT, - call_id: clampCallId(msg.tool_call_id), - output + call_id: clampResponsesCallId(msg.tool_call_id), + output: coerceResponsesOutput(msg.content) }); } } @@ -411,16 +429,19 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) if (body.tools && Array.isArray(body.tools)) { result.tools = body.tools.map(tool => { if (tool.type === OPENAI_BLOCK.FUNCTION) { + // Strict upstreams reject nameless/overlong tool declarations + const name = typeof tool.function?.name === "string" ? tool.function.name.trim() : ""; + if (!name) return null; return { type: OPENAI_BLOCK.FUNCTION, - name: tool.function.name, + name: name.slice(0, MAX_TOOL_NAME_LEN), description: String(tool.function.description || ""), parameters: normalizeToolParameters(tool.function.parameters), strict: tool.function.strict }; } return tool; - }); + }).filter(Boolean); } // Pass through other relevant fields diff --git a/open-sse/translator/response/openai-responses.js b/open-sse/translator/response/openai-responses.js index ff55bb4e..bd435f9c 100644 --- a/open-sse/translator/response/openai-responses.js +++ b/open-sse/translator/response/openai-responses.js @@ -446,6 +446,13 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { state.created = Math.floor(Date.now() / 1000); state.toolCallIndex = 0; state.currentToolCallId = null; + // item_id → chat tool_calls index. Deltas carry item_id; keying on it (not + // stream position) keeps parallel calls separate when upstream emits all + // output_item.added events before any done/delta. Lazily created so callers + // that build their own state object (stream.js) need no changes. + state.respToolChatIndex ??= new Map(); + // Indices that already received argument deltas (guards done-with-args). + state.respToolArgsEmitted ??= new Set(); } // Text content delta @@ -464,16 +471,29 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { return null; } - // Function call started (standard function_call or custom_tool_call) + // Function call started (standard function_call or custom_tool_call). + // Index is assigned here (not on done): attributing deltas by stream position + // merges parallel calls into index 0 whenever upstream emits all addeds + // before dones — the client then concatenates N JSON payloads into one + // tool input and fails validation. The server item id is the correlator. if (eventType === "response.output_item.added" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) { const item = data.item; state.currentToolCallId = item.call_id || fallbackToolCallId(); + state.respToolChatIndex ??= new Map(); + const key = item.id || data.item_id || state.currentToolCallId; + let idx; + if (key && state.respToolChatIndex.has(key)) { + idx = state.respToolChatIndex.get(key); // duplicate added (retry) — reuse + } else { + idx = state.toolCallIndex++; + if (key) state.respToolChatIndex.set(key, idx); + } return buildChunk( { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, { tool_calls: [{ - index: state.toolCallIndex, + index: idx, id: state.currentToolCallId, type: OPENAI_BLOCK.FUNCTION, function: { name: item.name || "", arguments: "" } @@ -482,20 +502,39 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { ); } - // Function call arguments delta (standard or custom_tool_call variant) + // Function call arguments delta (standard or custom_tool_call variant). + // Routed by item_id so interleaved parallel fragments stay on their own call. if (eventType === "response.function_call_arguments.delta" || eventType === "response.custom_tool_call_input.delta") { const argsDelta = data.delta || ""; if (!argsDelta) return null; + const known = data.item_id ? state.respToolChatIndex?.get(data.item_id) : undefined; + const idx = known ?? Math.max(0, (state.toolCallIndex || 1) - 1); + state.respToolArgsEmitted ??= new Set(); + state.respToolArgsEmitted.add(idx); return buildChunk( { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, - { tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsDelta } }] } + { tool_calls: [{ index: idx, function: { arguments: argsDelta } }] } ); } - // Function call done (standard or custom_tool_call variant) + // Function call done (standard or custom_tool_call variant). + // Index was assigned at added-time; nothing to advance. Some upstreams send + // complete arguments only here (no deltas) — emit them once in that case. if (eventType === "response.output_item.done" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) { - state.toolCallIndex++; + const key = data.item?.id || data.item_id; + const idx = (key && state.respToolChatIndex?.get(key)) ?? Math.max(0, (state.toolCallIndex || 1) - 1); + const fullArgs = data.item?.arguments; + if (typeof fullArgs === "string" && fullArgs) { + state.respToolArgsEmitted ??= new Set(); + if (!state.respToolArgsEmitted.has(idx)) { + state.respToolArgsEmitted.add(idx); + return buildChunk( + { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, + { tool_calls: [{ index: idx, function: { arguments: fullArgs } }] } + ); + } + } return null; } diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 335897ff..1c5a940e 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -27,6 +27,7 @@ describe("OpenCode Go model catalog", () => { "mimo-v2.5", "mimo-v2.5-pro", "minimax-m3", "minimax-m2.7", "minimax-m2.5", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "muse-spark-1.3-contributor", ]); }); }); diff --git a/tests/unit/opencode-go-muse-spark-responses.test.js b/tests/unit/opencode-go-muse-spark-responses.test.js new file mode 100644 index 00000000..2f696fd2 --- /dev/null +++ b/tests/unit/opencode-go-muse-spark-responses.test.js @@ -0,0 +1,165 @@ +import { describe, expect, it } from "vitest"; +import { PROVIDER_MODELS, getModelTargetFormat, getModelSupportedFormats } from "../../open-sse/config/providerModels.js"; +import { PROVIDERS } from "../../open-sse/config/providers.js"; +import { resolveTransport } from "../../open-sse/services/provider.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { getThinkingLevels } from "../../open-sse/providers/thinkingLevels.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; +import { OpenCodeGoExecutor } from "../../open-sse/executors/opencode-go.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import "../translator/registerAll.js"; +import { translateRequest } from "../../open-sse/translator/index.js"; + +const MODEL = "muse-spark-1.3-contributor"; +const PROVIDER = "opencode-go"; + +// Mirror of chatCore's per-model transport guard +function pickTransport(provider, sourceFormat, alias, model) { + const supported = getModelSupportedFormats(alias, model); + const rt = resolveTransport(provider, sourceFormat); + return supported?.includes(sourceFormat) ? rt : null; +} + +describe("ocg/muse-spark-1.3-contributor catalog", () => { + it("is registered responses-only", () => { + const entry = (PROVIDER_MODELS["opencode-go"] || []).find((m) => m.id === MODEL); + expect(entry).toBeDefined(); + expect(entry.targetFormat).toBe("openai-responses"); + expect(getModelSupportedFormats("opencode-go", MODEL)).toEqual(["openai-responses"]); + expect(getModelTargetFormat("ocg", MODEL)).toBe(FORMATS.OPENAI_RESPONSES); + expect(getModelTargetFormat("opencode-go", MODEL)).toBe(FORMATS.OPENAI_RESPONSES); + }); + + it("never takes the sourceFormat-matched transport (always translates)", () => { + expect(pickTransport(PROVIDER, "openai", "opencode-go", MODEL)).toBeNull(); + expect(pickTransport(PROVIDER, "claude", "opencode-go", MODEL)).toBeNull(); + expect(pickTransport(PROVIDER, "openai-responses", "opencode-go", MODEL)?.baseUrl) + .toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("advertises reasoning via the shared muse-spark pattern", () => { + expect(getCapabilitiesForModel(PROVIDER, MODEL)).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "openai", + }); + expect(getThinkingLevels(PROVIDER, MODEL)).toContain("xhigh"); + }); +}); + +describe("OpenCodeGoExecutor routing + sanitization", () => { + it("is wired for opencode-go and routes muse-spark to /responses", () => { + expect(getExecutor("opencode-go")).toBeInstanceOf(OpenCodeGoExecutor); + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl(MODEL)).toBe("https://opencode.ai/zen/go/v1/responses"); + // Even a stale runtimeTransport must not drag muse-spark onto chat/messages + expect(ex.buildUrl(MODEL, true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions" }, + })).toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("leaves non-muse models on the default/runtime transport", () => { + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl("kimi-k2.6")).toBe("https://opencode.ai/zen/go/v1/chat/completions"); + expect(ex.buildUrl("minimax-m3", true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/messages" }, + })).toBe("https://opencode.ai/zen/go/v1/messages"); + }); + + it("normalizes caps + reasoning and coerces tool items exactly once", () => { + const ex = new OpenCodeGoExecutor(); + const args = { path: "a\"b\nc\\d", emoji: "🚀 ü", nested: { q: "x'y\"z" } }; + const body = { + model: MODEL, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "function_call", call_id: "x".repeat(100), name: "read", arguments: args }, + { type: "function_call", call_id: "bad", name: " ", arguments: "{}" }, + { type: "function_call", call_id: "frag", name: "exec", arguments: "{not json" }, + { type: "function_call_output", call_id: "c1", output: { ok: true, text: "héllo \"w\"" } }, + { type: "function_call_output", call_id: "c2", output: null }, + ], + tools: [ + { type: "function", function: { name: "read", description: "r", parameters: { type: "object", properties: {} } } }, + { type: "function", function: { name: " ", parameters: {} } }, + ], + max_tokens: 2048, + reasoning_effort: "high", + }; + const out = ex.transformRequest(MODEL, body, true, {}); + expect(out.max_output_tokens).toBe(2048); + expect(out.max_tokens).toBeUndefined(); + expect(out.reasoning).toEqual({ effort: "high", summary: "auto" }); + expect(out.stream).toBe(true); + expect(out.store).toBe(false); + // nameless declaration dropped, nameless call dropped + expect(out.tools.map((t) => t.name)).toEqual(["read"]); + const calls = out.input.filter((i) => i.type === "function_call"); + expect(calls.map((c) => c.name)).toEqual(["read", "exec"]); + // overlong id clamped, object args stringified exactly once + expect(calls[0].call_id).toHaveLength(64); + expect(JSON.parse(calls[0].arguments)).toEqual(args); + // invalid fragment coerced, never double-encoded + expect(calls[1].arguments).toBe("{}"); + const outputs = out.input.filter((i) => i.type === "function_call_output"); + expect(JSON.parse(outputs[0].output)).toEqual({ ok: true, text: "héllo \"w\"" }); + expect(outputs[1].output).toBe(""); + }); +}); + +describe("chat/claude clients translate to Responses without breaking tools", () => { + const tricky = { cmd: "echo \"hi\"\nnewline\ttab\\slash", emoji: "🎉 café naïve", nested: { a: [1, "x'y"] } }; + + it("openai chat → responses keeps arguments parseable", () => { + const translated = translateRequest( + FORMATS.OPENAI, + FORMATS.OPENAI_RESPONSES, + MODEL, + { + model: `ocg/${MODEL}`, + messages: [ + { role: "system", content: [{ type: "text", text: "sys one" }, { type: "text", text: "sys two" }] }, + { role: "user", content: "run it" }, + { + role: "assistant", content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "exec", arguments: tricky } }], + }, + { role: "tool", tool_call_id: "call_1", content: tricky }, + ], + tools: [{ type: "function", function: { name: "exec", description: "e", parameters: { type: "object", properties: {} } } }], + }, + true, {}, PROVIDER, + ); + expect(translated.instructions).toBe("sys one\nsys two"); + const fc = translated.input.find((i) => i.type === "function_call"); + expect(JSON.parse(fc.arguments)).toEqual(tricky); + const fco = translated.input.find((i) => i.type === "function_call_output"); + expect(JSON.parse(fco.output)).toEqual(tricky); + }); + + it("claude messages → responses double-hop keeps tool input intact", () => { + const viaOpenAI = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, MODEL, { + system: "be terse", + messages: [ + { role: "user", content: [{ type: "text", text: "go" }] }, + { + role: "assistant", + content: [ + { type: "text", text: "calling" }, + { type: "tool_use", id: "tu_1", name: "exec", input: tricky }, + ], + }, + { + role: "user", + content: [{ type: "tool_result", tool_use_id: "tu_1", content: [{ type: "text", text: JSON.stringify(tricky) }] }], + }, + ], + tools: [{ name: "exec", description: "e", input_schema: { type: "object", properties: {} } }], + }, true, {}, PROVIDER); + const translated = translateRequest(FORMATS.OPENAI, FORMATS.OPENAI_RESPONSES, MODEL, viaOpenAI, true, {}, PROVIDER); + const fc = translated.input.find((i) => i.type === "function_call"); + expect(JSON.parse(fc.arguments)).toEqual(tricky); + const fco = translated.input.find((i) => i.type === "function_call_output"); + expect(JSON.parse(fco.output)).toEqual(tricky); + }); +}); diff --git a/tests/unit/responses-parallel-tool-calls.test.js b/tests/unit/responses-parallel-tool-calls.test.js new file mode 100644 index 00000000..e1bf5860 --- /dev/null +++ b/tests/unit/responses-parallel-tool-calls.test.js @@ -0,0 +1,140 @@ +// Parallel function_calls from a Responses upstream must stay on separate +// chat tool_calls indices. Regression: response/openai-responses.js attributed +// every arguments delta to the positional toolCallIndex (advanced only on +// output_item.done), so all-added-then-deltas ordering concatenated N JSON +// payloads into index 0 and clients failed with InputValidationError. +import { describe, expect, it } from "vitest"; +import "../translator/registerAll.js"; +import { openaiResponsesToOpenAIResponse } from "../../open-sse/translator/response/openai-responses.js"; +import { initState, translateResponse } from "../../open-sse/translator/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; + +const added = (id, call_id, name, type = "function_call") => ({ + type: "response.output_item.added", + item: { id, type, call_id, name, arguments: "" }, +}); +const delta = (item_id, text) => ({ + type: "response.function_call_arguments.delta", + item_id, + delta: text, +}); +const done = (id, call_id, name) => ({ + type: "response.output_item.done", + item: { id, type: "function_call", call_id, name }, +}); + +// Reassemble translated chunks the way an OpenAI client accumulator does. +function accumulate(calls, chunks) { + for (const chunk of chunks) { + if (!chunk) continue; + for (const tc of chunk.choices?.[0]?.delta?.tool_calls || []) { + const slot = (calls[tc.index] ??= { id: null, name: "", args: "" }); + if (tc.id) slot.id = tc.id; + if (tc.function?.name) slot.name = tc.function.name; + if (tc.function?.arguments) slot.args += tc.function.arguments; + } + } + return calls; +} + +function runStream(events) { + const state = {}; + const chunks = []; + for (const ev of events) { + const out = openaiResponsesToOpenAIResponse(ev, state); + if (out) chunks.push(out); + } + const flush = openaiResponsesToOpenAIResponse(null, state); + if (flush) chunks.push(flush); + return { state, chunks }; +} + +const PAYLOADS = [ + '{"file_path":"/docs/PRODUCT.md"}', + '{"file_path":"/docs/ROADMAP.md"}', + '{"file_path":"/docs/openapi.custom.yaml"}', + '{"file_path":"/docs/.gitignore"}', +]; + +function hostileOrdering() { + const events = PAYLOADS.map((_, i) => added(`fc_${i}`, `call_${i}`, "read_file")); + // Interleaved deltas AFTER all addeds — the ordering that used to merge all + // four payloads into index 0. + PAYLOADS.forEach((p, i) => events.push(delta(`fc_${i}`, p.slice(0, 20)), delta(`fc_${i}`, p.slice(20)))); + PAYLOADS.forEach((_, i) => events.push(done(`fc_${i}`, `call_${i}`, "read_file"))); + return events; +} + +describe("responses parallel tool calls keep their own index", () => { + it("all-added-then-deltas ordering yields 4 separately parseable calls", () => { + const { chunks } = runStream(hostileOrdering()); + const calls = accumulate({}, chunks); + expect(Object.keys(calls)).toHaveLength(4); + PAYLOADS.forEach((p, i) => { + expect(calls[i].id).toBe(`call_${i}`); + expect(calls[i].name).toBe("read_file"); + expect(JSON.parse(calls[i].args)).toEqual(JSON.parse(p)); + }); + }); + + it("sequential ordering still yields indices 0,1 in order", () => { + const events = [ + added("fc_0", "call_0", "read_file"), + delta("fc_0", PAYLOADS[0]), + done("fc_0", "call_0", "read_file"), + added("fc_1", "call_1", "read_file"), + delta("fc_1", PAYLOADS[1]), + done("fc_1", "call_1", "read_file"), + ]; + const { chunks } = runStream(events); + const calls = accumulate({}, chunks); + expect(Object.keys(calls)).toEqual(["0", "1"]); + expect(JSON.parse(calls[0].args)).toEqual(JSON.parse(PAYLOADS[0])); + expect(JSON.parse(calls[1].args)).toEqual(JSON.parse(PAYLOADS[1])); + }); + + it("done carrying full arguments (no deltas) emits them once", () => { + const state = {}; + const out1 = openaiResponsesToOpenAIResponse(added("fc_9", "call_9", "read_file"), state); + const out2 = openaiResponsesToOpenAIResponse({ + type: "response.output_item.done", + item: { id: "fc_9", type: "function_call", call_id: "call_9", name: "read_file", arguments: PAYLOADS[0] }, + }, state); + const calls = accumulate({}, [out1, out2]); + expect(JSON.parse(calls[0].args)).toEqual(JSON.parse(PAYLOADS[0])); + }); + + it("deltas without item_id fall back to the most recent call (legacy behavior)", () => { + const events = [ + added("fc_0", "call_0", "read_file"), + { type: "response.function_call_arguments.delta", delta: PAYLOADS[0] }, + done("fc_0", "call_0", "read_file"), + ]; + const { chunks } = runStream(events); + const calls = accumulate({}, chunks); + expect(JSON.parse(calls[0].args)).toEqual(JSON.parse(PAYLOADS[0])); + }); +}); + +describe("responses → claude end-to-end keeps parallel tool_use blocks separate", () => { + it("four read_file calls arrive as four parseable tool_use blocks", () => { + const state = initState(FORMATS.CLAUDE); + const out = []; + for (const ev of hostileOrdering()) { + for (const r of translateResponse(FORMATS.OPENAI_RESPONSES, FORMATS.CLAUDE, ev, state)) out.push(r); + } + for (const r of translateResponse(FORMATS.OPENAI_RESPONSES, FORMATS.CLAUDE, null, state)) out.push(r); + + const starts = out.filter((r) => r?.type === "content_block_start" && r?.content_block?.type === "tool_use"); + expect(starts).toHaveLength(4); + const partials = out.filter((r) => r?.delta?.type === "input_json_delta"); + expect(partials).toHaveLength(4); + const bodies = partials.map((r) => JSON.parse(r.delta.partial_json).file_path).sort(); + expect(bodies).toEqual([ + "/docs/.gitignore", + "/docs/PRODUCT.md", + "/docs/ROADMAP.md", + "/docs/openapi.custom.yaml", + ]); + }); +}); From f615a83cb250f745bac16642a2d86216a84e910b Mon Sep 17 00:00:00 2001 From: decolua Date: Sat, 5 Sep 2026 21:56:07 +0700 Subject: [PATCH 16/78] feat(dashboard): group Antigravity model quotas and trim hidden keys - Group Antigravity Gemini text models into single 'Gemini (Flash / Pro)' quota - Group Claude models into single 'Claude (Sonnet / Opus)' quota - Prune stale or legacy model keys from hidden quota visibility list Co-Authored-By: Claude Code --- .../usage/components/ProviderLimits/index.js | 22 ++++++ .../usage/components/ProviderLimits/utils.js | 79 ++++++++++++++++--- tests/unit/provider-quota-visibility.test.js | 34 ++++++-- 3 files changed, 119 insertions(+), 16 deletions(-) diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js index 683ea53d..568c4808 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js @@ -596,6 +596,17 @@ export default function ProviderLimits() { const providerVisibility = previous[provider] || {}; const hidden = new Set(providerVisibility.hidden || []); hidden.add(key); + if (provider === "antigravity") { + if (key === "gemini") { + for (const k of hidden) { + if (k.startsWith("gemini-") && !k.includes("image")) hidden.delete(k); + } + } else if (key === "claude") { + for (const k of hidden) { + if (k.startsWith("claude-")) hidden.delete(k); + } + } + } const next = { ...previous, [provider]: { @@ -614,6 +625,17 @@ export default function ProviderLimits() { const providerVisibility = previous[provider] || {}; const hidden = new Set(providerVisibility.hidden || []); hidden.delete(key); + if (provider === "antigravity") { + if (key === "gemini") { + for (const k of hidden) { + if (k.startsWith("gemini-") && !k.includes("image")) hidden.delete(k); + } + } else if (key === "claude") { + for (const k of hidden) { + if (k.startsWith("claude-")) hidden.delete(k); + } + } + } const next = { ...previous, [provider]: { diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js index 2a1a9236..96a77678 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js @@ -316,21 +316,33 @@ export function getQuotaVisibilityKey(quota) { return String(quota.modelKey || quota.name || "").trim(); } -function getProviderHiddenQuotaSet(provider, quotaVisibility) { +/** + * Trim hidden quota keys to only those matching currently valid quotas. + * Stale or obsolete model keys are dropped. + */ +export function trimHiddenQuotaKeys(hidden = [], quotas = []) { + if (!Array.isArray(hidden) || hidden.length === 0) return []; + const validKeys = new Set(quotas.map(getQuotaVisibilityKey).filter(Boolean)); + return [...new Set(hidden.map((k) => String(k).trim()).filter((k) => validKeys.has(k)))]; +} + +function getProviderHiddenQuotaSet(provider, quotaVisibility, quotas = []) { const hidden = quotaVisibility?.[provider]?.hidden; - return new Set(Array.isArray(hidden) ? hidden.map(String) : []); + if (!Array.isArray(hidden) || hidden.length === 0) return new Set(); + const trimmed = quotas.length > 0 ? trimHiddenQuotaKeys(hidden, quotas) : hidden; + return new Set(trimmed.map(String)); } export function filterQuotasByVisibility(provider, quotas = [], quotaVisibility = {}) { if (!Array.isArray(quotas) || quotas.length === 0) return []; - const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility); + const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility, quotas); if (hidden.size === 0) return quotas; return quotas.filter((quota) => !hidden.has(getQuotaVisibilityKey(quota))); } export function getHiddenQuotaRows(provider, quotas = [], quotaVisibility = {}) { if (!Array.isArray(quotas) || quotas.length === 0) return []; - const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility); + const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility, quotas); if (hidden.size === 0) return []; return quotas.filter((quota) => hidden.has(getQuotaVisibilityKey(quota))); } @@ -363,10 +375,55 @@ export function parseQuotaData(provider, data) { case "antigravity": if (data.quotas) { - Object.entries(data.quotas).forEach(([modelKey, quota]) => { + const entries = Object.entries(data.quotas); + const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image")); + const claudeModels = entries.filter(([k]) => k.startsWith("claude-")); + const imageModels = entries.filter(([k]) => k.includes("image")); + const otherModels = entries.filter(([k]) => !k.startsWith("gemini-") && !k.startsWith("claude-") && !k.includes("image")); + + if (geminiModels.length > 0) { + const rep = geminiModels.reduce((min, cur) => + (cur[1].remainingPercentage ?? 100) < (min[1].remainingPercentage ?? 100) ? cur : min + )[1]; + normalizedQuotas.push({ + name: "Gemini (Flash / Pro)", + modelKey: "gemini", + used: rep.used || 0, + total: rep.total || 0, + resetAt: rep.resetAt || null, + remainingPercentage: rep.remainingPercentage, + }); + } + + if (claudeModels.length > 0) { + const rep = claudeModels.reduce((min, cur) => + (cur[1].remainingPercentage ?? 100) < (min[1].remainingPercentage ?? 100) ? cur : min + )[1]; + normalizedQuotas.push({ + name: "Claude (Sonnet / Opus)", + modelKey: "claude", + used: rep.used || 0, + total: rep.total || 0, + resetAt: rep.resetAt || null, + remainingPercentage: rep.remainingPercentage, + }); + } + + imageModels.forEach(([modelKey, quota]) => { normalizedQuotas.push({ name: quota.displayName || modelKey, - modelKey: modelKey, // Keep modelKey for sorting + modelKey, + used: quota.used || 0, + total: quota.total || 0, + resetAt: quota.resetAt || null, + remainingPercentage: quota.remainingPercentage, + }); + }); + + otherModels.forEach(([modelKey, quota]) => { + normalizedQuotas.push({ + name: quota.displayName || modelKey, + modelKey, used: quota.used || 0, total: quota.total || 0, resetAt: quota.resetAt || null, @@ -613,9 +670,13 @@ export function parseQuotaData(provider, data) { const orderMap = new Map(modelOrder.map((m, i) => [m.id, i])); normalizedQuotas.sort((a, b) => { - // Use modelKey for antigravity, otherwise use name - const keyA = a.modelKey || a.name; - const keyB = b.modelKey || b.name; + // Use modelKey for antigravity (mapped to family anchor), otherwise use name + let keyA = a.modelKey || a.name; + let keyB = b.modelKey || b.name; + if (keyA === "gemini") keyA = "gemini-3.8-flash-high"; + if (keyA === "claude") keyA = "claude-sonnet-4-6"; + if (keyB === "gemini") keyB = "gemini-3.8-flash-high"; + if (keyB === "claude") keyB = "claude-sonnet-4-6"; const orderA = orderMap.get(keyA) ?? 999; const orderB = orderMap.get(keyB) ?? 999; return orderA - orderB; diff --git a/tests/unit/provider-quota-visibility.test.js b/tests/unit/provider-quota-visibility.test.js index d6616b81..8c69ee62 100644 --- a/tests/unit/provider-quota-visibility.test.js +++ b/tests/unit/provider-quota-visibility.test.js @@ -3,6 +3,7 @@ import { filterQuotasByVisibility, getHiddenQuotaRows, parseQuotaData, + trimHiddenQuotaKeys, } from "@/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; describe("provider quota visibility", () => { @@ -13,22 +14,26 @@ describe("provider quota visibility", () => { used: 200, total: 1000, resetAt: "2026-07-04T00:00:00Z", + remainingPercentage: 80, }, "claude-opus-4-6-thinking": { displayName: "Claude Opus 4.6 (Thinking)", used: 100, total: 1000, resetAt: "2026-07-04T00:00:00Z", + remainingPercentage: 90, }, }, }; - it("keeps Antigravity modelKey so hidden settings use stable quota ids", () => { + it("groups Antigravity model quotas into Gemini and Claude families", () => { const quotas = parseQuotaData("antigravity", data); expect(quotas.map((q) => q.modelKey)).toEqual([ - "gemini-pro-agent", - "claude-opus-4-6-thinking", + "gemini", + "claude", ]); + expect(quotas[0].name).toBe("Gemini (Flash / Pro)"); + expect(quotas[1].name).toBe("Claude (Sonnet / Opus)"); }); it("shows all quotas by default and hides configured provider rows", () => { @@ -36,19 +41,34 @@ describe("provider quota visibility", () => { expect(filterQuotasByVisibility("antigravity", quotas, {})).toHaveLength(2); const visibility = { - antigravity: { hidden: ["claude-opus-4-6-thinking"] }, + antigravity: { hidden: ["claude"] }, }; const visible = filterQuotasByVisibility("antigravity", quotas, visibility); const hidden = getHiddenQuotaRows("antigravity", quotas, visibility); - expect(visible.map((q) => q.modelKey)).toEqual(["gemini-pro-agent"]); - expect(hidden.map((q) => q.modelKey)).toEqual(["claude-opus-4-6-thinking"]); + expect(visible.map((q) => q.modelKey)).toEqual(["gemini"]); + expect(hidden.map((q) => q.modelKey)).toEqual(["claude"]); + }); + + it("trims stale or obsolete model keys", () => { + const quotas = parseQuotaData("antigravity", data); + const trimmed = trimHiddenQuotaKeys(["claude", "stale-model-xyz", "gemini-3.8-flash-low"], quotas); + expect(trimmed).toEqual(["claude"]); + + const visibility = { + antigravity: { hidden: ["claude", "stale-model-xyz"] }, + }; + const visible = filterQuotasByVisibility("antigravity", quotas, visibility); + const hidden = getHiddenQuotaRows("antigravity", quotas, visibility); + + expect(visible.map((q) => q.modelKey)).toEqual(["gemini"]); + expect(hidden.map((q) => q.modelKey)).toEqual(["claude"]); }); it("does not apply one provider hidden list to another provider", () => { const quotas = parseQuotaData("antigravity", data); const visibility = { - codex: { hidden: ["gemini-pro-agent"] }, + codex: { hidden: ["gemini"] }, }; expect(filterQuotasByVisibility("antigravity", quotas, visibility)).toHaveLength(2); }); From e214fb1c30ede189a363afea2cd577a265491752 Mon Sep 17 00:00:00 2001 From: decolua Date: Sat, 5 Sep 2026 22:05:18 +0700 Subject: [PATCH 17/78] feat(usage): add Claude Fable quota tracker support - Recognize Fable weekly windows and normalize to weekly fable (7d) - Fall back to 100% available weekly Fable window when Anthropic payload omits it - Forward remaining percentages and enforce canonical Claude quota order in Quota Tracker Co-Authored-By: Claude Code --- open-sse/services/usage/claude.js | 24 +++++++++++++++++-- .../usage/components/ProviderLimits/utils.js | 14 +++++++++++ 2 files changed, 36 insertions(+), 2 deletions(-) diff --git a/open-sse/services/usage/claude.js b/open-sse/services/usage/claude.js index ce3e01a0..d45731ab 100644 --- a/open-sse/services/usage/claude.js +++ b/open-sse/services/usage/claude.js @@ -102,14 +102,34 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) { quotas["weekly (7d)"] = createQuotaObject(data.seven_day); } - // Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus) + // Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus, seven_day_fable) + const MODEL_DISPLAY_NAMES = { + fable_5_1: "fable", + fable_5: "fable", + }; + for (const [key, value] of Object.entries(data)) { if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) { - const modelName = key.replace("seven_day_", ""); + const rawName = key.replace("seven_day_", ""); + const modelName = MODEL_DISPLAY_NAMES[rawName] || rawName; quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value); + } else if ((key === "fable" || key === "fable_5" || key === "fable_5_1") && hasUtilization(value)) { + quotas["weekly fable (7d)"] = createQuotaObject(value); } } + // Fallback: surface Fable quota row if weekly window exists but Fable was not returned yet + if (!quotas["weekly fable (7d)"] && hasUtilization(data.seven_day)) { + quotas["weekly fable (7d)"] = { + used: 0, + total: 100, + remaining: 100, + remainingPercentage: 100, + resetAt: parseResetTime(data.seven_day.resets_at), + unlimited: false, + }; + } + return { plan: "Claude Code", extraUsage: data.extra_usage ?? null, diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js index 96a77678..86fd30d8 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js @@ -510,6 +510,8 @@ export function parseQuotaData(provider, data) { name, used: quota.used || 0, total: quota.total || 0, + remaining: quota.remaining !== undefined ? quota.remaining : Math.max(0, (quota.total || 100) - (quota.used || 0)), + remainingPercentage: quota.remainingPercentage !== undefined ? quota.remainingPercentage : calculatePercentage(quota.used, quota.total), resetAt: quota.resetAt || null, }); }); @@ -664,6 +666,18 @@ export function parseQuotaData(provider, data) { return []; } + if (provider?.toLowerCase() === "claude") { + const CLAUDE_QUOTA_ORDER = { + "session (5h)": 0, + "weekly (7d)": 1, + "weekly fable (7d)": 2, + "weekly opus (7d)": 3, + "weekly sonnet (7d)": 4, + }; + normalizedQuotas.sort((a, b) => (CLAUDE_QUOTA_ORDER[a.name] ?? 99) - (CLAUDE_QUOTA_ORDER[b.name] ?? 99)); + return normalizedQuotas; + } + // Sort quotas according to PROVIDER_MODELS order const modelOrder = getModelsByProviderId(provider); if (modelOrder.length > 0) { From 11222eff0fb4b944324b8ab588cfc8c682de4538 Mon Sep 17 00:00:00 2001 From: Sina Sadeghi Date: Sat, 5 Sep 2026 22:39:22 +0700 Subject: [PATCH 18/78] feat(opencode-go): muse-spark-1.2 and Responses tool fixes (#3820) - Add muse-spark-1.2-contributor as responses-only model on OpenCode Go - Normalize object tool schemas without properties in OpenCode Go executor - Make fallback Responses call_ids unique across same-millisecond calls - Make Responses output coercion fail-soft for circular and non-stringifiable values --- open-sse/executors/opencode-go.js | 5 +++- open-sse/providers/registry/opencode-go.js | 1 + open-sse/translator/formats/responsesApi.js | 17 +++++++++++-- tests/unit/opencode-go-models.test.js | 11 +++++++- .../opencode-go-muse-spark-responses.test.js | 15 +++++++++++ .../responses-parallel-tool-calls.test.js | 25 +++++++++++++++++++ 6 files changed, 70 insertions(+), 4 deletions(-) diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js index e171be62..efb656ac 100644 --- a/open-sse/executors/opencode-go.js +++ b/open-sse/executors/opencode-go.js @@ -61,9 +61,12 @@ function normalizeResponsesTools(body) { const name = rawName.trim(); if (!name) return false; const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : ""); - const parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)) + let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)) ? tool.parameters : (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} }); + // Mirror the request translator: {type:"object"} without properties is rejected + // by strict Responses backends, so fill in the empty properties map. + if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} }; for (const k of Object.keys(tool)) delete tool[k]; tool.type = "function"; tool.name = name.slice(0, MAX_TOOL_NAME_LEN); diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index ac78da97..c6673223 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -52,6 +52,7 @@ export default { { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, // Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces // chatCore past the sourceFormat-matched transports into translation (see chatCore guard). + { id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, ], features: { diff --git a/open-sse/translator/formats/responsesApi.js b/open-sse/translator/formats/responsesApi.js index 26b737f1..5454f9f3 100644 --- a/open-sse/translator/formats/responsesApi.js +++ b/open-sse/translator/formats/responsesApi.js @@ -26,8 +26,13 @@ export function normalizeResponsesInput(input) { // Strict Responses upstreams reject overlong call_ids with InputValidationError (#393). export const MAX_RESPONSES_CALL_ID_LEN = 64; +// Fallback ids share one Date.now() when a batch of items is sanitized in a tight +// loop — a per-process sequence keeps same-millisecond ids unique so +// function_call ↔ function_call_output correlation never collides. +let responsesCallIdSeq = 0; + export function clampResponsesCallId(id) { - if (typeof id !== "string" || !id) return `call_${Date.now()}`; + if (typeof id !== "string" || !id) return `call_${Date.now()}_${(responsesCallIdSeq += 1)}`; return id.length > MAX_RESPONSES_CALL_ID_LEN ? id.substring(0, MAX_RESPONSES_CALL_ID_LEN) : id; } @@ -55,7 +60,15 @@ export function coerceResponsesArguments(value) { export function coerceResponsesOutput(value) { if (typeof value === "string") return value; if (value === undefined || value === null) return ""; - if (Array.isArray(value)) return value.map((c) => c?.text ?? JSON.stringify(c)).join(""); + if (Array.isArray(value)) { + return value.map((c) => { + try { + return c?.text ?? JSON.stringify(c); + } catch { + return String(c); + } + }).join(""); + } try { return JSON.stringify(value); } catch { diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 1c5a940e..7ece100a 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -27,7 +27,7 @@ describe("OpenCode Go model catalog", () => { "mimo-v2.5", "mimo-v2.5-pro", "minimax-m3", "minimax-m2.7", "minimax-m2.5", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", - "muse-spark-1.3-contributor", + "muse-spark-1.2-contributor", "muse-spark-1.3-contributor", ]); }); }); @@ -90,6 +90,15 @@ describe("OpenCode Go per-model transport guard (chatCore logic)", () => { } }); + it("routes Muse Spark (responses-only) to /responses, never to /messages", () => { + for (const m of ["muse-spark-1.2-contributor", "muse-spark-1.3-contributor"]) { + expect(getModelSupportedFormats("opencode-go", m)).toEqual(["openai-responses"]); + expect(pickTransport("opencode-go", "openai-responses", "opencode-go", m)?.baseUrl).toBe("https://opencode.ai/zen/go/v1/responses"); + expect(pickTransport("opencode-go", "claude", "opencode-go", m)).toBeNull(); + expect(pickTransport("opencode-go", "openai", "opencode-go", m)).toBeNull(); + } + }); + it("does NOT route MiniMax (no responses support) to /responses", () => { for (const m of CLAUDE_CAPABLE) { expect(pickTransport("opencode-go", "openai-responses", "opencode-go", m)).toBeNull(); diff --git a/tests/unit/opencode-go-muse-spark-responses.test.js b/tests/unit/opencode-go-muse-spark-responses.test.js index 2f696fd2..4f7dcb3d 100644 --- a/tests/unit/opencode-go-muse-spark-responses.test.js +++ b/tests/unit/opencode-go-muse-spark-responses.test.js @@ -105,6 +105,21 @@ describe("OpenCodeGoExecutor routing + sanitization", () => { expect(JSON.parse(outputs[0].output)).toEqual({ ok: true, text: "héllo \"w\"" }); expect(outputs[1].output).toBe(""); }); + + it("fills in properties for object tool schemas missing them", () => { + const ex = new OpenCodeGoExecutor(); + const body = { + model: MODEL, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + tools: [ + { type: "function", function: { name: "bare", parameters: { type: "object" } } }, + { type: "function", function: { name: "full", parameters: { type: "object", properties: { a: { type: "string" } } } } }, + ], + }; + const out = ex.transformRequest(MODEL, body, true, {}); + expect(out.tools.find((t) => t.name === "bare").parameters).toEqual({ type: "object", properties: {} }); + expect(out.tools.find((t) => t.name === "full").parameters).toEqual({ type: "object", properties: { a: { type: "string" } } }); + }); }); describe("chat/claude clients translate to Responses without breaking tools", () => { diff --git a/tests/unit/responses-parallel-tool-calls.test.js b/tests/unit/responses-parallel-tool-calls.test.js index e1bf5860..e33b00c1 100644 --- a/tests/unit/responses-parallel-tool-calls.test.js +++ b/tests/unit/responses-parallel-tool-calls.test.js @@ -6,6 +6,7 @@ import { describe, expect, it } from "vitest"; import "../translator/registerAll.js"; import { openaiResponsesToOpenAIResponse } from "../../open-sse/translator/response/openai-responses.js"; +import { clampResponsesCallId, coerceResponsesOutput, MAX_RESPONSES_CALL_ID_LEN } from "../../open-sse/translator/formats/responsesApi.js"; import { initState, translateResponse } from "../../open-sse/translator/index.js"; import { FORMATS } from "../../open-sse/translator/formats.js"; @@ -138,3 +139,27 @@ describe("responses → claude end-to-end keeps parallel tool_use blocks separat ]); }); }); + +describe("fallback call_ids stay unique within a batch", () => { + it("same-millisecond fallbacks never collide", () => { + const ids = new Set(Array.from({ length: 50 }, () => clampResponsesCallId(undefined))); + expect(ids.size).toBe(50); + for (const id of ids) { + expect(id.startsWith("call_")).toBe(true); + expect(id.length).toBeLessThanOrEqual(MAX_RESPONSES_CALL_ID_LEN); + } + expect(new Set([clampResponsesCallId(""), clampResponsesCallId(null)]).size).toBe(2); + }); +}); + +describe("output coercion stays fail-soft on unstringifiable values", () => { + it("never throws on BigInt/circular array elements", () => { + const circular = {}; + circular.self = circular; + const input = [1n, circular, { text: "ok" }]; + expect(() => coerceResponsesOutput(input)).not.toThrow(); + const out = coerceResponsesOutput(input); + expect(typeof out).toBe("string"); + expect(out).toContain("ok"); + }); +}); From eb712ca821f0ba6bc41043fbd14494c5af5daba5 Mon Sep 17 00:00:00 2001 From: decolua Date: Sat, 5 Sep 2026 22:57:00 +0700 Subject: [PATCH 19/78] # v0.5.69 (2026-09-05) ## Features - **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities - **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`) - **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys - **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819) - **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through - **CLI tools**: replace Copilot MITM with VS Code extension setup guide - **Gemini**: persist and replay `thoughtSignature` scoped by session namespace ## Fixes - **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792) - **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813) - **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797) - **Dashboard**: dynamic mode label for local/remote detection (#3801) - **Codex**: format reset credit API errors cleanly (#3778) - **Security**: guard cowork MCP tools probe against SSRF (#3783) - **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800) - **Logger**: suppress noisy background token refresh logs - **CLI**: export packed `.tgz` directly into workspace root instead of parent directory --- CHANGELOG.md | 22 ++++++++++++++++++++++ cli/package.json | 4 ++-- open-sse/providers/capabilities.js | 4 ++++ open-sse/providers/pricing.js | 1 + open-sse/providers/registry/codex.js | 1 + open-sse/providers/thinkingLevels.js | 1 + package.json | 2 +- tests/unit/capabilities.test.js | 11 +++++++++++ 8 files changed, 43 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index da6b2c05..39b3010c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,25 @@ +# v0.5.69 (2026-09-05) + +## Features +- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities +- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`) +- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys +- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819) +- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through +- **CLI tools**: replace Copilot MITM with VS Code extension setup guide +- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace + +## Fixes +- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792) +- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813) +- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797) +- **Dashboard**: dynamic mode label for local/remote detection (#3801) +- **Codex**: format reset credit API errors cleanly (#3778) +- **Security**: guard cowork MCP tools probe against SSRF (#3783) +- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800) +- **Logger**: suppress noisy background token refresh logs +- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory + # v0.5.65 (2026-09-03) ## Features diff --git a/cli/package.json b/cli/package.json index d5776b4e..db516ac4 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.65", + "version": "0.5.69", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" @@ -16,7 +16,7 @@ "scripts": { "dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js", "build": "node scripts/build-cli.js", - "pack:cli": "npm run build && npm pack --pack-destination ../..", + "pack:cli": "npm run build && npm pack --pack-destination ..", "publish:cli": "npm run build && npm publish", "postinstall": "node hooks/postinstall.js", "prepublishOnly": "npm run build" diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 7e1e757b..a9f8745a 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -154,6 +154,7 @@ export const PROVIDER_CAPABILITIES = { "deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 }, }, "codex": { + "gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, "gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS, @@ -283,6 +284,9 @@ export const PATTERN_CAPABILITIES = [ { pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } }, { pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } }, + // ── OpenAI GPT-6.x (vision + thinking + web search) ────────────── + { pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } }, + // ── OpenAI GPT-5.x (vision + thinking + web search) ────────────── { pattern: "*gpt-5*image*", caps: { imageOutput: true } }, { pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index 7cc302ff..7a7a0019 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -53,6 +53,7 @@ export const MODEL_PRICING = { "gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 }, "gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 }, "gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, + "gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, "o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 }, "o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 }, diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 427c47a4..1909fe97 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -45,6 +45,7 @@ export default { }, }, models: [ + { id: "gpt-6-astra", name: "GPT 6.0 Astra" }, { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, { id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index 8b2d6ead..7cc1b28f 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -35,6 +35,7 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh // Model-name pattern overrides (glob, first match wins) — more precise than format default. const PATTERN_THINKING = [ + { provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS }, { provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS }, diff --git a/package.json b/package.json index b2192e15..80f6d46b 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "9router-app", - "version": "0.5.65", + "version": "0.5.69", "description": "9Router web dashboard", "private": true, "scripts": { diff --git a/tests/unit/capabilities.test.js b/tests/unit/capabilities.test.js index ac544d47..3482a2c8 100644 --- a/tests/unit/capabilities.test.js +++ b/tests/unit/capabilities.test.js @@ -62,4 +62,15 @@ describe("getCapabilitiesForModel", () => { expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna-agentic")).toMatchObject(kiroGpt56Expected); expect(getCapabilitiesForModel("kiro", "gpt-5.6-sol-thinking-agentic")).toMatchObject(kiroGpt56Expected); }); + + it("reports Codex GPT 6.0 Astra as a vision and thinking capable model", () => { + expect(getCapabilitiesForModel("codex", "gpt-6-astra")).toMatchObject({ + vision: true, + reasoning: true, + search: true, + thinkingFormat: "openai", + contextWindow: 272000, + maxOutput: 128000, + }); + }); }); From e7b5f09d506072baec14c45119bc9acd9bec623e Mon Sep 17 00:00:00 2001 From: decolua Date: Wed, 9 Sep 2026 09:45:18 +0700 Subject: [PATCH 20/78] fix(gemini): normalize contents and handle intermediate tool responses in Antigravity Co-Authored-By: Claude Code --- open-sse/executors/antigravity.js | 18 +++++++-------- open-sse/translator/formats/gemini.js | 18 +++++++++++++++ .../translator/request/openai-to-gemini.js | 23 ++++++------------- 3 files changed, 33 insertions(+), 26 deletions(-) diff --git a/open-sse/executors/antigravity.js b/open-sse/executors/antigravity.js index f2ad630a..fb7beee0 100644 --- a/open-sse/executors/antigravity.js +++ b/open-sse/executors/antigravity.js @@ -5,7 +5,7 @@ import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, import { HTTP_STATUS } from "../config/runtimeConfig.js"; import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js"; -import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js"; +import { cleanJSONSchemaForAntigravity, normalizeGeminiContents } from "../translator/formats/gemini.js"; import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js"; import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js"; @@ -193,7 +193,7 @@ export class AntigravityExecutor extends BaseExecutor { // ─── Standard (non-image) request ─── // Fix contents for Claude models via Antigravity - const contents = body.request?.contents?.map(c => { + const rawContents = (body.request?.contents || []).map(c => { let role = c.role; // functionResponse must be role "user" for Claude models if (c.parts?.some(p => p.functionResponse)) { @@ -226,15 +226,13 @@ export class AntigravityExecutor extends BaseExecutor { return p; }); - const partsChanged = parts?.length !== c.parts?.length || modifiedParts?.some((p, idx) => p !== c.parts[idx]); - if (role !== c.role || partsChanged) { - return { - ...c, role, - parts: modifiedParts || parts, - }; - } - return c; + return { + ...c, + role, + parts: modifiedParts || parts || [], + }; }); + const contents = normalizeGeminiContents(rawContents); // Sanitize tool schemas and function names before sending to Antigravity. let tools = body.request?.tools; diff --git a/open-sse/translator/formats/gemini.js b/open-sse/translator/formats/gemini.js index 8a965fee..729e4c10 100644 --- a/open-sse/translator/formats/gemini.js +++ b/open-sse/translator/formats/gemini.js @@ -432,3 +432,21 @@ export function cleanJSONSchemaForAntigravity(schema) { return cleaned; } +// Merge adjacent same-role messages, strip empty parts, ensure initial user turn +export function normalizeGeminiContents(contents) { + const out = []; + for (const c of contents || []) { + if (!c?.role || !Array.isArray(c.parts)) continue; + const parts = c.parts.filter(p => p && Object.keys(p).length > 0); + if (parts.length === 0) continue; + const last = out.at(-1); + if (last?.role === c.role) last.parts.push(...parts); + else out.push({ ...c, parts: [...parts] }); + } + if (out.length > 0 && out[0].role !== "user") { + out.unshift({ role: "user", parts: [{ text: "..." }] }); + } + return out; +} + + diff --git a/open-sse/translator/request/openai-to-gemini.js b/open-sse/translator/request/openai-to-gemini.js index 2b05670d..24e7f262 100644 --- a/open-sse/translator/request/openai-to-gemini.js +++ b/open-sse/translator/request/openai-to-gemini.js @@ -15,7 +15,8 @@ import { generateRequestId, generateSessionId, generateProjectId, - cleanJSONSchemaForAntigravity + cleanJSONSchemaForAntigravity, + normalizeGeminiContents } from "../formats/gemini.js"; import { deriveSessionId, toNumericSessionId } from "../../utils/sessionManager.js"; import { ROLE, GEMINI_ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js"; @@ -35,17 +36,6 @@ function sanitizeGeminiFunctionName(name) { return sanitized.substring(0, 64); } -function normalizeGeminiContents(contents) { - const out = []; - for (const c of contents || []) { - if (!c?.role || !Array.isArray(c.parts) || c.parts.length === 0) continue; - const last = out.at(-1); - if (last?.role === c.role) last.parts.push(...c.parts); - else out.push({ ...c, parts: [...c.parts] }); - } - return out; -} - // Core: Convert OpenAI request to Gemini format (base for all variants) function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) { const result = { @@ -163,12 +153,14 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG } // Check if there are actual tool responses in the next messages - const hasActualResponses = toolCallIds.some(fid => toolResponses[fid]); + const isIntermediate = i < body.messages.length - 1; + const hasActualResponses = toolCallIds.some(fid => toolResponses[fid] !== undefined); - if (hasActualResponses) { + if (hasActualResponses || isIntermediate) { const toolParts = []; for (const fid of toolCallIds) { - if (!toolResponses[fid]) continue; + let resp = toolResponses[fid]; + if (resp === undefined) resp = ""; let name = tcID2Name[fid]; if (!name) { @@ -180,7 +172,6 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG } } - let resp = toolResponses[fid]; let parsedResp = tryParseJSON(resp); if (parsedResp === null) { parsedResp = { result: resp }; From 628ff1eab5cb5996ac52cfb34389eec6aa2fc482 Mon Sep 17 00:00:00 2001 From: decolua Date: Wed, 9 Sep 2026 09:45:23 +0700 Subject: [PATCH 21/78] fix(auth): set 24h maxAge for dashboard session cookie Co-Authored-By: Claude Code --- src/lib/auth/dashboardSession.js | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/lib/auth/dashboardSession.js b/src/lib/auth/dashboardSession.js index b4c65aae..8f81cec5 100644 --- a/src/lib/auth/dashboardSession.js +++ b/src/lib/auth/dashboardSession.js @@ -7,6 +7,7 @@ import { DATA_DIR } from "@/lib/dataDir"; import { getSettings } from "@/lib/localDb"; const DEFAULT_PASSWORD = "123456"; +const SESSION_MAX_AGE_SEC = 24 * 60 * 60; function loadJwtSecret() { if (process.env.JWT_SECRET) return process.env.JWT_SECRET; @@ -64,6 +65,7 @@ export async function setDashboardAuthCookie(cookieStore, request, claims = {}) secure: shouldUseSecureCookie(request), sameSite: "lax", path: "/", + maxAge: SESSION_MAX_AGE_SEC, }); } From e3bf94ee25fc4f472e323bbb961ca5cc3ae56367 Mon Sep 17 00:00:00 2001 From: Christian Gennari Date: Wed, 9 Sep 2026 09:54:40 +0700 Subject: [PATCH 22/78] feat(antigravity): add weekly quota tracking and free-tier handling (#3892) --- open-sse/providers/registry/antigravity.js | 1 + open-sse/services/usage/antigravity-weekly.js | 150 ++++++ open-sse/services/usage/google.js | 62 ++- .../usage/components/ProviderLimits/utils.js | 15 +- .../unit/antigravity-quota-gemini-3.6.test.js | 2 +- .../unit/antigravity-quota-gemini-3.7.test.js | 2 +- .../unit/antigravity-quota-gemini-3.8.test.js | 2 +- tests/unit/antigravity-usage-headers.test.js | 9 +- .../unit/antigravity-weekly-dashboard.test.js | 113 +++++ tests/unit/antigravity-weekly-quota.test.js | 464 ++++++++++++++++++ 10 files changed, 811 insertions(+), 9 deletions(-) create mode 100644 open-sse/services/usage/antigravity-weekly.js create mode 100644 tests/unit/antigravity-weekly-dashboard.test.js create mode 100644 tests/unit/antigravity-weekly-quota.test.js diff --git a/open-sse/providers/registry/antigravity.js b/open-sse/providers/registry/antigravity.js index ab1ee574..eba858e8 100644 --- a/open-sse/providers/registry/antigravity.js +++ b/open-sse/providers/registry/antigravity.js @@ -37,6 +37,7 @@ export default { }, usage: { quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`, + quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`, loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist", tokenUrl: "https://oauth2.googleapis.com/token", }, diff --git a/open-sse/services/usage/antigravity-weekly.js b/open-sse/services/usage/antigravity-weekly.js new file mode 100644 index 00000000..db92a224 --- /dev/null +++ b/open-sse/services/usage/antigravity-weekly.js @@ -0,0 +1,150 @@ +/** + * Antigravity weekly quota — best-effort retrieval from retrieveUserQuotaSummary. + * Failure never breaks existing per-model quota display. + */ + +import { U, parseResetTime, fetchWithTimeout } from "./shared.js"; +import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION } from "../../providers/shared.js"; + +// — Weekly quota summary config —————————————————————————————— +const WEEKLY_CONFIG = { + ...U("antigravity"), + userAgent: ANTIGRAVITY_IDE_USER_AGENT, +}; + +// — Cache: TTL + in-flight dedup per project ——————————————— +const WEEKLY_CACHE_TTL_MS = 180_000; // 3 minutes +const weeklyCache = new Map(); // cacheKey -> { result, expiresAt } | { promise } + +function cacheKey(accessToken, projectId) { + return `${accessToken}::${projectId || ""}`; +} + +// Exported for tests only +export function _clearWeeklyCache() { + weeklyCache.clear(); +} + +// — Group-name to stable key mapping —————————————————————— +const GROUP_MATCHERS = [ + { pattern: /gemini/i, key: "gemini_weekly", displayName: "Gemini (Weekly)" }, + { pattern: /claude|gpt/i, key: "claude_gpt_weekly", displayName: "Claude & GPT (Weekly)" }, +]; + +/** + * Parse a retrieveUserQuotaSummary response into normalized weekly quotas. + * Pure function — safe to unit-test without network. + * + * @param {Object|null} data Raw JSON response + * @returns {Object} e.g. { gemini_weekly: { used, total, ... }, claude_gpt_weekly: { ... } } + */ +export function parseWeeklyQuotaSummary(data) { + if (!data || typeof data !== "object") return {}; + + // Groups may live at data.groups or data.quotaSummary.groups + const groups = Array.isArray(data.groups) + ? data.groups + : Array.isArray(data.quotaSummary?.groups) + ? data.quotaSummary.groups + : null; + + if (!groups) return {}; + + const result = {}; + + for (const group of groups) { + if (!group || typeof group !== "object") continue; + const displayName = group.displayName || ""; + + const buckets = Array.isArray(group.buckets) ? group.buckets : []; + for (const bucket of buckets) { + if (!bucket || typeof bucket !== "object") continue; + + // Identify weekly buckets by checking bucketId + displayName for "weekly" + const bucketText = `${bucket.bucketId || ""} ${bucket.displayName || ""}`.toLowerCase(); + if (!bucketText.includes("weekly")) continue; + + // Skip disabled buckets + if (bucket.disabled === true) continue; + + const remainingFraction = Number(bucket.remainingFraction); + if (!Number.isFinite(remainingFraction)) continue; + + // Match group to a known family + for (const matcher of GROUP_MATCHERS) { + if (matcher.pattern.test(displayName)) { + const total = 1000; + const remaining = Math.round(total * remainingFraction); + const used = Math.max(0, total - remaining); + + result[matcher.key] = { + used, + total, + resetAt: parseResetTime(bucket.resetTime), + remainingPercentage: remainingFraction * 100, + unlimited: false, + displayName: matcher.displayName, + }; + break; // first matching bucket per family wins + } + } + } + } + + return result; +} + +/** + * Fetch weekly quota summary — cached, deduped, never throws. + */ +export async function fetchAntigravityWeeklyQuota(accessToken, projectId, proxyOptions = null) { + const key = cacheKey(accessToken, projectId); + + // Serve in-flight or cached + const hit = weeklyCache.get(key); + if (hit?.promise) return hit.promise; + if (hit && hit.expiresAt > Date.now()) return hit.result; + + const promise = (async () => { + try { + const url = WEEKLY_CONFIG.quotaSummaryApiUrl; + if (!url) return {}; + + const response = await fetchWithTimeout(url, { + method: "POST", + headers: { + "Authorization": `Bearer ${accessToken}`, + "User-Agent": WEEKLY_CONFIG.userAgent, + "Content-Type": "application/json", + "X-Client-Name": "antigravity", + "X-Client-Version": ANTIGRAVITY_IDE_VERSION, + }, + body: JSON.stringify({ + ...(projectId ? { project: projectId } : {}), + }), + }, 10000, proxyOptions); + + if (!response.ok) return {}; + + const data = await response.json(); + return parseWeeklyQuotaSummary(data); + } catch { + return {}; + } + })(); + + weeklyCache.set(key, { promise }); + + try { + const result = await promise; + if (result && Object.keys(result).length > 0) { + weeklyCache.set(key, { result, expiresAt: Date.now() + WEEKLY_CACHE_TTL_MS }); + } else { + weeklyCache.delete(key); + } + return result; + } catch { + weeklyCache.delete(key); + return {}; + } +} diff --git a/open-sse/services/usage/google.js b/open-sse/services/usage/google.js index 9afbe4ca..736722a7 100644 --- a/open-sse/services/usage/google.js +++ b/open-sse/services/usage/google.js @@ -5,6 +5,7 @@ import { CLIENT_METADATA } from "../../config/appConstants.js"; import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION, ANTIGRAVITY_OAUTH_CLIENT } from "../../providers/shared.js"; import { U, parseResetTime, normalizeCloudCodeProjectId, fetchWithTimeout } from "./shared.js"; +import { fetchAntigravityWeeklyQuota } from "./antigravity-weekly.js"; // Antigravity API config (from Quotio) — urls from registry, oauth client + dynamic UA kept here const ANTIGRAVITY_CONFIG = { @@ -157,8 +158,15 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro const data = await response.json(); const quotas = {}; - // Parse model quotas (inspired by vscode-antigravity-cockpit) - if (data.models) { + // Detect tier: free-tier accounts only have weekly quotas (no separate 5h window). + // On free-tier, fetchAvailableModels returns misleading per-model quota info + // (missing remainingFraction defaults to 0, or reflects the weekly limit not a 5h window). + const paidTierId = subscriptionInfo?.paidTier?.id; + const isFreeTier = !paidTierId || paidTierId === "free-tier"; + + // Parse model quotas only for paid-tier accounts. + // Free-tier accounts skip this — their only meaningful quota is the weekly limit. + if (!isFreeTier && data.models) { // Filter only recommended/important models (must match PROVIDER_MODELS ag ids) const importantModels = [ 'gemini-3.8-flash-high', @@ -212,6 +220,56 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro } } + // Best-effort weekly quota overlay — never blocks or breaks per-model results + try { + const weeklyQuotas = await fetchAntigravityWeeklyQuota( + accessToken, + projectId, + proxyOptions + ); + + // Reconcile weekly quota against model family status: + // If every model in a family is locked/exhausted (remainingPercentage === 0) + // until a future reset time, the weekly limit cannot be 100% available. + // On Google's Free Starter tier, retrieveUserQuotaSummary buggily reports + // remainingFraction: 1 even after the starter quota is depleted and all models 429. + const entries = Object.entries(quotas); + const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image")); + const claudeModels = entries.filter(([k]) => k.startsWith("claude-")); + + if (weeklyQuotas.gemini_weekly && geminiModels.length > 0) { + const allGeminiExhausted = geminiModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0); + if (allGeminiExhausted && weeklyQuotas.gemini_weekly.remainingPercentage > 0) { + const maxResetAt = geminiModels.reduce((max, [, q]) => + !max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null + ); + weeklyQuotas.gemini_weekly.used = weeklyQuotas.gemini_weekly.total; + weeklyQuotas.gemini_weekly.remainingPercentage = 0; + if (maxResetAt) { + weeklyQuotas.gemini_weekly.resetAt = maxResetAt; + } + } + } + + if (weeklyQuotas.claude_gpt_weekly && claudeModels.length > 0) { + const allClaudeExhausted = claudeModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0); + if (allClaudeExhausted && weeklyQuotas.claude_gpt_weekly.remainingPercentage > 0) { + const maxResetAt = claudeModels.reduce((max, [, q]) => + !max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null + ); + weeklyQuotas.claude_gpt_weekly.used = weeklyQuotas.claude_gpt_weekly.total; + weeklyQuotas.claude_gpt_weekly.remainingPercentage = 0; + if (maxResetAt) { + weeklyQuotas.claude_gpt_weekly.resetAt = maxResetAt; + } + } + } + + Object.assign(quotas, weeklyQuotas); + } catch { + // Silently ignore — weekly is best-effort + } + return { plan: subscriptionInfo?.currentTier?.name || "Unknown", quotas, diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js index 86fd30d8..ad321433 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js @@ -376,10 +376,12 @@ export function parseQuotaData(provider, data) { case "antigravity": if (data.quotas) { const entries = Object.entries(data.quotas); + const weeklyKeys = new Set(["gemini_weekly", "claude_gpt_weekly"]); const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image")); const claudeModels = entries.filter(([k]) => k.startsWith("claude-")); const imageModels = entries.filter(([k]) => k.includes("image")); - const otherModels = entries.filter(([k]) => !k.startsWith("gemini-") && !k.startsWith("claude-") && !k.includes("image")); + const weeklyModels = entries.filter(([k]) => weeklyKeys.has(k)); + const otherModels = entries.filter(([k]) => !k.startsWith("gemini-") && !k.startsWith("claude-") && !k.includes("image") && !weeklyKeys.has(k)); if (geminiModels.length > 0) { const rep = geminiModels.reduce((min, cur) => @@ -409,6 +411,17 @@ export function parseQuotaData(provider, data) { }); } + weeklyModels.forEach(([modelKey, quota]) => { + normalizedQuotas.push({ + name: quota.displayName || modelKey, + modelKey, + used: quota.used || 0, + total: quota.total || 0, + resetAt: quota.resetAt || null, + remainingPercentage: quota.remainingPercentage, + }); + }); + imageModels.forEach(([modelKey, quota]) => { normalizedQuotas.push({ name: quota.displayName || modelKey, diff --git a/tests/unit/antigravity-quota-gemini-3.6.test.js b/tests/unit/antigravity-quota-gemini-3.6.test.js index c67924d6..042983d1 100644 --- a/tests/unit/antigravity-quota-gemini-3.6.test.js +++ b/tests/unit/antigravity-quota-gemini-3.6.test.js @@ -4,7 +4,7 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } : { models: { "gemini-3.6-flash-high": { diff --git a/tests/unit/antigravity-quota-gemini-3.7.test.js b/tests/unit/antigravity-quota-gemini-3.7.test.js index e172be31..e1cad3a9 100644 --- a/tests/unit/antigravity-quota-gemini-3.7.test.js +++ b/tests/unit/antigravity-quota-gemini-3.7.test.js @@ -4,7 +4,7 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } : { models: { "gemini-3.7-flash-high": { diff --git a/tests/unit/antigravity-quota-gemini-3.8.test.js b/tests/unit/antigravity-quota-gemini-3.8.test.js index cb8637e5..8dd0056c 100644 --- a/tests/unit/antigravity-quota-gemini-3.8.test.js +++ b/tests/unit/antigravity-quota-gemini-3.8.test.js @@ -4,7 +4,7 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } : { models: { "gemini-3.8-flash-high": { diff --git a/tests/unit/antigravity-usage-headers.test.js b/tests/unit/antigravity-usage-headers.test.js index 74cf5287..363f5b6c 100644 --- a/tests/unit/antigravity-usage-headers.test.js +++ b/tests/unit/antigravity-usage-headers.test.js @@ -4,8 +4,10 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } - : { models: {} }, + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } + : url.includes(":retrieveUserQuotaSummary") + ? { groups: [] } + : { models: {} }, text: async () => "{}", })); @@ -21,7 +23,8 @@ describe("Antigravity usage headers", () => { await getAntigravityUsage("access-token", {}); - expect(proxyAwareFetch).toHaveBeenCalledTimes(2); + // loadCodeAssist + fetchAvailableModels + retrieveUserQuotaSummary + expect(proxyAwareFetch).toHaveBeenCalledTimes(3); for (const [, options] of proxyAwareFetch.mock.calls) { expect(options.headers["User-Agent"]).toBe("antigravity/ide/2.11.0 darwin/arm64"); expect(options.headers).not.toHaveProperty("x-request-source"); diff --git a/tests/unit/antigravity-weekly-dashboard.test.js b/tests/unit/antigravity-weekly-dashboard.test.js new file mode 100644 index 00000000..1c4a2784 --- /dev/null +++ b/tests/unit/antigravity-weekly-dashboard.test.js @@ -0,0 +1,113 @@ +import { describe, it, expect } from "vitest"; +import { parseQuotaData } from "@/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; + +describe("Antigravity dashboard normalization with weekly quotas", () => { + const data = { + quotas: { + "gemini-pro-agent": { + displayName: "Gemini 3.1 Pro (High)", + used: 200, + total: 1000, + resetAt: "2026-09-08T00:00:00Z", + remainingPercentage: 80, + }, + "claude-opus-4-6-thinking": { + displayName: "Claude Opus 4.6 (Thinking)", + used: 100, + total: 1000, + resetAt: "2026-09-08T00:00:00Z", + remainingPercentage: 90, + }, + gemini_weekly: { + displayName: "Gemini (Weekly)", + used: 250, + total: 1000, + resetAt: "2026-09-15T00:00:00Z", + remainingPercentage: 75, + }, + claude_gpt_weekly: { + displayName: "Claude & GPT (Weekly)", + used: 500, + total: 1000, + resetAt: "2026-09-14T00:00:00Z", + remainingPercentage: 50, + }, + }, + }; + + it("includes weekly rows with correct display names", () => { + const quotas = parseQuotaData("antigravity", data); + const names = quotas.map((q) => q.name); + + expect(names).toContain("Gemini (Flash / Pro)"); + expect(names).toContain("Claude (Sonnet / Opus)"); + expect(names).toContain("Gemini (Weekly)"); + expect(names).toContain("Claude & GPT (Weekly)"); + }); + + it("uses stable modelKey for weekly rows", () => { + const quotas = parseQuotaData("antigravity", data); + const keys = quotas.map((q) => q.modelKey); + + expect(keys).toContain("gemini_weekly"); + expect(keys).toContain("claude_gpt_weekly"); + }); + + it("weekly rows carry correct quota values", () => { + const quotas = parseQuotaData("antigravity", data); + const geminiWeekly = quotas.find((q) => q.modelKey === "gemini_weekly"); + const claudeWeekly = quotas.find((q) => q.modelKey === "claude_gpt_weekly"); + + expect(geminiWeekly).toMatchObject({ + used: 250, + total: 1000, + remainingPercentage: 75, + resetAt: "2026-09-15T00:00:00Z", + }); + expect(claudeWeekly).toMatchObject({ + used: 500, + total: 1000, + remainingPercentage: 50, + resetAt: "2026-09-14T00:00:00Z", + }); + }); + + it("weekly rows do NOT appear as otherModels", () => { + const quotas = parseQuotaData("antigravity", data); + const weeklyRows = quotas.filter((q) => + q.modelKey === "gemini_weekly" || q.modelKey === "claude_gpt_weekly" + ); + expect(weeklyRows).toHaveLength(2); + expect(weeklyRows[0].name).toMatch(/Weekly/); + expect(weeklyRows[1].name).toMatch(/Weekly/); + }); + + it("order: gemini family, claude family, weekly, then other", () => { + const quotas = parseQuotaData("antigravity", data); + const keys = quotas.map((q) => q.modelKey); + + const geminiIdx = keys.indexOf("gemini"); + const claudeIdx = keys.indexOf("claude"); + const geminiWeeklyIdx = keys.indexOf("gemini_weekly"); + const claudeWeeklyIdx = keys.indexOf("claude_gpt_weekly"); + + expect(geminiIdx).toBeLessThan(geminiWeeklyIdx); + expect(claudeIdx).toBeLessThan(claudeWeeklyIdx); + }); + + it("works with no weekly keys present (backward compat)", () => { + const noWeekly = { + quotas: { + "gemini-pro-agent": { + displayName: "Gemini 3.1 Pro (High)", + used: 200, + total: 1000, + remainingPercentage: 80, + }, + }, + }; + const quotas = parseQuotaData("antigravity", noWeekly); + expect(quotas).toHaveLength(1); + expect(quotas[0].name).toBe("Gemini (Flash / Pro)"); + }); +}); diff --git a/tests/unit/antigravity-weekly-quota.test.js b/tests/unit/antigravity-weekly-quota.test.js new file mode 100644 index 00000000..55cb3a81 --- /dev/null +++ b/tests/unit/antigravity-weekly-quota.test.js @@ -0,0 +1,464 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +// Mock proxyAwareFetch before any imports that use it +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import { + parseWeeklyQuotaSummary, + fetchAntigravityWeeklyQuota, + _clearWeeklyCache, +} from "../../open-sse/services/usage/antigravity-weekly.js"; + +// — Fixtures —————————————————————————————————————————————— +const GEMINI_GROUP = { + displayName: "Gemini Models", + buckets: [ + { + bucketId: "gemini-weekly-bucket", + displayName: "Weekly Limit", + remainingFraction: 0.75, + resetTime: "2026-09-15T00:00:00Z", + }, + { + bucketId: "gemini-daily-bucket", + displayName: "Daily Limit", + remainingFraction: 0.9, + resetTime: "2026-09-09T00:00:00Z", + }, + ], +}; + +const CLAUDE_GPT_GROUP = { + displayName: "Claude and GPT models", + buckets: [ + { + bucketId: "claude-gpt-weekly", + displayName: "Weekly Quota", + remainingFraction: 0.5, + resetTime: "2026-09-14T00:00:00Z", + }, + ], +}; + +const FULL_RESPONSE = { groups: [GEMINI_GROUP, CLAUDE_GPT_GROUP] }; + +const NESTED_RESPONSE = { + quotaSummary: { + groups: [GEMINI_GROUP, CLAUDE_GPT_GROUP], + }, +}; + +// — parseWeeklyQuotaSummary ——————————————————————————————— +describe("parseWeeklyQuotaSummary", () => { + it("extracts Gemini weekly quota from top-level groups", () => { + const result = parseWeeklyQuotaSummary(FULL_RESPONSE); + expect(result.gemini_weekly).toMatchObject({ + used: 250, + total: 1000, + remainingPercentage: 75, + displayName: "Gemini (Weekly)", + unlimited: false, + }); + expect(result.gemini_weekly.resetAt).toBe("2026-09-15T00:00:00.000Z"); + }); + + it("extracts Claude & GPT weekly quota", () => { + const result = parseWeeklyQuotaSummary(FULL_RESPONSE); + expect(result.claude_gpt_weekly).toMatchObject({ + used: 500, + total: 1000, + remainingPercentage: 50, + displayName: "Claude & GPT (Weekly)", + unlimited: false, + }); + expect(result.claude_gpt_weekly.resetAt).toBe("2026-09-14T00:00:00.000Z"); + }); + + it("handles alternate nested quotaSummary.groups shape", () => { + const result = parseWeeklyQuotaSummary(NESTED_RESPONSE); + expect(result.gemini_weekly).toBeDefined(); + expect(result.claude_gpt_weekly).toBeDefined(); + expect(result.gemini_weekly.remainingPercentage).toBe(75); + expect(result.claude_gpt_weekly.remainingPercentage).toBe(50); + }); + + it("skips non-weekly buckets", () => { + const data = { + groups: [{ + displayName: "Gemini Models", + buckets: [ + { + bucketId: "gemini-daily-bucket", + displayName: "Daily Limit", + remainingFraction: 0.9, + resetTime: "2026-09-09T00:00:00Z", + }, + ], + }], + }; + const result = parseWeeklyQuotaSummary(data); + expect(result).toEqual({}); + }); + + it("skips disabled weekly buckets", () => { + const data = { + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "gemini-weekly-bucket", + displayName: "Weekly Limit", + remainingFraction: 0.75, + resetTime: "2026-09-15T00:00:00Z", + disabled: true, + }], + }], + }; + const result = parseWeeklyQuotaSummary(data); + expect(result).toEqual({}); + }); + + it("returns empty object for null/undefined input", () => { + expect(parseWeeklyQuotaSummary(null)).toEqual({}); + expect(parseWeeklyQuotaSummary(undefined)).toEqual({}); + expect(parseWeeklyQuotaSummary("string")).toEqual({}); + }); + + it("returns empty object for response with no groups", () => { + expect(parseWeeklyQuotaSummary({})).toEqual({}); + expect(parseWeeklyQuotaSummary({ groups: "not-array" })).toEqual({}); + expect(parseWeeklyQuotaSummary({ quotaSummary: {} })).toEqual({}); + }); + + it("handles groups with no buckets gracefully", () => { + const data = { + groups: [{ displayName: "Gemini Models" }], + }; + expect(parseWeeklyQuotaSummary(data)).toEqual({}); + }); + + it("handles bucket with non-finite remainingFraction", () => { + const data = { + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "weekly-bucket", + displayName: "Weekly", + remainingFraction: "not-a-number", + }], + }], + }; + expect(parseWeeklyQuotaSummary(data)).toEqual({}); + }); + + it("ignores groups that don't match known families", () => { + const data = { + groups: [{ + displayName: "Unknown AI Provider", + buckets: [{ + bucketId: "weekly-bucket", + displayName: "Weekly", + remainingFraction: 0.5, + }], + }], + }; + expect(parseWeeklyQuotaSummary(data)).toEqual({}); + }); +}); + +// — fetchAntigravityWeeklyQuota ——————————————————————————— +describe("fetchAntigravityWeeklyQuota", () => { + beforeEach(() => { + proxyAwareFetch.mockReset(); + _clearWeeklyCache(); + }); + + it("fetches and returns parsed weekly quota on success", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => FULL_RESPONSE, + }); + + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result.gemini_weekly).toBeDefined(); + expect(result.claude_gpt_weekly).toBeDefined(); + }); + + it("returns {} on HTTP 401", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 401 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on HTTP 403", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 403 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on HTTP 404", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 404 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on HTTP 429", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 429 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on network error", async () => { + proxyAwareFetch.mockRejectedValue(new Error("network timeout")); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on malformed JSON response", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => { throw new SyntaxError("Unexpected token"); }, + }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("deduplicates concurrent requests for the same account", async () => { + let resolveResponse; + proxyAwareFetch.mockReturnValue(new Promise(resolve => { + resolveResponse = resolve; + })); + + const p1 = fetchAntigravityWeeklyQuota("token", "project-1"); + const p2 = fetchAntigravityWeeklyQuota("token", "project-1"); + + resolveResponse({ ok: true, json: async () => FULL_RESPONSE }); + + const [r1, r2] = await Promise.all([p1, p2]); + expect(r1).toEqual(r2); + expect(proxyAwareFetch).toHaveBeenCalledTimes(1); + }); + + it("serves cached result within TTL", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => FULL_RESPONSE, + }); + + await fetchAntigravityWeeklyQuota("token", "project-1"); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + + expect(proxyAwareFetch).toHaveBeenCalledTimes(1); + expect(result.gemini_weekly).toBeDefined(); + }); + + it("sends correct headers and body", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => ({ groups: [] }), + }); + + await fetchAntigravityWeeklyQuota("token", "project-1"); + + expect(proxyAwareFetch).toHaveBeenCalledWith( + "https://daily-cloudcode-pa.googleapis.com/v1internal:retrieveUserQuotaSummary", + expect.objectContaining({ + method: "POST", + headers: expect.objectContaining({ + "Authorization": "Bearer token", + "User-Agent": "antigravity/ide/2.11.0 darwin/arm64", + "Content-Type": "application/json", + "X-Client-Name": "antigravity", + }), + body: JSON.stringify({ project: "project-1" }), + }), + expect.any(Object), + ); + }); +}); + +// — Integration: weekly failure does not affect existing quotas ————— +describe("weekly quota isolation from existing quota", () => { + beforeEach(() => { + proxyAwareFetch.mockReset(); + _clearWeeklyCache(); + }); + + it("existing getAntigravityUsage succeeds even when weekly RPC fails", async () => { + proxyAwareFetch.mockImplementation(async (url) => { + if (url.includes(":loadCodeAssist")) { + return { + ok: true, + status: 200, + json: async () => ({ cloudaicompanionProject: "p1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } }), + }; + } + if (url.includes(":fetchAvailableModels")) { + return { + ok: true, + status: 200, + json: async () => ({ + models: { + "gemini-3.8-flash-high": { + displayName: "Gemini 3.8 Flash (High)", + quotaInfo: { remainingFraction: 0.85, resetTime: "2026-09-15T00:00:00Z" }, + }, + }, + }), + }; + } + if (url.includes(":retrieveUserQuotaSummary")) { + throw new Error("weekly endpoint unavailable"); + } + return { ok: false, status: 404 }; + }); + + const { getAntigravityUsage } = await import("../../open-sse/services/usage/google.js"); + const result = await getAntigravityUsage("token", {}); + + expect(result.quotas["gemini-3.8-flash-high"]).toMatchObject({ + used: 150, + total: 1000, + remainingPercentage: 85, + }); + expect(result.quotas.gemini_weekly).toBeUndefined(); + expect(result.quotas.claude_gpt_weekly).toBeUndefined(); + expect(result.message).toBeUndefined(); + }); + + it("free-tier accounts only show weekly quotas, not per-model short-window quotas", async () => { + proxyAwareFetch.mockImplementation(async (url) => { + if (url.includes(":loadCodeAssist")) { + return { + ok: true, + status: 200, + json: async () => ({ cloudaicompanionProject: "p1", currentTier: { name: "Starter" }, paidTier: { id: "free-tier", name: "Antigravity Starter Quota" } }), + }; + } + if (url.includes(":fetchAvailableModels")) { + return { + ok: true, + status: 200, + json: async () => ({ + models: { + "gemini-3.8-flash-high": { + displayName: "Gemini 3.8 Flash (High)", + quotaInfo: { remainingFraction: 1, resetTime: "2026-09-15T00:00:00Z" }, + }, + "claude-sonnet-4-6": { + displayName: "Claude Sonnet 4.6", + // Missing remainingFraction — free tier exhausted + quotaInfo: { resetTime: "2026-09-13T12:00:00Z" }, + }, + }, + }), + }; + } + if (url.includes(":retrieveUserQuotaSummary")) { + return { + ok: true, + status: 200, + json: async () => ({ + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "gemini-weekly", + displayName: "Weekly Limit Remaining", + remainingFraction: 1, + resetTime: "2026-09-15T00:00:00Z", + }], + }, { + displayName: "Claude and GPT models", + buckets: [{ + bucketId: "3p-weekly", + displayName: "Weekly Limit Remaining", + remainingFraction: 0, + resetTime: "2026-09-13T12:00:00Z", + }], + }], + }), + }; + } + return { ok: false, status: 404 }; + }); + + const { getAntigravityUsage } = await import("../../open-sse/services/usage/google.js"); + const result = await getAntigravityUsage("token", {}); + + // Per-model quotas should be absent (free-tier accounts skip model parsing) + expect(result.quotas["gemini-3.8-flash-high"]).toBeUndefined(); + expect(result.quotas["claude-sonnet-4-6"]).toBeUndefined(); + + // Only weekly quotas should appear + expect(result.quotas.gemini_weekly).toMatchObject({ + used: 0, + total: 1000, + remainingPercentage: 100, + }); + expect(result.quotas.claude_gpt_weekly).toMatchObject({ + used: 1000, + total: 1000, + remainingPercentage: 0, + }); + }); + + it("reconciles weekly quota to 0% when all paid-tier family models are exhausted", async () => { + proxyAwareFetch.mockImplementation(async (url) => { + if (url.includes(":loadCodeAssist")) { + return { + ok: true, + status: 200, + json: async () => ({ cloudaicompanionProject: "p1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } }), + }; + } + if (url.includes(":fetchAvailableModels")) { + return { + ok: true, + status: 200, + json: async () => ({ + models: { + "gemini-3.8-flash-high": { + displayName: "Gemini 3.8 Flash (High)", + // Exhausted model: no remainingFraction, future resetTime + quotaInfo: { resetTime: "2026-09-13T12:00:00Z" }, + }, + }, + }), + }; + } + if (url.includes(":retrieveUserQuotaSummary")) { + return { + ok: true, + status: 200, + json: async () => ({ + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "gemini-weekly", + displayName: "Weekly Limit Remaining", + remainingFraction: 1, + resetTime: "2026-09-15T00:00:00Z", + }], + }], + }), + }; + } + return { ok: false, status: 404 }; + }); + + const { getAntigravityUsage } = await import("../../open-sse/services/usage/google.js"); + const result = await getAntigravityUsage("token", {}); + + // Per-model quota should show exhausted + expect(result.quotas["gemini-3.8-flash-high"].remainingPercentage).toBe(0); + // Weekly quota should be reconciled to 0% with the family reset time + expect(result.quotas.gemini_weekly).toMatchObject({ + used: 1000, + total: 1000, + remainingPercentage: 0, + resetAt: "2026-09-13T12:00:00.000Z", + }); + }); +}); From 4ad1e7a4ba264ebef95a539f0d33867c983a88e4 Mon Sep 17 00:00:00 2001 From: B1nh M1nh <43268322+b1nhm1nh@users.noreply.github.com> Date: Wed, 9 Sep 2026 10:18:35 +0700 Subject: [PATCH 23/78] fix(usage): parse Fable weekly limit from limits[] instead of fabricating a row (#3847) --- open-sse/services/usage/claude.js | 36 ++++++++++++++----------------- 1 file changed, 16 insertions(+), 20 deletions(-) diff --git a/open-sse/services/usage/claude.js b/open-sse/services/usage/claude.js index d45731ab..d93a0682 100644 --- a/open-sse/services/usage/claude.js +++ b/open-sse/services/usage/claude.js @@ -102,32 +102,28 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) { quotas["weekly (7d)"] = createQuotaObject(data.seven_day); } - // Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus, seven_day_fable) - const MODEL_DISPLAY_NAMES = { - fable_5_1: "fable", - fable_5: "fable", - }; - + // Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus) for (const [key, value] of Object.entries(data)) { if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) { - const rawName = key.replace("seven_day_", ""); - const modelName = MODEL_DISPLAY_NAMES[rawName] || rawName; + const modelName = key.replace("seven_day_", ""); quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value); - } else if ((key === "fable" || key === "fable_5" || key === "fable_5_1") && hasUtilization(value)) { - quotas["weekly fable (7d)"] = createQuotaObject(value); } } - // Fallback: surface Fable quota row if weekly window exists but Fable was not returned yet - if (!quotas["weekly fable (7d)"] && hasUtilization(data.seven_day)) { - quotas["weekly fable (7d)"] = { - used: 0, - total: 100, - remaining: 100, - remainingPercentage: 100, - resetAt: parseResetTime(data.seven_day.resets_at), - unlimited: false, - }; + // Model-scoped weekly limits (e.g. Fable) arrive in limits[], not as + // seven_day_* keys: { kind: "weekly_scoped", percent, resets_at, + // scope: { model: { display_name: "Fable" } } }. No limits entry means + // the account has no such window — omit the row, never fabricate one. + if (Array.isArray(data.limits)) { + for (const limit of data.limits) { + if (limit?.kind !== "weekly_scoped") continue; + const modelName = String(limit?.scope?.model?.display_name || "").trim().toLowerCase(); + if (!modelName || typeof limit.percent !== "number") continue; + quotas[`weekly ${modelName} (7d)`] = createQuotaObject({ + utilization: Math.max(0, Math.min(100, limit.percent)), + resets_at: limit.resets_at, + }); + } } return { From 7fee56bacd72c27eb0d7c73568ab9400dc3f253d Mon Sep 17 00:00:00 2001 From: Sutarto Jordan Chrisfivo Date: Wed, 9 Sep 2026 10:24:30 +0700 Subject: [PATCH 24/78] fix(providers): clear stale locks after validation (#3830) Clear stale connection health state (modelLock_*, backoffLevel, rateLimitedUntil, errorCode) whenever a connection is explicitly marked active after successful validation or OAuth re-login. Closes #3810 --- src/lib/db/repos/connectionsRepo.js | 28 +++++++- tests/unit/db-sqlite-vs-lowdb.test.js | 98 +++++++++++++++++++++++++++ 2 files changed, 124 insertions(+), 2 deletions(-) diff --git a/src/lib/db/repos/connectionsRepo.js b/src/lib/db/repos/connectionsRepo.js index 4181843f..78abfc90 100644 --- a/src/lib/db/repos/connectionsRepo.js +++ b/src/lib/db/repos/connectionsRepo.js @@ -10,6 +10,28 @@ const OPTIONAL_FIELDS = [ "consecutiveUseCount", "idToken", "lastRefreshAt", ]; +const MODEL_LOCK_PREFIX = "modelLock_"; + +function resetHealthStateOnActivation(existing, patch) { + if (patch?.testStatus !== "active") return patch; + + const normalized = { + ...patch, + testStatus: "active", + lastError: Object.hasOwn(patch, "lastError") ? patch.lastError : null, + lastErrorAt: Object.hasOwn(patch, "lastErrorAt") ? patch.lastErrorAt : null, + errorCode: null, + rateLimitedUntil: null, + backoffLevel: 0, + }; + + for (const key of Object.keys(existing || {})) { + if (key.startsWith(MODEL_LOCK_PREFIX)) normalized[key] = null; + } + + return normalized; +} + function rowToConn(row) { if (!row) return null; const extra = parseJson(row.data, {}); @@ -147,7 +169,8 @@ export async function createProviderConnection(data) { // access_token: never dedup — user manages duplicates manually if (existing) { - const merged = { ...existing, ...data, updatedAt: now }; + const normalized = resetHealthStateOnActivation(existing, data); + const merged = { ...existing, ...normalized, updatedAt: now }; upsert(db, merged); result = merged; return; @@ -196,7 +219,8 @@ export async function updateProviderConnection(id, data) { const row = db.get(`SELECT * FROM providerConnections WHERE id = ?`, [id]); if (!row) { result = null; return; } const existing = rowToConn(row); - const merged = { ...existing, ...data, updatedAt: new Date().toISOString() }; + const normalized = resetHealthStateOnActivation(existing, data); + const merged = { ...existing, ...normalized, updatedAt: new Date().toISOString() }; upsert(db, merged); if (data.priority !== undefined) reorderInTx(db, existing.provider); result = merged; diff --git a/tests/unit/db-sqlite-vs-lowdb.test.js b/tests/unit/db-sqlite-vs-lowdb.test.js index 52a80884..4955cd88 100644 --- a/tests/unit/db-sqlite-vs-lowdb.test.js +++ b/tests/unit/db-sqlite-vs-lowdb.test.js @@ -101,6 +101,104 @@ describe("DB SQLite layer — public API parity", () => { expect(back.providerSpecificData).toEqual({ foo: "bar" }); }); + it("providerConnections: successful validation clears stale routing locks", async () => { + const c = await sqliteDb.createProviderConnection({ + provider: "health-reset-update", + authType: "oauth", + email: "update@example.com", + accessToken: "old-token", + }); + await sqliteDb.updateProviderConnection(c.id, { + testStatus: "unavailable", + lastError: "Access denied", + lastErrorAt: "2026-09-05T00:00:00.000Z", + errorCode: 403, + backoffLevel: 3, + rateLimitedUntil: "2099-01-01T00:00:00.000Z", + modelLock_modelA: "2099-01-01T00:00:00.000Z", + modelLock_modelB: "2099-01-01T00:00:00.000Z", + }); + + await sqliteDb.updateProviderConnection(c.id, { testStatus: "active" }); + + const back = await sqliteDb.getProviderConnectionById(c.id); + expect(back).toMatchObject({ + testStatus: "active", + lastError: null, + lastErrorAt: null, + errorCode: null, + backoffLevel: 0, + rateLimitedUntil: null, + modelLock_modelA: null, + modelLock_modelB: null, + }); + }); + + it("providerConnections: re-saving valid OAuth credentials clears stale routing locks", async () => { + const existing = await sqliteDb.createProviderConnection({ + provider: "health-reset-resave", + authType: "oauth", + email: "resave@example.com", + accessToken: "old-token", + }); + await sqliteDb.updateProviderConnection(existing.id, { + testStatus: "unavailable", + lastError: "Access denied", + errorCode: 403, + backoffLevel: 2, + modelLock_modelA: "2099-01-01T00:00:00.000Z", + }); + + const resaved = await sqliteDb.createProviderConnection({ + provider: "health-reset-resave", + authType: "oauth", + email: "resave@example.com", + accessToken: "new-token", + testStatus: "active", + }); + + expect(resaved.id).toBe(existing.id); + const back = await sqliteDb.getProviderConnectionById(existing.id); + expect(back).toMatchObject({ + accessToken: "new-token", + testStatus: "active", + lastError: null, + errorCode: null, + backoffLevel: 0, + modelLock_modelA: null, + }); + }); + + it("providerConnections: active soft warnings survive the health reset", async () => { + const c = await sqliteDb.createProviderConnection({ + provider: "health-reset-warning", + authType: "oauth", + email: "warning@example.com", + }); + await sqliteDb.updateProviderConnection(c.id, { + testStatus: "unavailable", + lastError: "Old failure", + modelLock_modelA: "2099-01-01T00:00:00.000Z", + }); + + const warningAt = "2026-09-06T00:00:00.000Z"; + await sqliteDb.updateProviderConnection(c.id, { + testStatus: "active", + lastError: "Connected, but credits are exhausted", + lastErrorAt: warningAt, + }); + + const back = await sqliteDb.getProviderConnectionById(c.id); + expect(back).toMatchObject({ + testStatus: "active", + lastError: "Connected, but credits are exhausted", + lastErrorAt: warningAt, + errorCode: null, + backoffLevel: 0, + modelLock_modelA: null, + }); + }); + it("providerConnections: GitHub OAuth uses account identity as fallback name", async () => { const c = await sqliteDb.createProviderConnection({ provider: "github", From 35b950be81314158999b63566cf3bbad502dcfd7 Mon Sep 17 00:00:00 2001 From: mrnim94 <50592567+mrnim94@users.noreply.github.com> Date: Wed, 9 Sep 2026 10:45:44 +0700 Subject: [PATCH 25/78] fix(kiro): route requests through current runtime surfaces and fix 400 REQUEST_BODY_INVALID (#3776) --- open-sse/executors/kiro.js | 40 +++++++++++++------ open-sse/translator/request/claude-to-kiro.js | 3 -- open-sse/translator/request/openai-to-kiro.js | 3 -- tests/translator/claude-kiro-direct.test.js | 7 ++-- .../kiro-api-key-endpoint-routing.test.js | 14 +++---- tests/unit/kiro-minimal-wire-payload.test.js | 19 +++++++++ tests/unit/openai-to-kiro.test.js | 2 +- 7 files changed, 57 insertions(+), 31 deletions(-) create mode 100644 tests/unit/kiro-minimal-wire-payload.test.js diff --git a/open-sse/executors/kiro.js b/open-sse/executors/kiro.js index 77616618..e12aced6 100644 --- a/open-sse/executors/kiro.js +++ b/open-sse/executors/kiro.js @@ -259,6 +259,19 @@ export class KiroExecutor extends BaseExecutor { } } + // CLIRO parity for the Amazon surfaces: the Kiro runtime accepts the + // SSO bearer header + agent-mode marker. Without these the deprecated + // path gateway answers REQUEST_BODY_INVALID for modern payloads. + if (credentials?.accessToken) { + headers["x-amz-sso-bearer"] = credentials.accessToken; + } + headers["x-amzn-kiro-agent-mode"] = "spec"; + headers["x-amzn-codewhisperer-machine-id"] = "kiro-desktop"; + const profileArn = credentials?.providerSpecificData?.profileArn; + if (profileArn) { + headers["x-amzn-codewhisperer-profile-arn"] = profileArn; + } + return headers; } @@ -285,9 +298,13 @@ export class KiroExecutor extends BaseExecutor { // 403 "bearer token invalid", so they must hit the CodeWhisperer // *.amazonaws.com surface, and in the region the token was minted in // (the baseUrls are hardcoded us-east-1). - const isCodeWhispererSurface = - authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc"; - if (!isCodeWhispererSurface) return baseUrls; + // Kiro deprecated the legacy path-style GenerateAssistantResponse on + // runtime.*.kiro.dev (IDE 1.0.228+ moved to POST / + x-amz-target). The + // path gateway now answers valid modern payloads with 400 + // REQUEST_BODY_INVALID, and 400 is terminal in BaseExecutor, so kiro.dev + // must never be the first surface for any auth method. Amazon surfaces + // reject foreign tokens with 401/403, which DO fall through, so trying + // q/codewhisperer first is safe for every auth method (CLIRO parity). const region = (credentials?.providerSpecificData?.region || "us-east-1").trim(); const regionalize = (u) => @@ -297,20 +314,17 @@ export class KiroExecutor extends BaseExecutor { const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize); const others = baseUrls.filter((u) => !u.includes("amazonaws.com")); - if (authMethod === "api_key") { - const q = amazon.filter((u) => u.includes("://q.")); - const remaining = amazon.filter((u) => !u.includes("://q.")); - return q.length > 0 - ? [...q, ...remaining, ...others] - : [...amazon, ...others]; - } - - return amazon.length > 0 ? [...amazon, ...others] : baseUrls; + const q = amazon.filter((u) => u.includes("://q.")); + const remaining = amazon.filter((u) => !u.includes("://q.")); + return q.length > 0 + ? [...q, ...remaining, ...others] + : [...amazon, ...others]; } buildUrl(model, stream, urlIndex = 0, credentials = null) { const baseUrls = this.getOrderedBaseUrls(credentials); - return baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl; + const url = baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl; + return url; } // Retry only endpoint/auth-surface failures. Payload-invalid HTTP 400 must be diff --git a/open-sse/translator/request/claude-to-kiro.js b/open-sse/translator/request/claude-to-kiro.js index 3cfad109..9a7cf377 100644 --- a/open-sse/translator/request/claude-to-kiro.js +++ b/open-sse/translator/request/claude-to-kiro.js @@ -316,14 +316,11 @@ export function claudeToKiroRequest(model, body, stream, credentials) { conversationState: { chatTriggerType: "MANUAL", conversationId, - agentContinuationId: continuationId, - agentTaskType: "vibe", currentMessage: { userInputMessage, }, history: canonical.history, }, - agentMode: "vibe", }; if (profileArn) payload.profileArn = profileArn; diff --git a/open-sse/translator/request/openai-to-kiro.js b/open-sse/translator/request/openai-to-kiro.js index 1d9bedec..990b607f 100644 --- a/open-sse/translator/request/openai-to-kiro.js +++ b/open-sse/translator/request/openai-to-kiro.js @@ -397,8 +397,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) { conversationState: { chatTriggerType: "MANUAL", conversationId, - agentContinuationId: continuationId, - agentTaskType: "vibe", currentMessage: { userInputMessage: { content: replayCurrent.content || "", @@ -414,7 +412,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) { }, history: canonical.history }, - agentMode: "vibe", }; if (profileArn) { diff --git a/tests/translator/claude-kiro-direct.test.js b/tests/translator/claude-kiro-direct.test.js index 3fca7be1..ab122b0c 100644 --- a/tests/translator/claude-kiro-direct.test.js +++ b/tests/translator/claude-kiro-direct.test.js @@ -26,9 +26,8 @@ describe("Claude → Kiro (direct route)", () => { expect(first.conversationState.conversationId).toBe("hermes-session-123-claude-replay"); expect(second.conversationState.conversationId).toBe("hermes-session-123-claude-replay"); - expect(first.conversationState.agentContinuationId).toBeTruthy(); - expect(second.conversationState.agentContinuationId).toBe(first.conversationState.agentContinuationId); - expect(first.conversationState.agentTaskType).toBe("vibe"); + expect(first.conversationState).not.toHaveProperty("agentContinuationId"); + expect(second.conversationState).not.toHaveProperty("agentTaskType"); expect(second.conversationState.history[0].userInputMessage.content).toBe( first.conversationState.currentMessage.userInputMessage.content ); @@ -84,7 +83,7 @@ describe("Claude → Kiro (direct route)", () => { expect(out.systemPrompt).toContain( "enabled" ); - expect(out.agentMode).toBe("vibe"); + expect(out).not.toHaveProperty("agentMode"); }); it("does not send additionalModelRequestFields for Kiro models without effort support", () => { diff --git a/tests/unit/kiro-api-key-endpoint-routing.test.js b/tests/unit/kiro-api-key-endpoint-routing.test.js index a0750adc..22cf97fc 100644 --- a/tests/unit/kiro-api-key-endpoint-routing.test.js +++ b/tests/unit/kiro-api-key-endpoint-routing.test.js @@ -20,26 +20,26 @@ describe("Kiro auth-aware endpoint routing", () => { ]); }); - it("keeps Builder ID OAuth on the Kiro runtime surface", () => { + it("routes Builder ID OAuth through Amazon Q first (runtime path deprecated)", () => { expect(executor.getOrderedBaseUrls(credentials("builder-id"))).toEqual([ - RUNTIME, - CODEWHISPERER, Q, + CODEWHISPERER, + RUNTIME, ]); }); - it("keeps external IdP on CodeWhisperer before Amazon Q", () => { + it("routes external IdP through Amazon Q first", () => { expect(executor.getOrderedBaseUrls(credentials("external_idp"))).toEqual([ - CODEWHISPERER, Q, + CODEWHISPERER, RUNTIME, ]); }); - it("regionalizes AWS endpoints for IDC without changing Kiro runtime", () => { + it("regionalizes AWS endpoints for IDC with Q first", () => { expect(executor.getOrderedBaseUrls(credentials("idc", "eu-west-1"))).toEqual([ - "https://codewhisperer.eu-west-1.amazonaws.com/generateAssistantResponse", "https://q.eu-west-1.amazonaws.com/generateAssistantResponse", + "https://codewhisperer.eu-west-1.amazonaws.com/generateAssistantResponse", RUNTIME, ]); }); diff --git a/tests/unit/kiro-minimal-wire-payload.test.js b/tests/unit/kiro-minimal-wire-payload.test.js new file mode 100644 index 00000000..97075d4b --- /dev/null +++ b/tests/unit/kiro-minimal-wire-payload.test.js @@ -0,0 +1,19 @@ +import { describe, expect, it } from "vitest"; +import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to-kiro.js"; +import { claudeToKiroRequest } from "../../open-sse/translator/request/claude-to-kiro.js"; + +for (const [name, translate, body] of [ + ["OpenAI", openaiToKiroRequest, { messages: [{ role: "user", content: "hello" }] }], + ["Claude", claudeToKiroRequest, { messages: [{ role: "user", content: "hello" }] }], +]) { + describe(`${name} Kiro minimal wire payload`, () => { + it("omits unsupported agent fields", () => { + const payload = translate("kiro/claude-sonnet-4.5", body, true, {}); + expect(payload).not.toHaveProperty("agentMode"); + expect(payload.conversationState).not.toHaveProperty("agentContinuationId"); + expect(payload.conversationState).not.toHaveProperty("agentTaskType"); + expect(payload.conversationState.chatTriggerType).toBe("MANUAL"); + expect(payload.conversationState.currentMessage.userInputMessage.origin).toBe("AI_EDITOR"); + }); + }); +} diff --git a/tests/unit/openai-to-kiro.test.js b/tests/unit/openai-to-kiro.test.js index 17773bdb..965bd2d1 100644 --- a/tests/unit/openai-to-kiro.test.js +++ b/tests/unit/openai-to-kiro.test.js @@ -606,7 +606,7 @@ describe("openaiToKiroRequest", () => { ); expect(second.conversationState.conversationId).toBe("hermes-session-openai-replay"); - expect(second.conversationState.agentContinuationId).toBe(first.conversationState.agentContinuationId); + expect(second.conversationState).not.toHaveProperty("agentContinuationId"); expect(second.conversationState.history[0].userInputMessage.content).toBe( first.conversationState.currentMessage.userInputMessage.content ); From 40dffbce538e6df8914362c5eb91d256397562bb Mon Sep 17 00:00:00 2001 From: 617929zcxc <101911800+617929zcxc@users.noreply.github.com> Date: Thu, 10 Sep 2026 18:53:23 +0700 Subject: [PATCH 26/78] feat(providers): add standalone Qwen provider --- open-sse/providers/registry/index.js | 3 ++- open-sse/providers/registry/qwen.js | 36 ++++++++++++++++++++++++++++ 2 files changed, 38 insertions(+), 1 deletion(-) create mode 100644 open-sse/providers/registry/qwen.js diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index ab71e2e7..48a49ebe 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -123,7 +123,7 @@ import p119 from "./selfhosted-embedding.js"; import p120 from "./fish-audio.js"; import p121 from "./alitp-intl.js"; import p122 from "./xquik.js"; - +import p124 from "./qwen.js"; export default [ p0, p1, @@ -247,4 +247,5 @@ export default [ p120, p121, p122, + p124, ]; diff --git a/open-sse/providers/registry/qwen.js b/open-sse/providers/registry/qwen.js new file mode 100644 index 00000000..d51e88a8 --- /dev/null +++ b/open-sse/providers/registry/qwen.js @@ -0,0 +1,36 @@ +export default { + id: "qwen", + priority: 12, + alias: "qwen", + aliases: ["qw"], + display: { + name: "Qwen", + icon: "sparkles", + color: "#6366F1", + textIcon: "Qw", + website: "https://www.alibabacloud.com/en/product/model-studio", + notice: { + apiKeyUrl: + "https://modelstudio.console.alibabacloud.com/?apiKey=1", + }, + }, + category: "apikey", + transport: { + baseUrl: + "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions", + headers: {}, + quirks: { preserveCacheControl: true }, + }, + models: [ + // Flagship & API models + { id: "qwen3.8-max", name: "Qwen3.8 Max" }, + { id: "qwen3.8-flash", name: "Qwen3.8 Flash" }, + { id: "qwen3.7-plus", name: "Qwen3.7 Plus" }, + { id: "qwen3.7-flash", name: "Qwen3.7 Flash" }, + // Open-source models + { id: "qwen3.8-2.4t-a95b", name: "Qwen3.8 2.4T MoE (Open)" }, + { id: "qwen3.8-27b", name: "Qwen3.8 27B (Open)" }, + { id: "qwen3.6-35b-a3b", name: "Qwen3.6 35B MoE (Open)" }, + { id: "qwen3.6-27b", name: "Qwen3.6 27B (Open)" }, + ], +}; From eee3515e5419cd0824bedb4beb0e0d91dd662b50 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 21:08:32 +0700 Subject: [PATCH 27/78] feat(opencode-go): add newly published Go models Add the models the provider docs now list but the registry lacked: chat/completions glm-5.3, kimi-k3, deepseek-flash, longcat-2.0, hy4-preview, hy3 + /messages qwen3.8-max, qwen3.8-flash responses only grok-4.6, gpt-5.6-luna Endpoints follow the table at https://opencode.ai/docs/go/. chat-only models stay on the sourceFormat-matched transport guard so a Claude client is never routed to /messages for a model that lacks it. Co-Authored-By: Claude Code --- open-sse/providers/registry/opencode-go.js | 15 +++++++++++++-- tests/unit/opencode-go-models.test.js | 18 +++++++++++------- 2 files changed, 24 insertions(+), 9 deletions(-) diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index c6673223..a715a98e 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -33,25 +33,36 @@ export default { { format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } }, { format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, ], + // supportedFormats follow the endpoint table in https://opencode.ai/docs/go/ models: [ { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] }, + { id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] }, { id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] }, { id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] }, { id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] }, { id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] }, + { id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] }, { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] }, + { id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] }, + { id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] }, { id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] }, { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] }, { id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] }, { id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] }, { id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] }, + { id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] }, + { id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, - // Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces - // chatCore past the sourceFormat-matched transports into translation (see chatCore guard). + { id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] }, + { id: "hy3", name: "Hy3", supportedFormats: ["openai"] }, + // Served by /zen/go/v1/responses only — the responses-only entry forces chatCore + // past the sourceFormat-matched transports into translation (see chatCore guard). + { id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, ], diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 7ece100a..0ed0804b 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -4,9 +4,11 @@ import { PROVIDERS } from "../../open-sse/config/providers.js"; import { resolveTransport } from "../../open-sse/services/provider.js"; // Chat-only models (no /messages, no /responses support on opencode-go) -const CHAT_ONLY = ["glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "mimo-v2.5", "mimo-v2.5-pro"]; +const CHAT_ONLY = ["glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", + "deepseek-flash", "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", "hy4-preview", "hy3"]; // Models that also expose the Anthropic /messages endpoint -const CLAUDE_CAPABLE = ["minimax-m3", "minimax-m2.7", "minimax-m2.5", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus"]; +const CLAUDE_CAPABLE = ["minimax-m3", "minimax-m2.7", "minimax-m2.5", + "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus"]; // Models that also expose the OpenAI /responses endpoint const RESPONSES_CAPABLE = ["deepseek-v4-pro", "deepseek-v4-flash"]; @@ -22,11 +24,13 @@ describe("OpenCode Go model catalog", () => { it("matches the documented model IDs", () => { const ids = (PROVIDER_MODELS["opencode-go"] || []).map((m) => m.id); expect(ids).toEqual([ - "glm-5.3-flash", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", - "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", - "mimo-v2.5", "mimo-v2.5-pro", + "glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", + "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "deepseek-flash", + "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", "minimax-m3", "minimax-m2.7", "minimax-m2.5", - "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "hy4-preview", "hy3", + "grok-4.6", "gpt-5.6-luna", "muse-spark-1.2-contributor", "muse-spark-1.3-contributor", ]); }); @@ -91,7 +95,7 @@ describe("OpenCode Go per-model transport guard (chatCore logic)", () => { }); it("routes Muse Spark (responses-only) to /responses, never to /messages", () => { - for (const m of ["muse-spark-1.2-contributor", "muse-spark-1.3-contributor"]) { + for (const m of ["muse-spark-1.2-contributor", "muse-spark-1.3-contributor", "grok-4.6", "gpt-5.6-luna"]) { expect(getModelSupportedFormats("opencode-go", m)).toEqual(["openai-responses"]); expect(pickTransport("opencode-go", "openai-responses", "opencode-go", m)?.baseUrl).toBe("https://opencode.ai/zen/go/v1/responses"); expect(pickTransport("opencode-go", "claude", "opencode-go", m)).toBeNull(); From a7047a07d475207a6685b1d288f1aa57c027f405 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 21:23:24 +0700 Subject: [PATCH 28/78] fix(codex): restore Version header and single-source the CLI version The image handler's `version` header was commented out, so Codex image requests reached chatgpt.com without the Version identity the backend expects. Restore it and route every Codex identity header through one constant. The CLI version now lives on registry codex.transport as `cliVersion` (the same pattern gemini-cli uses) and is re-exported as CODEX_CLI_VERSION, so the registry User-Agent, the image handler and the connection test can no longer drift apart. Bumped 0.136.0 -> 0.154.0 (current stable). Co-Authored-By: Claude Code --- open-sse/config/appConstants.js | 3 +++ open-sse/handlers/imageProviders/codex.js | 6 +++--- open-sse/providers/registry/codex.js | 7 ++++++- src/app/api/providers/[id]/test/testUtils.js | 3 ++- tests/__baseline__/providers-baseline.json | 3 ++- tests/unit/image-generation.test.js | 2 +- 6 files changed, 17 insertions(+), 7 deletions(-) diff --git a/open-sse/config/appConstants.js b/open-sse/config/appConstants.js index 3e18633e..4df1463e 100644 --- a/open-sse/config/appConstants.js +++ b/open-sse/config/appConstants.js @@ -7,6 +7,9 @@ import { createRequire } from "module"; export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion; export const GEMINI_CLI_API_CLIENT = PROVIDERS["gemini-cli"]?.apiClient; +// === Codex CLI === derive từ registry codex.transport +export const CODEX_CLI_VERSION = PROVIDERS["codex"]?.cliVersion; + // Map Node arch to Gemini CLI arch string (x64/x86/arm64/...) function geminiCLIArch() { const a = arch(); diff --git a/open-sse/handlers/imageProviders/codex.js b/open-sse/handlers/imageProviders/codex.js index 218302ab..385ec5b6 100644 --- a/open-sse/handlers/imageProviders/codex.js +++ b/open-sse/handlers/imageProviders/codex.js @@ -2,10 +2,10 @@ import { randomUUID } from "node:crypto"; import { nowSec } from "./_base.js"; import { PROVIDERS } from "../../config/providers.js"; +import { CODEX_CLI_VERSION } from "../../config/appConstants.js"; const CODEX_RESPONSES_URL = PROVIDERS["codex"].baseUrl; -const CODEX_USER_AGENT = "codex_cli_rs/0.136.0"; -const CODEX_VERSION = "0.136.0"; +const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`; const CODEX_ORIGINATOR = "codex_cli_rs"; const CODEX_MODEL_SUFFIX = "-image"; const CODEX_REF_DETAIL = "high"; @@ -157,7 +157,7 @@ export default { "originator": CODEX_ORIGINATOR, "session_id": randomUUID(), "user-agent": CODEX_USER_AGENT, - "version": CODEX_VERSION, + "version": CODEX_CLI_VERSION, "x-client-request-id": randomUUID(), }; }, diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 1909fe97..bf436a10 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -1,5 +1,9 @@ import { withCodexReviewModels } from "../models/helpers.js"; +// Codex CLI version seen by OpenAI's backend — single source for the Version / +// User-Agent identity headers. Bump when the installed codex CLI is upgraded. +const CODEX_CLI_VERSION = "0.154.0"; + export default { id: "codex", priority: 30, @@ -34,9 +38,10 @@ export default { baseUrl: "https://chatgpt.com/backend-api/codex/responses", format: "openai-responses", forceStream: true, + cliVersion: CODEX_CLI_VERSION, headers: { originator: "codex_cli_rs", - "User-Agent": "codex_cli_rs/0.136.0", + "User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`, }, usage: { url: "https://chatgpt.com/backend-api/wham/usage", diff --git a/src/app/api/providers/[id]/test/testUtils.js b/src/app/api/providers/[id]/test/testUtils.js index bd0c4782..9572e69e 100644 --- a/src/app/api/providers/[id]/test/testUtils.js +++ b/src/app/api/providers/[id]/test/testUtils.js @@ -4,6 +4,7 @@ import { testProxyUrl } from "@/lib/network/proxyTest"; import { isOpenAICompatibleProvider, isAnthropicCompatibleProvider } from "@/shared/constants/providers"; import { getDefaultModel } from "open-sse/config/providerModels.js"; import { resolveOllamaLocalHost, PROVIDERS } from "open-sse/config/providers.js"; +import { CODEX_CLI_VERSION } from "open-sse/config/appConstants.js"; import { refreshProviderCredentials, shouldRefreshCredentials, @@ -27,7 +28,7 @@ const OAUTH_TEST_CONFIG = { method: "POST", authHeader: "Authorization", authPrefix: "Bearer ", - extraHeaders: { "Content-Type": "application/json", "originator": "codex_cli_rs", "User-Agent": "codex_cli_rs/0.136.0" }, + extraHeaders: { "Content-Type": "application/json", "originator": "codex_cli_rs", "User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}` }, // Minimal invalid body — triggers fast 400 without consuming quota body: JSON.stringify({ model: "gpt-5.3-codex", input: [], stream: false, store: false }), // 400 (bad request) means auth succeeded; only 401/403 means token is bad diff --git a/tests/__baseline__/providers-baseline.json b/tests/__baseline__/providers-baseline.json index db0b8bad..2dd09f55 100644 --- a/tests/__baseline__/providers-baseline.json +++ b/tests/__baseline__/providers-baseline.json @@ -192,9 +192,10 @@ "baseUrl": "https://chatgpt.com/backend-api/codex/responses", "format": "openai-responses", "forceStream": true, + "cliVersion": "0.154.0", "headers": { "originator": "codex_cli_rs", - "User-Agent": "codex_cli_rs/0.136.0" + "User-Agent": "codex_cli_rs/0.154.0" }, "usage": { "url": "https://chatgpt.com/backend-api/wham/usage", diff --git a/tests/unit/image-generation.test.js b/tests/unit/image-generation.test.js index 12dce95d..4df94afc 100644 --- a/tests/unit/image-generation.test.js +++ b/tests/unit/image-generation.test.js @@ -351,7 +351,7 @@ describe("handleImageGenerationCore", () => { headers: expect.objectContaining({ authorization: "Bearer codex-token", "chatgpt-account-id": "account-123", - version: "0.136.0", + version: "0.154.0", }), }) ); From 4a390685b3d38054305bb014a63910a827d6c652 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 21:56:30 +0700 Subject: [PATCH 29/78] feat(cli): group model selector by provider with search Replace the flat numbered model list with provider-grouped browsing (combos first, then providers by alias order), full-text search across all models, and manual custom model ID entry. A single available category opens directly into its model list. Also bump root and cli packages to 0.5.75 and ignore packed `9router-*` tarballs. Co-Authored-By: Claude Code --- .gitignore | 3 +- cli/package.json | 2 +- cli/src/cli/utils/modelSelector.js | 227 +++++++++++++++++++++++------ package.json | 2 +- 4 files changed, 186 insertions(+), 48 deletions(-) diff --git a/.gitignore b/.gitignore index 0f0fb03c..6773d550 100644 --- a/.gitignore +++ b/.gitignore @@ -88,4 +88,5 @@ graphify-out/* .next-analyze/* # Kiro local workspace state -.kiro/ \ No newline at end of file +.kiro/ +9router-* diff --git a/cli/package.json b/cli/package.json index db516ac4..58b9dfa8 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.69", + "version": "0.5.75", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" diff --git a/cli/src/cli/utils/modelSelector.js b/cli/src/cli/utils/modelSelector.js index 438220d1..a2bd3fdc 100644 --- a/cli/src/cli/utils/modelSelector.js +++ b/cli/src/cli/utils/modelSelector.js @@ -55,7 +55,7 @@ async function getAvailableModelsGrouped() { } /** - * Display model list and prompt for selection + * Display model list and prompt for selection with provider grouping & search * @param {string} title - Title to display * @param {string} currentValue - Current selected value (optional) * @param {Object} options - { excludeCombos?: boolean } @@ -70,62 +70,199 @@ async function selectModelFromList(title, currentValue = "", options = {}) { if (totalModels === 0) { return null; } - - // Build flat list for selection - const allModels = []; - - // Display - clearScreen(); - console.log(`\n🎯 ${title}`); - console.log("=".repeat(50)); - if (currentValue) { - console.log(`Current: ${currentValue}\n`); - } else { - console.log(); - } - - let idx = 1; - - // Combos first (skipped when excludeCombos is true) + + // All models for flat search + const allModelsList = [ + ...combos, + ...Object.values(groups).flat() + ]; + + // Build category list + const categories = []; if (combos.length > 0) { - console.log("[Combos]"); - combos.forEach(combo => { - console.log(` ${idx}. ${combo}`); - allModels.push(combo); - idx++; + categories.push({ + id: "combos", + name: "[Combos]", + models: combos }); - console.log(); } - - // Provider groups in order (by alias) + const sortedProviders = Object.keys(groups).sort((a, b) => { const idxA = PROVIDER_ALIAS_ORDER.indexOf(a); const idxB = PROVIDER_ALIAS_ORDER.indexOf(b); return (idxA === -1 ? 999 : idxA) - (idxB === -1 ? 999 : idxB); }); - - sortedProviders.forEach(provider => { + + sortedProviders.forEach((provider) => { const providerName = PROVIDER_ALIAS_NAMES[provider] || provider; - console.log(`[${providerName}]`); - groups[provider].forEach(model => { - console.log(` ${idx}. ${model}`); - allModels.push(model); - idx++; + categories.push({ + id: provider, + name: providerName, + models: groups[provider] }); - console.log(); }); - - console.log(" 0. Cancel\n"); - - // Prompt for number input - const input = await prompt("Enter number: "); - const num = parseInt(input, 10); - - if (isNaN(num) || num === 0 || num < 0 || num > allModels.length) { - return null; + + let filterQuery = null; + + while (true) { + clearScreen(); + console.log(`\n🎯 ${title}`); + console.log("=".repeat(50)); + if (currentValue) { + console.log(`Current: ${currentValue}\n`); + } else { + console.log(); + } + + // Active search view + if (filterQuery !== null) { + const q = filterQuery.toLowerCase().trim(); + const matched = allModelsList.filter((m) => m.toLowerCase().includes(q)); + + console.log(`🔍 Search results for "${filterQuery}": (${matched.length} found)\n`); + if (matched.length === 0) { + console.log(" No matching models found.\n"); + console.log(" 0. ← Back to providers"); + console.log(" s. Search again\n"); + const act = await prompt("Select option: "); + if (act.toLowerCase() === "s") { + const newQ = await prompt("Enter search keyword: "); + filterQuery = newQ.trim() || null; + } else { + filterQuery = null; + } + continue; + } + + matched.forEach((m, i) => { + console.log(` ${i + 1}. ${m}`); + }); + console.log("\n 0. ← Back to providers"); + console.log(" s. Search again\n"); + + const input = await prompt("Enter number to select (or 0/s): "); + if (input.toLowerCase() === "s") { + const newQ = await prompt("Enter search keyword: "); + filterQuery = newQ.trim() || null; + continue; + } + const num = parseInt(input, 10); + if (isNaN(num) || num === 0) { + filterQuery = null; + continue; + } + if (num > 0 && num <= matched.length) { + return matched[num - 1]; + } + continue; + } + + // If only 1 category exists, jump straight into its model list + if (categories.length === 1) { + const singleCategory = categories[0]; + console.log(`[${singleCategory.name}]`); + singleCategory.models.forEach((m, i) => { + console.log(` ${i + 1}. ${m}`); + }); + console.log(); + console.log(" s. 🔍 Search models"); + console.log(" m. ✍️ Enter custom model ID"); + console.log(" 0. Cancel\n"); + + const input = await prompt("Enter choice (number / s / m / 0): "); + const trimmed = input.trim(); + if (!trimmed || trimmed === "0") return null; + + const lower = trimmed.toLowerCase(); + if (lower === "s") { + const q = await prompt("Enter search keyword: "); + if (q.trim()) filterQuery = q.trim(); + continue; + } + if (lower === "m") { + const customModel = await prompt("Enter custom model ID: "); + if (customModel.trim()) return customModel.trim(); + continue; + } + + const num = parseInt(trimmed, 10); + if (!isNaN(num) && num > 0 && num <= singleCategory.models.length) { + return singleCategory.models[num - 1]; + } + filterQuery = trimmed; + continue; + } + + // Multiple categories view + console.log("[Providers & Groups]"); + categories.forEach((cat, i) => { + console.log(` ${i + 1}. ${cat.name} (${cat.models.length} models)`); + }); + + console.log(); + console.log(" s. 🔍 Search models"); + console.log(" m. ✍️ Enter custom model ID"); + console.log(" 0. Cancel\n"); + + const input = await prompt("Enter choice (number / keyword / s / m): "); + const trimmed = input.trim(); + + if (!trimmed || trimmed === "0") { + return null; + } + + const lower = trimmed.toLowerCase(); + if (lower === "s") { + const q = await prompt("Enter search keyword: "); + if (q.trim()) { + filterQuery = q.trim(); + } + continue; + } + + if (lower === "m") { + const customModel = await prompt("Enter custom model ID: "); + if (customModel.trim()) { + return customModel.trim(); + } + continue; + } + + const num = parseInt(trimmed, 10); + // Selected a category + if (!isNaN(num) && num > 0 && num <= categories.length) { + const selectedCategory = categories[num - 1]; + + while (true) { + clearScreen(); + console.log(`\n🎯 ${title} > ${selectedCategory.name}`); + console.log("=".repeat(50)); + if (currentValue) { + console.log(`Current: ${currentValue}\n`); + } else { + console.log(); + } + + selectedCategory.models.forEach((m, i) => { + console.log(` ${i + 1}. ${m}`); + }); + console.log("\n 0. ← Back\n"); + + const modelChoice = await prompt("Enter number to select (0 to back): "); + const modelNum = parseInt(modelChoice, 10); + if (isNaN(modelNum) || modelNum === 0) { + break; + } + if (modelNum > 0 && modelNum <= selectedCategory.models.length) { + return selectedCategory.models[modelNum - 1]; + } + } + continue; + } + + // User typed text directly -> treat as search query + filterQuery = trimmed; } - - return allModels[num - 1]; } module.exports = { diff --git a/package.json b/package.json index 80f6d46b..15b7eca7 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "9router-app", - "version": "0.5.69", + "version": "0.5.75", "description": "9Router web dashboard", "private": true, "scripts": { From 807553e24662a219fecabca38b1fc9626e2b7fbc Mon Sep 17 00:00:00 2001 From: zmf Date: Thu, 10 Sep 2026 21:57:22 +0700 Subject: [PATCH 30/78] feat(codebuddy-cn): replace deepseek-v4-flash with deepseek-v4.1-flash MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The server's product-config payload (which the IDE plugin fetches from copilot.tencent.com) publishes deepseek-v4.1-flash and no longer lists deepseek-v4-flash, so the old id is dropped — same pattern as the previous catalog refreshes (#3648, #3802). The v4-flash endpoint still answers 200, but the published list is the contract. Per the server table, maxOutput rises 50000 -> 128000 while contextWindow stays 1000000. - registry/codebuddy-cn.js: models[] entry swapped to the new id - capabilities.js: per-model entry swapped, maxOutput -> 128000 No changes needed in thinkingLevels.js (the deepseek-v4* pattern already matches the new id and publishes low/high/xhigh, matching the server's supportedEfforts or pricing.js (the deepseek-v* glob yields the same rates). EOF ) --- open-sse/providers/capabilities.js | 5 ++++- open-sse/providers/registry/codebuddy-cn.js | 6 ++++-- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index a9f8745a..dd9fa381 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -209,7 +209,10 @@ export const PROVIDER_CAPABILITIES = { "glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 }, "kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 }, "deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 }, - "deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 }, + // deepseek-v4.1-flash replaces v4-flash (dropped from the server list; + // the old endpoint still answers 200 but the published list is the + // contract). maxOutput 128000 per the server's product-config payload. + "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 }, }, // Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the // registry `name` is display-only and capability lookup matches on the raw diff --git a/open-sse/providers/registry/codebuddy-cn.js b/open-sse/providers/registry/codebuddy-cn.js index fdb53e68..672a05d8 100644 --- a/open-sse/providers/registry/codebuddy-cn.js +++ b/open-sse/providers/registry/codebuddy-cn.js @@ -58,7 +58,9 @@ export default { // (endpoint returns 11102 "model service info not found"), plus // glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview / // deepseek-v3-2-volc (absent from the server list, though still answering - // 200) and hy3-x (paid tier, not used here). + // 200) and hy3-x (paid tier, not used here). deepseek-v4-flash removed + // 2026-09: replaced server-side by deepseek-v4.1-flash (same low/high/ + // xhigh efforts; endpoint still answers 200 but the list is the contract). // "-x" suffix = paid tier of the same model (free id rides the promo quota). { id: "hy3", name: "Hy3" }, { id: "hy4-preview", name: "Hy4-Preview" }, @@ -66,7 +68,7 @@ export default { { id: "glm-5.3-flash", name: "GLM-5.3-Flash" }, { id: "kimi-k3-1", name: "Kimi-K3" }, { id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" }, - { id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" }, + { id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" }, ], oauth: { baseUrl: "https://copilot.tencent.com", From 3288bbc47e75c90a9ebf02ac90a5fe934b41b90c Mon Sep 17 00:00:00 2001 From: coozgan Date: Thu, 10 Sep 2026 22:05:10 +0700 Subject: [PATCH 31/78] feat(video): add OpenRouter and Vertex AI (Veo) video generation Video generation was xAI-only. Adds an adapter layer under open-sse/handlers/videoProviders/ so /v1/videos/* can target OpenRouter or Google Cloud credentials. A provider with no adapter keeps the exact previous behaviour (raw body to {baseUrl}/{action}, poll {baseUrl}/{id}, verbatim passthrough), so the xAI path is unchanged. - openrouter: async job shape identical to xAI; creation POSTs to the /videos collection root (no /generations suffix) and the registry HTTP-Referer / X-Title headers are applied. Bodies pass through verbatim. - vertex: two-way translation, since Veo does not speak the OpenAI-ish videos shape. create -> :predictLongRunning { instances[], parameters{} }, poll -> :fetchPredictOperation (Veo has no REST GET poll). The operation resource name is base64url-encoded into the job id so GET /v1/videos/{id} stays a flat path. Access tokens are minted from Service Account JSON via the existing refreshVertexToken; raw API keys are rejected up front. The operation response maps back onto the { id, status, video, videos } shape clients already poll. - videoCore: the request plan is rebuilt per attempt, so the 401 -> refresh once -> retry once path picks up the refreshed token. Adapter validation errors return 400 before any upstream call, so a malformed request can never create a billable job. - videoGeneration: GET /v1/videos/{id} resolves the provider from the pinned x-connection-id connection, then ?provider=, then falls back to the xAI default. - registry: openrouter and vertex gain videoConfig, the video serviceKind and video-kind models (Veo 3.1 / 3 / 2, Sora 2 Pro, Seedance 2.0). --- open-sse/handlers/videoCore.js | 67 ++++- open-sse/handlers/videoProviders/index.js | 13 + .../handlers/videoProviders/openrouter.js | 39 +++ open-sse/handlers/videoProviders/vertex.js | 145 ++++++++++ open-sse/providers/registry/openrouter.js | 11 +- open-sse/providers/registry/vertex.js | 9 +- src/sse/handlers/videoGeneration.js | 19 +- tests/unit/video-providers.test.js | 263 ++++++++++++++++++ tests/unit/xai-video-handler.test.js | 1 + 9 files changed, 551 insertions(+), 16 deletions(-) create mode 100644 open-sse/handlers/videoProviders/index.js create mode 100644 open-sse/handlers/videoProviders/openrouter.js create mode 100644 open-sse/handlers/videoProviders/vertex.js create mode 100644 tests/unit/video-providers.test.js diff --git a/open-sse/handlers/videoCore.js b/open-sse/handlers/videoCore.js index 98d60157..45f14d86 100644 --- a/open-sse/handlers/videoCore.js +++ b/open-sse/handlers/videoCore.js @@ -2,6 +2,7 @@ import { createErrorResult } from "../utils/error.js"; import { HTTP_STATUS } from "../config/runtimeConfig.js"; import { refreshTokenByProvider } from "../services/tokenRefresh.js"; import { PROVIDER_MEDIA } from "../providers/index.js"; +import { getVideoAdapter } from "./videoProviders/index.js"; // Upstream fetch deadline for video job submission/polling (the job itself is // async upstream — this only bounds the HTTP round-trip, not video rendering). @@ -94,21 +95,49 @@ export async function handleVideoProxyCore({ return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`); } - const method = requestId ? "GET" : "POST"; - const url = buildUpstreamUrl(config, action, requestId); + const adapter = getVideoAdapter(provider); const fetchSignal = combineSignals(signal, timeoutMs); - const doFetch = (token) => - fetch(url, { + // Default (xAI shape) request plan; adapters override URL/method/headers/body. + const defaultPlan = () => { + const method = requestId ? "GET" : "POST"; + return { method, - headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }), + url: buildUpstreamUrl(config, action, requestId), + headers: buildHeaders({ + token: credentials?.accessToken || credentials?.apiKey, + contentType: method === "POST" ? contentType : null, + idempotencyKey: method === "POST" ? idempotencyKey : null, + }), body: method === "POST" ? rawBody : undefined, - signal: fetchSignal, - }); + }; + }; + // Rebuilt per attempt so the auth retry below picks up the refreshed token. + const doFetch = async () => { + const plan = adapter + ? await adapter.buildRequest({ + config, action, requestId, rawBody, contentType, idempotencyKey, credentials, log, + token: credentials?.accessToken || credentials?.apiKey, + }) + : defaultPlan(); + if (plan.error) return { planError: plan.error }; + return { + response: await fetch(plan.url, { + method: plan.method, + headers: plan.headers, + body: plan.body, + signal: fetchSignal, + }), + }; + }; + + const method = requestId ? "GET" : "POST"; let upstream; try { - upstream = await doFetch(credentials?.accessToken || credentials?.apiKey); + const first = await doFetch(); + if (first.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${first.planError}`); + upstream = first.response; } catch (error) { if (error?.name === "AbortError" || error?.name === "TimeoutError") { return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`); @@ -136,7 +165,9 @@ export async function handleVideoProxyCore({ await upstream.body?.cancel?.(); } catch { /* noop */ } try { - upstream = await doFetch(credentials.accessToken || credentials.apiKey); + const retry = await doFetch(); + if (retry.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${retry.planError}`); + upstream = retry.response; } catch (error) { return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials)); } @@ -152,13 +183,25 @@ export async function handleVideoProxyCore({ return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`); } - // Success: pass the upstream JSON through untouched (request_id / status / video.url). + // Success: pass the upstream JSON through untouched (request_id / status / video.url), + // unless the adapter maps a provider-native shape onto it (Vertex operations). + let outBody = bodyText; + let outType = upstream.headers.get("content-type") || "application/json"; + if (adapter?.transformResponse) { + try { + outBody = JSON.stringify(adapter.transformResponse(JSON.parse(bodyText))); + outType = "application/json"; + } catch { + // Non-JSON or unexpected shape — fall back to the raw upstream body. + } + } + return { success: true, - response: new Response(bodyText, { + response: new Response(outBody, { status: upstream.status, headers: { - "Content-Type": upstream.headers.get("content-type") || "application/json", + "Content-Type": outType, "Access-Control-Allow-Origin": "*", }, }), diff --git a/open-sse/handlers/videoProviders/index.js b/open-sse/handlers/videoProviders/index.js new file mode 100644 index 00000000..28173972 --- /dev/null +++ b/open-sse/handlers/videoProviders/index.js @@ -0,0 +1,13 @@ +// Video provider adapters. +// +// Default (no adapter) = xAI shape: raw body forwarded to {baseUrl}/{action}, +// polled at {baseUrl}/{id}, upstream JSON passed through verbatim. +// A provider only needs an adapter when its wire format differs from that. +import openrouter from "./openrouter.js"; +import vertex from "./vertex.js"; + +const ADAPTERS = { openrouter, vertex }; + +export function getVideoAdapter(provider) { + return ADAPTERS[provider] || null; +} diff --git a/open-sse/handlers/videoProviders/openrouter.js b/open-sse/handlers/videoProviders/openrouter.js new file mode 100644 index 00000000..90198a39 --- /dev/null +++ b/open-sse/handlers/videoProviders/openrouter.js @@ -0,0 +1,39 @@ +// OpenRouter video jobs — https://openrouter.ai/docs/api/api-reference/videos +// +// Same async shape as xAI (POST → { id, status }, GET → status/unsigned_urls), +// two differences only: creation POSTs to the collection root (no `/generations` +// suffix) and the account headers come from the registry entry. +// Response bodies are passed through verbatim. + +// ponytail: generations only — OpenRouter has no edits/extensions endpoint today. +const SUPPORTED_ACTIONS = new Set(["generations"]); + +function headers(config, token) { + return { + Accept: "application/json", + ...(config.headers || {}), + ...(token ? { Authorization: `Bearer ${token}` } : {}), + }; +} + +export default { + buildRequest({ config, action, requestId, rawBody, contentType, token }) { + const base = config.baseUrl.replace(/\/$/, ""); + + if (requestId) { + return { method: "GET", url: `${base}/${encodeURIComponent(requestId)}`, headers: headers(config, token) }; + } + if (!SUPPORTED_ACTIONS.has(action)) { + return { error: `OpenRouter video supports 'generations' only (got '${action}')` }; + } + if (contentType && !contentType.includes("application/json")) { + return { error: "OpenRouter video requires an application/json body" }; + } + return { + method: "POST", + url: base, + headers: { ...headers(config, token), "Content-Type": "application/json" }, + body: rawBody, + }; + }, +}; diff --git a/open-sse/handlers/videoProviders/vertex.js b/open-sse/handlers/videoProviders/vertex.js new file mode 100644 index 00000000..25394b94 --- /dev/null +++ b/open-sse/handlers/videoProviders/vertex.js @@ -0,0 +1,145 @@ +// Vertex AI (Veo) video jobs. +// +// Vertex does NOT speak the OpenAI-ish /v1/videos shape, so unlike OpenRouter +// this adapter translates both directions: +// create → POST {model}:predictLongRunning { instances[], parameters{} } → { name } +// poll → POST {model}:fetchPredictOperation { operationName } → { done, response } +// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation +// +// The operation name is a resource path (contains "/"), so it is base64url-encoded +// into the job id returned to the client — GET /v1/videos/{id} stays a flat path. +import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefresh.js"; + +const DEFAULT_LOCATION = "us-central1"; + +const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url"); +const decodeJobId = (id) => Buffer.from(id, "base64url").toString("utf8"); + +async function resolveAuth(credentials, log) { + const saJson = parseVertexSaJson(credentials?.apiKey); + const projectId = + saJson?.project_id || + credentials?.projectId || + credentials?.providerSpecificData?.projectId; + const location = credentials?.providerSpecificData?.location || DEFAULT_LOCATION; + + if (!projectId) { + return { error: "Vertex video requires a project_id — use Service Account JSON or set providerSpecificData.projectId" }; + } + + let token = credentials?.accessToken; + if (saJson) { + const minted = await refreshVertexToken(saJson, log); + if (!minted?.accessToken) return { error: "Vertex video: failed to mint access token from service account JSON" }; + token = minted.accessToken; + } + if (!token) return { error: "Vertex video requires Service Account JSON or an OAuth access token (raw API keys are not supported)" }; + + return { token, projectId, location }; +} + +/** OpenAI-ish video body → Vertex predictLongRunning body. */ +function toVertexBody(body) { + const instance = { prompt: body.prompt }; + // Image-to-video: accept the Vertex-native shape or a bare data URL / base64 string. + const image = body.image ?? body.image_url; + if (image && typeof image === "object") { + instance.image = image; + } else if (typeof image === "string") { + const match = image.match(/^data:([^;]+);base64,(.*)$/s); + instance.image = match + ? { bytesBase64Encoded: match[2], mimeType: match[1] } + : { gcsUri: image }; + } + if (body.video && typeof body.video === "object") instance.video = body.video; + + const parameters = {}; + if (body.n != null) parameters.sampleCount = Number(body.n); + if (body.duration != null) parameters.durationSeconds = Number(body.duration); + if (body.aspect_ratio) parameters.aspectRatio = body.aspect_ratio; + if (body.resolution) parameters.resolution = body.resolution; + if (body.seed != null) parameters.seed = body.seed; + if (body.negative_prompt) parameters.negativePrompt = body.negative_prompt; + // Without storageUri Vertex returns inline base64 bytes; a GCS bucket keeps + // the poll response small and is what production callers want. + if (body.storage_uri) parameters.storageUri = body.storage_uri; + if (body.generate_audio != null) parameters.generateAudio = !!body.generate_audio; + + return { instances: [instance], ...(Object.keys(parameters).length ? { parameters } : {}) }; +} + +/** Vertex operation → the async-job shape 9Router clients already poll for. */ +function fromVertexOperation(json) { + if (!json?.name) return json; + const id = encodeJobId(json.name); + if (json.error) { + return { id, request_id: id, status: "failed", error: json.error }; + } + if (!json.done) { + return { id, request_id: id, status: "pending" }; + } + const samples = + json.response?.videos || + json.response?.generateVideoResponse?.generatedSamples || + []; + const videos = samples.map((s) => ({ + url: s.gcsUri || s.video?.uri || s.uri || null, + b64_json: s.bytesBase64Encoded || s.video?.bytesBase64Encoded || null, + mime_type: s.mimeType || s.video?.mimeType || "video/mp4", + })); + return { id, request_id: id, status: "completed", video: videos[0] || null, videos }; +} + +export default { + async buildRequest({ config, action, requestId, rawBody, contentType, credentials, log }) { + if (contentType && !contentType.includes("application/json")) { + return { error: "Vertex video requires an application/json body" }; + } + + const auth = await resolveAuth(credentials, log); + if (auth.error) return { error: auth.error }; + const { token, projectId, location } = auth; + const base = (config.baseUrl || "https://aiplatform.googleapis.com").replace(/\/$/, ""); + const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` }; + + if (requestId) { + let operationName; + try { + operationName = decodeJobId(requestId); + } catch { + return { error: "Invalid Vertex video job id" }; + } + const modelPath = operationName.split("/operations/")[0]; + if (!modelPath || modelPath === operationName) return { error: "Invalid Vertex video job id" }; + return { + method: "POST", + url: `${base}/v1/${modelPath}:fetchPredictOperation`, + headers, + body: JSON.stringify({ operationName }), + }; + } + + if (action !== "generations") { + // ponytail: Veo extend/edit go through generations with `video`/`image` in the body. + return { error: `Vertex video supports 'generations' only (got '${action}')` }; + } + + let body; + try { + body = JSON.parse(typeof rawBody === "string" ? rawBody : rawBody.toString("utf8")); + } catch { + return { error: "Invalid JSON body" }; + } + if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" }; + if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" }; + + return { + method: "POST", + url: `${base}/v1/projects/${projectId}/locations/${location}/publishers/google/models/${body.model}:predictLongRunning`, + headers, + body: JSON.stringify(toVertexBody(body)), + }; + }, + + transformResponse: fromVertexOperation, +}; diff --git a/open-sse/providers/registry/openrouter.js b/open-sse/providers/registry/openrouter.js index a0df2a52..68a92857 100644 --- a/open-sse/providers/registry/openrouter.js +++ b/open-sse/providers/registry/openrouter.js @@ -40,8 +40,11 @@ export default { { id: "openai/gpt-image-1", name: "GPT Image 1 (via OpenRouter)", params: ["n","size","quality","response_format"], kind: "image" }, { id: "google/imagen-3.0-generate-002", name: "Imagen 3 (via OpenRouter)", params: ["n","size"], kind: "image" }, { id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell (via OpenRouter)", params: ["n","size"], kind: "image" }, + { id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" }, + { id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" }, + { id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" }, ], - serviceKinds: ["llm","embedding","tts","imageToText"], + serviceKinds: ["llm","embedding","tts","imageToText","video"], ttsConfig: { baseUrl: "https://openrouter.ai/api/v1/chat/completions", defaultModel: "openai/gpt-4o-mini-tts", @@ -57,6 +60,12 @@ export default { baseUrl: "https://openrouter.ai/api/v1/images/generations", headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"}, }, + // Async video jobs (POST /videos → { id, status }, GET /videos/{id} polls). + // Docs: https://openrouter.ai/docs/api/api-reference/videos + videoConfig: { + baseUrl: "https://openrouter.ai/api/v1/videos", + headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"}, + }, modelsFetcher: { url: "https://openrouter.ai/api/v1/models", type: "openrouter-free" }, passthroughModels: true, }; diff --git a/open-sse/providers/registry/vertex.js b/open-sse/providers/registry/vertex.js index b8765de3..a1c60e60 100644 --- a/open-sse/providers/registry/vertex.js +++ b/open-sse/providers/registry/vertex.js @@ -27,6 +27,13 @@ export default { { id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" }, { id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" }, { id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" }, + { id: "veo-3.1-generate-preview", name: "Veo 3.1 (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" }, + { id: "veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" }, + { id: "veo-3.0-generate-001", name: "Veo 3", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" }, + { id: "veo-2.0-generate-001", name: "Veo 2", params: ["duration","aspect_ratio","negative_prompt","seed","storage_uri"], kind: "video" }, ], - serviceKinds: ["llm","imageToText"], + serviceKinds: ["llm","imageToText","video"], + // Veo via predictLongRunning + fetchPredictOperation (adapter: handlers/videoProviders/vertex.js). + // Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation + videoConfig: { baseUrl: "https://aiplatform.googleapis.com" }, }; diff --git a/src/sse/handlers/videoGeneration.js b/src/sse/handlers/videoGeneration.js index 67142899..af19da19 100644 --- a/src/sse/handlers/videoGeneration.js +++ b/src/sse/handlers/videoGeneration.js @@ -5,7 +5,7 @@ import { extractApiKey, isValidApiKey, } from "../services/auth.js"; -import { getSettings } from "@/lib/localDb"; +import { getSettings, getProviderConnectionById } from "@/lib/localDb"; import { getModelInfo } from "../services/model.js"; import { handleVideoProxyCore, getVideoConfig, sanitizeSecrets } from "open-sse/handlers/videoCore.js"; import { errorResponse, unavailableResponse } from "open-sse/utils/error.js"; @@ -17,6 +17,21 @@ import * as log from "../utils/logger.js"; // (bare model id, or multipart bodies we deliberately don't parse) land here. const DEFAULT_VIDEO_PROVIDER = "xai"; +/** + * Poll requests carry no model, so the provider comes from the pinned + * connection (`x-connection-id`, returned on create) or an explicit + * `?provider=` — falling back to the historical xAI default. + */ +async function resolveGetProvider(request, connectionId) { + if (connectionId) { + const conn = await getProviderConnectionById(connectionId).catch(() => null); + if (conn?.provider && getVideoConfig(conn.provider)) return conn.provider; + } + const queried = new URL(request.url).searchParams.get("provider"); + if (queried && getVideoConfig(queried)) return queried; + return DEFAULT_VIDEO_PROVIDER; +} + // Creation POSTs are billable jobs — only rotate to another account for // errors that upstream rejects BEFORE creating a job (auth/quota). A 5xx may // have created the job, so it is returned to the caller instead of re-sent. @@ -185,8 +200,8 @@ export async function handleVideoGet(request, requestId) { if (!requestId) return errorResponse(HTTP_STATUS.BAD_REQUEST, "Missing video request id"); - const provider = DEFAULT_VIDEO_PROVIDER; const preferredConnectionId = request.headers.get("x-connection-id") || null; + const provider = await resolveGetProvider(request, preferredConnectionId); const credentials = await getProviderCredentials(provider, null, null, { preferredConnectionId }); if (!credentials || credentials.allRateLimited) { diff --git a/tests/unit/video-providers.test.js b/tests/unit/video-providers.test.js new file mode 100644 index 00000000..680db5a5 --- /dev/null +++ b/tests/unit/video-providers.test.js @@ -0,0 +1,263 @@ +/** + * Unit tests for the OpenRouter + Vertex (Veo) video adapters. + * + * Covers: + * - registry wiring (videoConfig, video serviceKind, video-kind models) + * - OpenRouter: POST to the collection root, GET poll, verbatim passthrough + * - Vertex: predictLongRunning body translation, fetchPredictOperation polling, + * operation-name round-trip through the job id, response mapping + * - xAI default path is unchanged by the adapter hook + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +vi.mock("open-sse/services/tokenRefresh.js", async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, refreshTokenByProvider: vi.fn(), refreshVertexToken: vi.fn() }; +}); + +import { handleVideoProxyCore, getVideoConfig } from "open-sse/handlers/videoCore.js"; +import { refreshVertexToken } from "open-sse/services/tokenRefresh.js"; +import { PROVIDER_MEDIA, PROVIDER_MODELS } from "open-sse/providers/index.js"; + +const originalFetch = global.fetch; +const jsonResponse = (body, status = 200) => + new Response(JSON.stringify(body), { status, headers: { "Content-Type": "application/json" } }); + +// Vertex operation names are resource paths; the adapter base64url-encodes them. +const OPERATION_NAME = + "projects/proj-1/locations/us-central1/publishers/google/models/veo-3.1-generate-preview/operations/op-abc"; +const JOB_ID = Buffer.from(OPERATION_NAME, "utf8").toString("base64url"); + +describe("registry wiring", () => { + it("exposes videoConfig + video serviceKind for openrouter and vertex", () => { + expect(getVideoConfig("openrouter").baseUrl).toBe("https://openrouter.ai/api/v1/videos"); + expect(getVideoConfig("vertex").baseUrl).toBe("https://aiplatform.googleapis.com"); + expect(PROVIDER_MEDIA.openrouter.serviceKinds).toContain("video"); + expect(PROVIDER_MEDIA.vertex.serviceKinds).toContain("video"); + }); + + it("registers video-kind models on both providers", () => { + const or = PROVIDER_MODELS.openrouter.find((m) => m.id === "google/veo-3.1"); + const vx = PROVIDER_MODELS.vertex.find((m) => m.id === "veo-3.1-generate-preview"); + expect(or?.kind).toBe("video"); + expect(vx?.kind).toBe("video"); + }); +}); + +describe("openrouter video adapter", () => { + beforeEach(() => { global.fetch = vi.fn(); }); + afterEach(() => { global.fetch = originalFetch; }); + + it("POSTs creation to the collection root (no /generations suffix)", async () => { + global.fetch.mockResolvedValueOnce(jsonResponse({ id: "job-1", status: "pending" })); + + const raw = '{"model":"google/veo-3.1","prompt":"a paper boat"}'; + const result = await handleVideoProxyCore({ + provider: "openrouter", + action: "generations", + rawBody: raw, + contentType: "application/json", + credentials: { apiKey: "sk-or-key" }, + }); + + expect(result.success).toBe(true); + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe("https://openrouter.ai/api/v1/videos"); + expect(init.method).toBe("POST"); + expect(init.body).toBe(raw); // verbatim + expect(init.headers.Authorization).toBe("Bearer sk-or-key"); + expect(init.headers["HTTP-Referer"]).toBe("https://endpoint-proxy.local"); + expect(await result.response.json()).toEqual({ id: "job-1", status: "pending" }); + }); + + it("polls GET /videos/{id} and passes the payload through verbatim", async () => { + const payload = { id: "job-1", status: "completed", unsigned_urls: ["https://cdn/v.mp4"] }; + global.fetch.mockResolvedValueOnce(jsonResponse(payload)); + + const result = await handleVideoProxyCore({ + provider: "openrouter", + requestId: "job-1", + credentials: { apiKey: "sk-or-key" }, + }); + + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe("https://openrouter.ai/api/v1/videos/job-1"); + expect(init.method).toBe("GET"); + expect(await result.response.json()).toEqual(payload); + }); + + it("rejects unsupported actions before any upstream call (no billable job)", async () => { + const result = await handleVideoProxyCore({ + provider: "openrouter", + action: "extensions", + rawBody: "{}", + contentType: "application/json", + credentials: { apiKey: "sk-or-key" }, + }); + + expect(result.success).toBe(false); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + }); +}); + +describe("vertex (veo) video adapter", () => { + beforeEach(() => { + global.fetch = vi.fn(); + refreshVertexToken.mockReset(); + }); + afterEach(() => { global.fetch = originalFetch; }); + + const saJson = JSON.stringify({ + type: "service_account", + client_email: "sa@proj-1.iam.gserviceaccount.com", + private_key: "-----BEGIN PRIVATE KEY-----\nx\n-----END PRIVATE KEY-----\n", + project_id: "proj-1", + }); + + it("translates the create body to predictLongRunning and returns a poll-able job id", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce(jsonResponse({ name: OPERATION_NAME })); + + const result = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ + model: "veo-3.1-generate-preview", + prompt: "a neon city", + duration: 8, + aspect_ratio: "16:9", + resolution: "720p", + n: 1, + }), + contentType: "application/json", + credentials: { apiKey: saJson }, + }); + + expect(result.success).toBe(true); + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe( + "https://aiplatform.googleapis.com/v1/projects/proj-1/locations/us-central1/publishers/google/models/veo-3.1-generate-preview:predictLongRunning" + ); + expect(init.headers.Authorization).toBe("Bearer vertex-tok"); + expect(JSON.parse(init.body)).toEqual({ + instances: [{ prompt: "a neon city" }], + parameters: { sampleCount: 1, durationSeconds: 8, aspectRatio: "16:9", resolution: "720p" }, + }); + + // Response is mapped onto the async-job shape clients already poll. + expect(await result.response.json()).toEqual({ + id: JOB_ID, + request_id: JOB_ID, + status: "pending", + }); + }); + + it("maps a data-URL image onto the Vertex image instance", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce(jsonResponse({ name: OPERATION_NAME })); + + await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ + model: "veo-3.1-generate-preview", + prompt: "animate this", + image: "data:image/png;base64,AAAB", + }), + contentType: "application/json", + credentials: { apiKey: saJson }, + }); + + expect(JSON.parse(global.fetch.mock.calls[0][1].body).instances[0].image).toEqual({ + bytesBase64Encoded: "AAAB", + mimeType: "image/png", + }); + }); + + it("polls via fetchPredictOperation and maps a completed operation", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce( + jsonResponse({ + name: OPERATION_NAME, + done: true, + response: { videos: [{ gcsUri: "gs://bucket/v.mp4", mimeType: "video/mp4" }] }, + }) + ); + + const result = await handleVideoProxyCore({ + provider: "vertex", + requestId: JOB_ID, + credentials: { apiKey: saJson }, + }); + + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe( + "https://aiplatform.googleapis.com/v1/projects/proj-1/locations/us-central1/publishers/google/models/veo-3.1-generate-preview:fetchPredictOperation" + ); + expect(init.method).toBe("POST"); // Vertex polls with POST, not GET + expect(JSON.parse(init.body)).toEqual({ operationName: OPERATION_NAME }); + + expect(await result.response.json()).toEqual({ + id: JOB_ID, + request_id: JOB_ID, + status: "completed", + video: { url: "gs://bucket/v.mp4", b64_json: null, mime_type: "video/mp4" }, + videos: [{ url: "gs://bucket/v.mp4", b64_json: null, mime_type: "video/mp4" }], + }); + }); + + it("maps a failed operation to status failed", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce( + jsonResponse({ name: OPERATION_NAME, done: true, error: { code: 3, message: "bad prompt" } }) + ); + + const result = await handleVideoProxyCore({ + provider: "vertex", + requestId: JOB_ID, + credentials: { apiKey: saJson }, + }); + + const body = await result.response.json(); + expect(body.status).toBe("failed"); + expect(body.error.message).toBe("bad prompt"); + }); + + it("rejects missing project id and raw API keys before any upstream call", async () => { + const noProject = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ model: "veo-3.1-generate-preview", prompt: "x" }), + contentType: "application/json", + credentials: { apiKey: "AIzaRawKey" }, + }); + expect(noProject.success).toBe(false); + expect(noProject.status).toBe(400); + + const noToken = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ model: "veo-3.1-generate-preview", prompt: "x" }), + contentType: "application/json", + credentials: { apiKey: "AIzaRawKey", providerSpecificData: { projectId: "proj-1" } }, + }); + expect(noToken.success).toBe(false); + expect(noToken.status).toBe(400); + + expect(global.fetch).not.toHaveBeenCalled(); + }); + + it("rejects an invalid job id without calling upstream", async () => { + refreshVertexToken.mockResolvedValue({ accessToken: "vertex-tok" }); + const result = await handleVideoProxyCore({ + provider: "vertex", + requestId: Buffer.from("not-an-operation", "utf8").toString("base64url"), + credentials: { apiKey: saJson }, + }); + expect(result.success).toBe(false); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + }); +}); diff --git a/tests/unit/xai-video-handler.test.js b/tests/unit/xai-video-handler.test.js index 563ee08c..a18411b4 100644 --- a/tests/unit/xai-video-handler.test.js +++ b/tests/unit/xai-video-handler.test.js @@ -29,6 +29,7 @@ vi.mock("@/sse/services/auth.js", () => authMocks); vi.mock("@/sse/services/tokenRefresh.js", () => tokenMocks); vi.mock("@/lib/localDb", () => ({ getSettings: vi.fn(async () => ({ requireApiKey: false })), + getProviderConnectionById: vi.fn(async () => ({ id: "conn-5", provider: "xai" })), getComboByName: vi.fn(async () => null), getModelAliases: vi.fn(async () => ({})), getProviderNodes: vi.fn(async () => []), From 832a34659e3ea4a97373bdbfef41462f855f5f68 Mon Sep 17 00:00:00 2001 From: Hai Trinh Date: Thu, 10 Sep 2026 22:06:17 +0700 Subject: [PATCH 32/78] feat(codex): add GPT Image 2.5, Flare and Sunburst image models - Add gpt-image-1.5, gpt-image-2, gpt-image-2.5, gpt-image-2.5-flare and gpt-image-2.5-sunburst as Codex image models with multi-image support - Add gpt-image-2.5, gpt-image-2.5-flare and gpt-image-2.5-sunburst to the OpenAI provider catalog - Route tool-backed image models through the Codex responses model while passing the selected model to the image_generation tool, pinning tool_choice and deriving generate/edit from the presence of references - Cover the Codex gpt-image-2.5 request shape with a unit test --- open-sse/handlers/imageProviders/codex.js | 26 ++++++++++++-- open-sse/providers/registry/codex.js | 5 +++ open-sse/providers/registry/openai.js | 3 ++ tests/unit/image-generation.test.js | 41 +++++++++++++++++++++++ 4 files changed, 72 insertions(+), 3 deletions(-) diff --git a/open-sse/handlers/imageProviders/codex.js b/open-sse/handlers/imageProviders/codex.js index 385ec5b6..afaf8e97 100644 --- a/open-sse/handlers/imageProviders/codex.js +++ b/open-sse/handlers/imageProviders/codex.js @@ -9,6 +9,14 @@ const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`; const CODEX_ORIGINATOR = "codex_cli_rs"; const CODEX_MODEL_SUFFIX = "-image"; const CODEX_REF_DETAIL = "high"; +const CODEX_IMAGES_MAIN_MODEL = "gpt-5.5"; +const CODEX_TOOL_IMAGE_MODELS = new Set([ + "gpt-image-1.5", + "gpt-image-2", + "gpt-image-2.5", + "gpt-image-2.5-flare", + "gpt-image-2.5-sunburst", +]); function decodeAccountId(idToken) { try { @@ -27,6 +35,13 @@ function stripImageSuffix(model) { return model.endsWith(CODEX_MODEL_SUFFIX) ? model.slice(0, -CODEX_MODEL_SUFFIX.length) : model; } +function resolveCodexImageModels(model) { + if (CODEX_TOOL_IMAGE_MODELS.has(model)) { + return { responsesModel: CODEX_IMAGES_MAIN_MODEL, toolModel: model }; + } + return { responsesModel: stripImageSuffix(model), toolModel: null }; +} + function toDataUrl(input) { if (!input || typeof input !== "string") return null; if (/^data:image\//i.test(input) || /^https?:\/\//i.test(input)) return input; @@ -167,21 +182,26 @@ export default { const single = toDataUrl(body.image); if (single) refs.push(single); const detail = body.image_detail || CODEX_REF_DETAIL; + const { responsesModel, toolModel } = resolveCodexImageModels(model); const imgTool = { type: "image_generation", output_format: (body.output_format || "png").toLowerCase() }; + if (toolModel) { + imgTool.action = refs.length > 0 ? "edit" : "generate"; + imgTool.model = toolModel; + } if (body.size && body.size !== "") imgTool.size = body.size; if (body.quality && body.quality !== "") imgTool.quality = body.quality; if (body.background && body.background !== "") imgTool.background = body.background; return { - model: stripImageSuffix(model), + model: responsesModel, instructions: "", input: [{ type: "message", role: "user", content: buildContent(body.prompt, refs, detail) }], tools: [imgTool], - tool_choice: "auto", + tool_choice: toolModel ? { type: "image_generation" } : "auto", parallel_tool_calls: false, prompt_cache_key: randomUUID(), stream: true, store: false, - reasoning: null, + reasoning: toolModel ? { effort: "medium", summary: "auto" } : null, }; }, // Custom: codex parses SSE → either pipe to client or collect b64 diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index bf436a10..9eaafe74 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -65,6 +65,11 @@ export default { { id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" }, { id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" }, { id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" }, + { id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-2", name: "GPT Image 2", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-1.5", name: "GPT Image 1.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, diff --git a/open-sse/providers/registry/openai.js b/open-sse/providers/registry/openai.js index 9a1ca57b..4a5f0eb1 100644 --- a/open-sse/providers/registry/openai.js +++ b/open-sse/providers/registry/openai.js @@ -57,6 +57,9 @@ export default { { id: "whisper-1", name: "Whisper 1", params: ["language","response_format","temperature","prompt"], kind: "stt" }, { id: "gpt-4o-transcribe", name: "GPT-4o Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" }, { id: "gpt-4o-mini-transcribe", name: "GPT-4o Mini Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" }, + { id: "gpt-image-2.5", name: "GPT Image 2.5", params: ["n","size","quality","response_format"], kind: "image" }, + { id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", params: ["n","size","quality","response_format"], kind: "image" }, + { id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", params: ["n","size","quality","response_format"], kind: "image" }, { id: "gpt-image-1", name: "GPT Image 1", params: ["n","size","quality","response_format"], kind: "image" }, { id: "dall-e-3", name: "DALL-E 3", params: ["size","quality","style","response_format"], kind: "image" }, { id: "dall-e-2", name: "DALL-E 2", params: ["n","size","response_format"], kind: "image" }, diff --git a/tests/unit/image-generation.test.js b/tests/unit/image-generation.test.js index 4df94afc..6a22cd40 100644 --- a/tests/unit/image-generation.test.js +++ b/tests/unit/image-generation.test.js @@ -367,6 +367,47 @@ describe("handleImageGenerationCore", () => { expect(responseBody.data[0].b64_json).toBe("base64codeximage"); }); + it("generates image with Codex gpt-image-2.5 tool model", async () => { + global.fetch.mockResolvedValueOnce( + new Response( + [ + "event: response.output_item.done", + 'data: {"item":{"type":"image_generation_call","result":"base64codeximage"}}', + "", + "", + ].join("\n"), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ) + ); + + const result = await handleImageGenerationCore({ + body: { + prompt: "A futuristic city", + size: "1024x1024", + output_format: "png", + }, + modelInfo: { provider: "codex", model: "gpt-image-2.5" }, + credentials: { + accessToken: "codex-token", + providerSpecificData: { chatgptAccountId: "account-123" }, + }, + log: null, + }); + + expect(result.success).toBe(true); + const fetchCall = global.fetch.mock.calls[0]; + const requestBody = JSON.parse(fetchCall[1].body); + expect(requestBody.model).toBe("gpt-5.5"); + expect(requestBody.tools).toEqual([ + { type: "image_generation", output_format: "png", size: "1024x1024", action: "generate", model: "gpt-image-2.5" }, + ]); + expect(requestBody.tool_choice).toEqual({ type: "image_generation" }); + expect(requestBody.reasoning).toEqual({ effort: "medium", summary: "auto" }); + + const responseBody = await result.response.json(); + expect(responseBody.data[0].b64_json).toBe("base64codeximage"); + }); + it("generates image with Cloudflare Workers AI JSON response", async () => { global.fetch.mockResolvedValueOnce( new Response( From 1f10f9e5c4603673aae6720bf91f8c62aedf6730 Mon Sep 17 00:00:00 2001 From: LLL <2798142644@qq.com> Date: Thu, 10 Sep 2026 22:06:49 +0700 Subject: [PATCH 33/78] fix(qoder): report usage to all clients and stop inlining large attachments - Coalesce Qoder's empty finish-in-delta frame with the later choices:[] usage frame so OpenAI and Claude clients receive prompt_tokens, completion_tokens and cache-hit tokens (the dashboard already saw them) - Upload inlined images through /api/v2/image/upload like qodercli, and stub oversized non-image files instead of stuffing 30MB+ data URIs into agent_chat_generation - Emit response.completed -> response.usage for chat-native upstreams so /v1/responses clients (Codex CLI, sub2api) no longer log 0/0/0 - Keep Claude message_delta.usage working when usage arrives without choices[0] - Escalate to the smallest advertised Qoder context tier (200K/400K/1M) when the estimated prompt no longer fits max_input_tokens - Pass apiKey for PAT connections and list hidden enable:false catalog keys from /v1/models --- open-sse/executors/qoder.js | 133 +++++-- .../handlers/chatCore/nonStreamingHandler.js | 8 +- .../handlers/chatCore/sseToJsonHandler.js | 8 +- open-sse/providers/capabilities.js | 6 +- open-sse/services/qoderModels.js | 24 ++ open-sse/shared/qoder/attachments.js | 341 ++++++++++++++++++ open-sse/shared/qoder/constants.js | 33 ++ open-sse/shared/qoder/contextTier.js | 160 ++++++++ open-sse/shared/qoder/sse.js | 208 +++++++++++ open-sse/transformer/responsesTransformer.js | 27 +- open-sse/translator/concerns/usage.js | 31 ++ .../translator/response/openai-responses.js | 34 +- .../translator/response/openai-to-claude.js | 81 +++-- open-sse/utils/stream.js | 17 +- src/app/api/v1/models/route.js | 13 +- tests/unit/openai-responses-usage.test.js | 182 ++++++++++ tests/unit/openai-to-claude.test.js | 27 ++ tests/unit/qoder-context-tier.test.js | 193 ++++++++++ tests/unit/qoder.test.js | 210 ++++++++++- 19 files changed, 1615 insertions(+), 121 deletions(-) create mode 100644 open-sse/shared/qoder/attachments.js create mode 100644 open-sse/shared/qoder/contextTier.js create mode 100644 open-sse/shared/qoder/sse.js create mode 100644 tests/unit/openai-responses-usage.test.js create mode 100644 tests/unit/qoder-context-tier.test.js diff --git a/open-sse/executors/qoder.js b/open-sse/executors/qoder.js index 86055aba..8ee21ac5 100644 --- a/open-sse/executors/qoder.js +++ b/open-sse/executors/qoder.js @@ -31,14 +31,16 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { SSE_DONE } from "../utils/sseConstants.js"; import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js"; import { - QODER_CHAT_URL_ENCODED, - QODER_CHAT_BASE_ALT, QODER_CHAT_SIG_PATH, - QODER_MODEL_MAP, + QODER_CONTEXT_TIER_ENV, + qoderInferenceBase, } from "../shared/qoder/constants.js"; import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js"; import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js"; import { encodeDataUri } from "../translator/concerns/image.js"; +import { createQoderSseCoalescer } from "../shared/qoder/sse.js"; +import { rewriteQoderMessageAttachments } from "../shared/qoder/attachments.js"; +import { resolveQoderContextTier, applyQoderContextTier } from "../shared/qoder/contextTier.js"; /** * Hoist role:"system" messages out of the messages array (Qoder rejects @@ -70,15 +72,16 @@ function normalizeMessages(messages) { * * Text-only content is flattened to a plain string (Qoder's historical * shape). When images are present the content stays an array and image - * blocks are kept as OpenAI-style `image_url` parts — verified against the - * upstream: it accepts both http(s) URLs and inline base64 data: URIs - * directly, no pre-upload to the /image/upload OSS flow required (that is - * a qodercli client-side choice, not a protocol requirement). The legacy + * blocks are kept as OpenAI-style `image_url` parts. Native qodercli + * uploads inlined bytes to `/api/v2/image/upload` first and then sends + * the OSS URL — `buildQoderRequestBody` does that rewrite before this + * runs. Tiny leftover data URIs are still accepted. The legacy * top-level `image_urls` / `chat_context.imageUrls` slots stay null — * qodercli leaves them null too. * * Claude-style `{type:"image", source:{...}}` blocks are converted to - * `image_url` so claude-format clients also round-trip. + * `image_url`. File/document blocks that survived rewrite become short + * stubs so 30MB PDFs never land in agent_chat_generation. */ function normalizeContent(content) { if (typeof content === "string") return content; @@ -88,10 +91,24 @@ function normalizeContent(content) { const blocks = []; const textParts = []; let hasImage = false; + + const pushText = (text) => { + if (!text) return; + if (hasImage || blocks.length) blocks.push({ type: OPENAI_BLOCK.TEXT, text }); + else textParts.push(text); + }; + + const imageUrlOf = (item) => { + if (typeof item.image_url === "string" && item.image_url) return item.image_url; + if (typeof item.image_url?.url === "string" && item.image_url.url) return item.image_url.url; + return null; + }; + for (const item of content) { if (!item || typeof item !== "object") continue; - if (item.type === OPENAI_BLOCK.IMAGE_URL && typeof item.image_url?.url === "string" && item.image_url.url) { - blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: item.image_url.url } }); + const imageUrl = item.type === OPENAI_BLOCK.IMAGE_URL ? imageUrlOf(item) : null; + if (imageUrl) { + blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: imageUrl } }); hasImage = true; } else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) { // Claude base64/url image → OpenAI image_url equivalent. @@ -103,13 +120,14 @@ function normalizeContent(content) { blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } }); hasImage = true; } + } else if (item.type === OPENAI_BLOCK.FILE) { + const name = item.file?.filename || item.file?.name || "file"; + pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`); + } else if (item.type === CLAUDE_BLOCK.DOCUMENT) { + const name = item.title || "document"; + pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`); } else if (typeof item.text === "string" && item.text) { - if (hasImage || blocks.length) { - // Keep ordering faithful once images are in play. - blocks.push({ type: OPENAI_BLOCK.TEXT, text: item.text }); - } else { - textParts.push(item.text); - } + pushText(item.text); } } @@ -189,7 +207,7 @@ function truncate(s, n) { /** * Map the OpenAI-style request body into the exact shape Qoder expects. */ -async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }) { +async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null }) { const qoderKey = String(model || "").replace(/^qoder\//, ""); // Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP. @@ -208,7 +226,30 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio modelConfig = { ...retried, key: qoderKey }; } - const { messages, systemText } = normalizeMessages(body.messages || []); + const incoming = Array.isArray(body.messages) + ? body.messages.map((m) => { + if (!m || typeof m !== "object") return m; + return { + ...m, + content: Array.isArray(m.content) + ? m.content.map((b) => (b && typeof b === "object" ? { ...b } : b)) + : m.content, + }; + }) + : []; + try { + await rewriteQoderMessageAttachments(incoming, { + credentials, + log, + proxyOptions, + signal, + uploadFn, + }); + } catch (err) { + log?.warn?.("QODER", `attachment rewrite failed: ${err.message}`); + } + + const { messages, systemText } = normalizeMessages(incoming); const tools = body.tools; const isReasoning = !!modelConfig.is_reasoning; const maxOutputTokens = Number(modelConfig.max_output_tokens) || 0; @@ -227,7 +268,21 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio const sessionId = stableHash("qoder-session", psd.userId, qoderKey); const recordId = stableChatRecordId(qoderKey, messages, tools, maxTokens); - return { + // Context-window tier (200K/400K/1M): the IDE picks one from model_config.context_config; + // qodercli-style requests default to the smallest. Escalate when the prompt no longer fits. + const tierChoice = resolveQoderContextTier( + modelConfig, + { system: systemText, messages, tools }, + { preference: process.env[QODER_CONTEXT_TIER_ENV] }, + ); + if (tierChoice) { + log?.info?.( + "QODER", + `context tier ${tierChoice.tier.name} (${tierChoice.tier.tokenCount} tokens, ${tierChoice.reason}) for ~${tierChoice.estimatedTokens} prompt tokens`, + ); + } + + const built = { qoderKey, payload: { request_id: uuidv4(), @@ -275,6 +330,8 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio }, modelConfig, }; + if (tierChoice) applyQoderContextTier(built.payload, tierChoice.tier); + return built; } /** @@ -338,6 +395,11 @@ async function peekFirstQoderFrame(reader, decoder) { * response.text() which hangs until the socket closes — so on terminal * events we cancel the upstream reader and close our stream immediately. * + * Usage: Qoder puts finish_reason on `delta` and sends token counts on a + * later `choices: []` frame. Downstream OpenAI/Claude clients only read + * usage from the finish chunk, so we coalesce those two frames (see + * createQoderSseCoalescer) before forwarding. + * * NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl). * If detected, return 403 response so chatCore marks connection unavailable * and triggers combo fallback instead of leaking error text into chat. @@ -364,6 +426,11 @@ async function wrapQoderSSE(response, model) { const upstreamDrained = peek.upstreamDone === true; const encoder = new TextEncoder(); let doneEmitted = false; + const coalescer = createQoderSseCoalescer({ model, encoder, sseDone: SSE_DONE }); + + const syncDone = () => { + if (coalescer.doneEmitted) doneEmitted = true; + }; // Process one already-extracted SSE line (no trailing newline). const processLine = (line, controller) => { @@ -374,15 +441,17 @@ async function wrapQoderSSE(response, model) { const data = trimmed.slice(5).trimStart(); if (data === "[DONE]") { - controller.enqueue(encoder.encode(SSE_DONE)); - doneEmitted = true; + coalescer.flush(controller); + syncDone(); return; } let envelope; try { envelope = JSON.parse(data); } catch { return; } const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200; - const inner = typeof envelope.body === "string" ? envelope.body : ""; + const inner = typeof envelope.body === "string" + ? envelope.body + : envelope.body != null ? JSON.stringify(envelope.body) : ""; if (statusVal !== 200) { const msg = inner || `upstream status ${statusVal}`; const errChunk = JSON.stringify({ @@ -398,14 +467,8 @@ async function wrapQoderSSE(response, model) { return; } if (!inner) return; - if (inner === "[DONE]") { - controller.enqueue(encoder.encode(SSE_DONE)); - doneEmitted = true; - return; - } - // Strip embedded newlines so the SSE frame stays a single event. - const sanitized = inner.replace(/\r?\n/g, ""); - controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`)); + coalescer.handleInner(inner, controller); + syncDone(); }; const stream = new ReadableStream({ @@ -464,7 +527,7 @@ async function wrapQoderSSE(response, model) { } finally { if (!doneEmitted) { try { - controller.enqueue(encoder.encode(SSE_DONE)); + coalescer.flush(controller); doneEmitted = true; } catch { /* already closed */ } } @@ -493,13 +556,7 @@ export class QoderExecutor extends BaseExecutor { } buildUrl(credentials) { - // Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt- - // with "Login expired" (403). Device tokens (dt-...) stay on api3. - const raw = credentials?.apiKey || credentials?.accessToken; - if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) { - return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`; - } - return QODER_CHAT_URL_ENCODED; + return `${qoderInferenceBase(credentials)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`; } // Override execute entirely — Qoder needs: diff --git a/open-sse/handlers/chatCore/nonStreamingHandler.js b/open-sse/handlers/chatCore/nonStreamingHandler.js index becc0842..3b6cfc6d 100644 --- a/open-sse/handlers/chatCore/nonStreamingHandler.js +++ b/open-sse/handlers/chatCore/nonStreamingHandler.js @@ -10,6 +10,7 @@ import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, sav import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { decloakToolNames } from "../../utils/claudeCloaking.js"; import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js"; +import { toResponsesUsage } from "../../translator/concerns/usage.js"; function parseToolArguments(value) { if (!value) return {}; @@ -130,11 +131,8 @@ function openAICompletionToResponses(responseBody, customToolNames = null) { background: false, error: null, output, - usage: { - input_tokens: usage.prompt_tokens || usage.input_tokens || 0, - output_tokens: usage.completion_tokens || usage.output_tokens || 0, - total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0), - }, + // Keep cached/reasoning details (input_tokens_details) — proxies bill cache hits from them + usage: toResponsesUsage(usage) || { input_tokens: 0, output_tokens: 0, total_tokens: 0 }, }; } diff --git a/open-sse/handlers/chatCore/sseToJsonHandler.js b/open-sse/handlers/chatCore/sseToJsonHandler.js index 6801be89..0da5c69f 100644 --- a/open-sse/handlers/chatCore/sseToJsonHandler.js +++ b/open-sse/handlers/chatCore/sseToJsonHandler.js @@ -5,6 +5,7 @@ import { FORMATS } from "../../translator/formats.js"; import { PROVIDERS } from "../../config/providers.js"; import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js"; +import { toResponsesUsage } from "../../translator/concerns/usage.js"; // Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES; @@ -97,11 +98,8 @@ function chatCompletionToResponses(responseBody, customToolNames = null) { background: false, error: null, output, - usage: { - input_tokens: usage.prompt_tokens || usage.input_tokens || 0, - output_tokens: usage.completion_tokens || usage.output_tokens || 0, - total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0), - }, + // Keep cached/reasoning details (input_tokens_details) — proxies bill cache hits from them + usage: toResponsesUsage(usage) || { input_tokens: 0, output_tokens: 0, total_tokens: 0 }, }; } diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index dd9fa381..e011b788 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -222,9 +222,9 @@ export const PROVIDER_CAPABILITIES = { // windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more). // max_output_tokens arrives as 0 for every model, so outputs are // best-guess from the real model family. Vision tags below follow the - // upstream is_vl flag per explicit request, even though the executor - // currently sends image_urls:null (image pass-through over the agent_chat - // SSE protocol is unverified). reasoning:true on all of them — every model can + // upstream is_vl flag. The executor uploads inlined images to + // /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null + // (same as qodercli). reasoning:true on all of them — every model can // reason; the upstream is_reasoning flag only drives model_config selection. // thinkingFormat keeps the true-model family for documentation/UI, but // thinkingCanDisable:false everywhere: the executor only forwards diff --git a/open-sse/services/qoderModels.js b/open-sse/services/qoderModels.js index 572931e5..e9a7879b 100644 --- a/open-sse/services/qoderModels.js +++ b/open-sse/services/qoderModels.js @@ -343,6 +343,30 @@ export async function resolveQoderModels(credentials, options = {}) { } } +/** + * Every model key the chat endpoint accepts for this credential: the IDE-visible + * models first, then catalog entries flagged `enable:false` (hidden in the IDE + * picker, e.g. by an account policy, but still served by agent_chat_generation — + * see fetchQoderCatalogRaw). /v1/models uses this so the advertised list matches + * what the router will actually route instead of collapsing to one or two keys. + */ +export function routableQoderModels(catalog) { + if (!catalog) return []; + const out = []; + const seen = new Set(); + for (const m of catalog.models || []) { + if (!m?.id || seen.has(m.id)) continue; + seen.add(m.id); + out.push({ id: m.id, name: m.name || m.id, hidden: false }); + } + for (const [key, cfg] of catalog.rawConfigs || []) { + if (!key || seen.has(key)) continue; + seen.add(key); + out.push({ id: key, name: cfg?.display_name || key, hidden: true }); + } + return out; +} + export function invalidateQoderCatalog(credentials) { if (!credentials) return; catalogCache.delete(cacheKey(credentials)); diff --git a/open-sse/shared/qoder/attachments.js b/open-sse/shared/qoder/attachments.js new file mode 100644 index 00000000..d2024529 --- /dev/null +++ b/open-sse/shared/qoder/attachments.js @@ -0,0 +1,341 @@ +/** + * Native qodercli does NOT stuff image/PDF bytes into agent_chat_generation. + * It PUTs them to /algo/api/v2/image/upload (COSY-signed multipart) and then + * sends the returned OSS URL. Agents like Claude Code send OpenAI/Claude + * data-URIs instead, which 9router previously forwarded verbatim — 10MB + * images become 30MB+ JSON and upstream 413s even though the model window + * is ~200k tokens. + * + * This module: + * 1. Uploads inlined images to Qoder's file API (cached by sha256). + * 2. Replaces huge non-image file blocks with a short stub. + * 3. Caps leftover data-URIs so the chat JSON stays small. + */ + +import { createHash } from "crypto"; +import { v4 as uuidv4 } from "uuid"; + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { parseDataUri } from "../../translator/concerns/image.js"; +import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../../translator/schema/blocks.js"; +import { MAX_IMAGE_BYTES } from "../../config/mediaConfig.js"; +import { buildCosyHeaders } from "./cosy.js"; +import { + QODER_IMAGE_UPLOAD_SIG_PATH, + QODER_INLINE_FALLBACK_MAX_BYTES, + QODER_MAX_PAYLOAD_BYTES, + qoderInferenceBase, +} from "./constants.js"; + +const IMAGE_MIME_RE = /^image\//i; +const DATA_URI_RE = /data:[^;]+;base64,[A-Za-z0-9+/=\s]+/g; + +function mimeExt(mime) { + const m = String(mime || "").toLowerCase(); + if (m.includes("png")) return "png"; + if (m.includes("jpeg") || m.includes("jpg")) return "jpg"; + if (m.includes("gif")) return "gif"; + if (m.includes("webp")) return "webp"; + if (m.includes("bmp")) return "bmp"; + if (m.includes("pdf")) return "pdf"; + return "bin"; +} + +function decodedBytes(b64) { + if (typeof b64 !== "string" || !b64) return 0; + const compact = b64.replace(/\s/g, ""); + return Math.floor(compact.length * 3 / 4); +} + +function stubText({ name, mime, bytes, reason }) { + const label = name || mime || "attachment"; + const size = bytes ? `, ${bytes} bytes` : ""; + return `[file omitted: ${label}${size} — ${reason}]`; +} + +export function buildMultipartFile(buffer, { fieldName = "file", fileName, mediaType } = {}) { + const boundary = `----9routerQoder${Date.now().toString(16)}${Math.random().toString(16).slice(2)}`; + const filename = fileName || `upload.${mimeExt(mediaType)}`; + const head = Buffer.from( + `--${boundary}\r\nContent-Disposition: form-data; name="${fieldName}"; filename="${filename}"\r\nContent-Type: ${mediaType || "application/octet-stream"}\r\n\r\n`, + ); + const tail = Buffer.from(`\r\n--${boundary}--\r\n`); + const body = Buffer.concat([head, buffer, tail]); + return { boundary, body }; +} + +function extractUrlFromUploadResponse(json) { + if (!json || typeof json !== "object") return null; + const result = json.result && typeof json.result === "object" ? json.result : json; + const arrays = [result.imageUrls, result.image_urls, json.imageUrls, json.image_urls]; + for (const arr of arrays) { + if (Array.isArray(arr) && typeof arr[0] === "string" && arr[0]) return arr[0]; + } + const keys = ["imageUrl", "image_url", "url", "ossUrl", "oss_url", "originalUrl", "originUrl", "link", "image"]; + for (const key of keys) { + const v = result[key] ?? json[key]; + if (typeof v === "string" && v) return v; + } + if (typeof json.body === "string") { + try { return extractUrlFromUploadResponse(JSON.parse(json.body)); } catch { /* ignore */ } + } + return null; +} + +async function defaultUploadImage({ buffer, mediaType, credentials, proxyOptions, signal }) { + const requestId = uuidv4(); + const url = `${qoderInferenceBase(credentials)}${`/algo${QODER_IMAGE_UPLOAD_SIG_PATH}`}?request_id=${requestId}`; + const { boundary, body } = buildMultipartFile(buffer, { + fileName: `image.${mimeExt(mediaType)}`, + mediaType: mediaType || "application/octet-stream", + }); + const psd = credentials?.providerSpecificData || {}; + const cosyHeaders = buildCosyHeaders(body, url, { + userId: psd.userId, + authToken: credentials.accessToken, + name: credentials.displayName || "", + email: credentials.email || "", + machineId: psd.machineId || "", + }); + const headers = { + ...cosyHeaders, + Accept: "application/json", + "Content-Type": `multipart/form-data; boundary=${boundary}`, + "Content-Length": String(body.length), + "AI-CLIENT-TIMESTAMP": String(Math.floor(Date.now() / 1000)), + "Accept-Encoding": "identity", + }; + const res = await proxyAwareFetch( + url, + { method: "PUT", headers, body, signal }, + proxyOptions, + ); + if (!res.ok) { + const text = await res.text().catch(() => ""); + throw new Error(`HTTP ${res.status}${text ? `: ${text.slice(0, 180)}` : ""}`); + } + const json = await res.json().catch(() => null); + const uploaded = extractUrlFromUploadResponse(json); + if (!uploaded) throw new Error("upload response missing url"); + return uploaded; +} + +async function uploadImageData({ base64, mediaType, credentials, proxyOptions, signal, log, uploadFn, cache }) { + const compact = String(base64 || "").replace(/\s/g, ""); + if (!compact) return null; + const bytes = decodedBytes(compact); + if (bytes > MAX_IMAGE_BYTES) { + log?.warn?.("QODER", `image ${bytes} bytes exceeds upload cap, stubbing`); + return { stub: true, bytes, mime: mediaType }; + } + let buffer; + try { + buffer = Buffer.from(compact, "base64"); + } catch { + return { stub: true, bytes, mime: mediaType }; + } + const digest = createHash("sha256").update(buffer).digest("hex"); + if (cache?.has(digest)) return { url: cache.get(digest), bytes, mime: mediaType }; + + const doUpload = uploadFn || defaultUploadImage; + try { + const url = await doUpload({ buffer, mediaType, credentials, proxyOptions, signal }); + if (typeof url === "string" && url) { + cache?.set(digest, url); + return { url, bytes, mime: mediaType }; + } + } catch (err) { + log?.warn?.("QODER", `image upload failed (${err.message}); ${bytes <= QODER_INLINE_FALLBACK_MAX_BYTES ? "keeping inline" : "stubbing"}`); + } + if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return { keep: true, bytes, mime: mediaType }; + return { stub: true, bytes, mime: mediaType }; +} + +function imageUrlBlock(url) { + return { type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } }; +} + +async function rewriteBlock(block, ctx) { + if (!block || typeof block !== "object") return block; + + if (block.type === OPENAI_BLOCK.IMAGE_URL) { + const raw = typeof block.image_url === "string" ? block.image_url : block.image_url?.url; + if (typeof raw !== "string" || !raw) return null; + if (raw.startsWith("http://") || raw.startsWith("https://")) return imageUrlBlock(raw); + const parsed = parseDataUri(raw); + if (!parsed) return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "attachment", reason: "unreadable data URI" }) }; + if (!IMAGE_MIME_RE.test(parsed.mimeType)) { + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "file", mime: parsed.mimeType, bytes: decodedBytes(parsed.base64), reason: "non-image bytes are not inlined into Qoder context" }) }; + } + const up = await uploadImageData({ ...ctx, base64: parsed.base64, mediaType: parsed.mimeType }); + if (up?.url) return imageUrlBlock(up.url); + if (up?.keep) return imageUrlBlock(raw); + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", mime: parsed.mimeType, bytes: up?.bytes, reason: "upload failed; not inlined" }) }; + } + + if (block.type === OPENAI_BLOCK.IMAGE || block.type === CLAUDE_BLOCK.IMAGE) { + const src = block.source || {}; + if (src.type === "url" && typeof src.url === "string") return imageUrlBlock(src.url); + if (src.type === "base64" && src.data) { + const mime = src.media_type || "image/png"; + const up = await uploadImageData({ ...ctx, base64: src.data, mediaType: mime }); + if (up?.url) return imageUrlBlock(up.url); + if (up?.keep) return imageUrlBlock(`data:${mime};base64,${src.data}`); + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", mime, bytes: up?.bytes, reason: "upload failed; not inlined" }) }; + } + } + + if (block.type === OPENAI_BLOCK.FILE && block.file) { + const file = block.file; + const name = file.filename || file.name || "file"; + const dataUri = typeof file.file_data === "string" ? file.file_data : null; + const parsed = dataUri ? parseDataUri(dataUri) : null; + const b64 = parsed?.base64 || (typeof file.file_data === "string" && !file.file_data.startsWith("data:") ? file.file_data : null); + const mime = parsed?.mimeType || file.format || "application/octet-stream"; + if (b64 && IMAGE_MIME_RE.test(mime)) { + const up = await uploadImageData({ ...ctx, base64: b64, mediaType: mime }); + if (up?.url) return imageUrlBlock(up.url); + } + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name, mime, bytes: decodedBytes(b64 || ""), reason: "Qoder reads documents via its file API, not inlined bytes" }) }; + } + + if (block.type === CLAUDE_BLOCK.DOCUMENT && block.source) { + const src = block.source; + const name = block.title || "document"; + if (src.type === "base64" && src.data) { + const mime = src.media_type || "application/pdf"; + if (IMAGE_MIME_RE.test(mime)) { + const up = await uploadImageData({ ...ctx, base64: src.data, mediaType: mime }); + if (up?.url) return imageUrlBlock(up.url); + } + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name, mime, bytes: decodedBytes(src.data), reason: "Qoder reads documents via its file API, not inlined bytes" }) }; + } + } + + if (typeof block.text === "string" && block.text.includes("data:") && block.text.length > 8192) { + const next = block.text.replace(DATA_URI_RE, (m) => { + const parsed = parseDataUri(m.trim()); + const bytes = parsed ? decodedBytes(parsed.base64) : m.length; + if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return m; + return stubText({ mime: parsed?.mimeType, bytes, reason: "inlined data URI stripped from Qoder context" }); + }); + return { ...block, text: next }; + } + + return block; +} + +async function rewriteContent(content, ctx) { + if (typeof content === "string") { + if (content.includes("data:") && content.length > 8192) { + return content.replace(DATA_URI_RE, (m) => { + const parsed = parseDataUri(m.trim()); + const bytes = parsed ? decodedBytes(parsed.base64) : m.length; + if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return m; + return stubText({ mime: parsed?.mimeType, bytes, reason: "inlined data URI stripped from Qoder context" }); + }); + } + return content; + } + if (!Array.isArray(content)) return content; + const out = []; + for (const block of content) { + const next = await rewriteBlock(block, ctx); + if (next == null) continue; + out.push(next); + } + return out.length ? out : ""; +} + +function payloadBytes(messages) { + try { + return Buffer.byteLength(JSON.stringify(messages), "utf8"); + } catch { + return 0; + } +} + +function stripRemainingDataUris(messages) { + for (const msg of messages || []) { + if (typeof msg?.content === "string" && msg.content.includes("data:")) { + msg.content = msg.content.replace(DATA_URI_RE, (m) => + stubText({ bytes: m.length, reason: "payload over Qoder size budget" }), + ); + } else if (Array.isArray(msg?.content)) { + msg.content = msg.content.map((block) => { + if (block?.type === OPENAI_BLOCK.IMAGE_URL) { + const raw = typeof block.image_url === "string" ? block.image_url : block.image_url?.url; + if (typeof raw === "string" && raw.startsWith("data:")) { + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", reason: "payload over Qoder size budget" }) }; + } + } + if (typeof block?.text === "string" && block.text.includes("data:")) { + return { ...block, text: block.text.replace(DATA_URI_RE, (m) => + stubText({ bytes: m.length, reason: "payload over Qoder size budget" }), + ) }; + } + return block; + }); + } + } +} + +/** + * Rewrite OpenAI-shaped messages in place: upload images, stub huge files. + * @returns {Promise<{imageUrls: string[], uploaded: number, stubbed: number}>} + */ +export async function rewriteQoderMessageAttachments(messages, { + credentials, + log, + proxyOptions = null, + signal = null, + uploadFn = null, +} = {}) { + const stats = { imageUrls: [], uploaded: 0, stubbed: 0 }; + if (!Array.isArray(messages) || messages.length === 0) return stats; + + const ctx = { credentials, log, proxyOptions, signal, uploadFn, cache: new Map() }; + + for (const msg of messages) { + if (!msg || typeof msg !== "object") continue; + if (Array.isArray(msg.images)) { + // Ollama-style sidecar; fold into content so normalizeMessages can see them. + const extras = msg.images.map((url) => imageUrlBlock(String(url))); + msg.content = Array.isArray(msg.content) + ? [...msg.content, ...extras] + : [{ type: OPENAI_BLOCK.TEXT, text: typeof msg.content === "string" ? msg.content : "" }, ...extras]; + delete msg.images; + } + msg.content = await rewriteContent(msg.content, ctx); + } + + // Collect surviving http(s) image URLs for callers that want image_urls. + for (const msg of messages) { + if (!Array.isArray(msg?.content)) continue; + for (const block of msg.content) { + const url = block?.type === OPENAI_BLOCK.IMAGE_URL + ? (typeof block.image_url === "string" ? block.image_url : block.image_url?.url) + : null; + if (typeof url === "string" && /^https?:\/\//i.test(url)) stats.imageUrls.push(url); + if (block?.type === OPENAI_BLOCK.TEXT && typeof block.text === "string" && block.text.startsWith("[file omitted:")) stats.stubbed += 1; + } + } + stats.uploaded = stats.imageUrls.length; + + if (payloadBytes(messages) > QODER_MAX_PAYLOAD_BYTES) { + log?.warn?.("QODER", `request still ${payloadBytes(messages)} bytes after rewrite; stripping leftover data URIs`); + stripRemainingDataUris(messages); + } + + return stats; +} + +/** Test helper kept for callers; upload memo is now per-request. */ +export function clearQoderUploadCache() {} + +export const __test__ = { + extractUrlFromUploadResponse, + decodedBytes, + stubText, + payloadBytes, +}; diff --git a/open-sse/shared/qoder/constants.js b/open-sse/shared/qoder/constants.js index 861e8f30..849f67ed 100644 --- a/open-sse/shared/qoder/constants.js +++ b/open-sse/shared/qoder/constants.js @@ -33,6 +33,39 @@ export const QODER_CHAT_SIG_PATH = "/api/v2/service/pro/sse/agent_chat_generatio export const QODER_CHAT_URL = `${QODER_CHAT_BASE}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common`; export const QODER_CHAT_URL_ENCODED = `${QODER_CHAT_URL}&Encode=1`; export const QODER_MODEL_LIST_URL = `${QODER_CHAT_BASE}/algo/api/v2/model/list`; +// Official qodercli uploads images here (COSY-signed PUT multipart, field "file") +// instead of inlining base64 into agent_chat_generation. +export const QODER_IMAGE_UPLOAD_SIG_PATH = "/api/v2/image/upload"; + +// Drop remaining inlined binaries if the Qoder JSON body would still exceed this. +// 30MB+ payloads are what blow past Claude-Code's ~200k context on the wire. +export const QODER_MAX_PAYLOAD_BYTES = 6 * 1024 * 1024; +// If OSS upload fails, keep tiny data-URIs; anything larger is stubbed. +export const QODER_INLINE_FALLBACK_MAX_BYTES = 512 * 1024; + +// Context-window tier selection (see shared/qoder/contextTier.js). The IDE exposes the +// model's context_config tiers (200K/400K/1M); we auto-escalate when the estimated prompt +// (+ headroom, tokenizer variance) no longer fits the current max_input_tokens. +export const QODER_CONTEXT_TIER_HEADROOM = 0.15; +export const QODER_CONTEXT_TIER_ENV = "QODER_CONTEXT_TIER"; +export const QODER_CONTEXT_TIER_MODES = Object.freeze({ AUTO: "auto", MAX: "max", DEFAULT: "default" }); + +/** + * Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt- with + * "Login expired" (403). Device tokens (dt-...) stay on api3. PATs (pt-...) + * are exchanged for jt- before this is consulted. + */ +export function qoderInferenceBase(credentials) { + const raw = credentials?.apiKey || credentials?.accessToken; + if ( + typeof raw === "string" && + !raw.startsWith("pt-") && + (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-")) + ) { + return QODER_CHAT_BASE_ALT; + } + return QODER_CHAT_BASE; +} // COSY header constants. These are not arbitrary — the upstream signature // validation matches them against the values used at signing time. diff --git a/open-sse/shared/qoder/contextTier.js b/open-sse/shared/qoder/contextTier.js new file mode 100644 index 00000000..523c2fe5 --- /dev/null +++ b/open-sse/shared/qoder/contextTier.js @@ -0,0 +1,160 @@ +/** + * Qoder context-window tiers. + * + * Each Qoder model_config ships a `context_config` list (e.g. 200K / 400K / 1M for + * qmodel_38max) while `max_input_tokens` only carries the tier the IDE currently has + * selected (~180K by default). The Qoder IDE lets the user switch tiers from the model + * picker; a qodercli-style client (which is what 9router impersonates) has no picker, + * so a long Claude-Code / Codex session that grew past the default tier is rejected + * upstream even though the model itself supports 1M. + * + * This module emulates the IDE: estimate the prompt size, pick the smallest advertised + * tier that fits (never below the model's current default), and mirror the choice into + * the same three places the IDE writes: + * parameters.context_length + * chat_context.extra.ideModelConfigOverride.max_input_tokens + * model_config.max_input_tokens + * + * Override with QODER_CONTEXT_TIER = auto (default) | max | default | . + * Pure functions, no I/O — the executor wires them into buildQoderRequestBody. + */ + +import { QODER_CONTEXT_TIER_HEADROOM, QODER_CONTEXT_TIER_MODES } from "./constants.js"; + +const UNIT = { K: 1_000, M: 1_000_000 }; + +/** "200K" | "1M" | "204800" | 204800 → integer token count (0 when unparseable). */ +export function parseTierTokenCount(value) { + if (typeof value === "number") return Number.isFinite(value) && value > 0 ? Math.floor(value) : 0; + if (typeof value !== "string") return 0; + const m = value.trim().toUpperCase().match(/^(\d+(?:\.\d+)?)\s*([KM])?$/); + if (!m) return 0; + const n = Number(m[1]) * (UNIT[m[2]] || 1); + return Number.isFinite(n) && n > 0 ? Math.floor(n) : 0; +} + +function tierName(entry, tokenCount) { + const raw = entry.name ?? entry.label ?? entry.display_name ?? entry.displayName ?? entry.key ?? entry.id; + if (typeof raw === "string" && raw.trim()) return raw.trim(); + if (tokenCount >= UNIT.M && tokenCount % UNIT.M === 0) return `${tokenCount / UNIT.M}M`; + if (tokenCount >= UNIT.K && tokenCount % UNIT.K === 0) return `${tokenCount / UNIT.K}K`; + return String(tokenCount); +} + +/** + * Normalize a model_config into sorted tiers: [{ name, tokenCount, isDefault }] ascending. + * Accepts snake_case and camelCase shapes; returns [] when the model has no tiers. + */ +export function getQoderContextTiers(modelConfig) { + const list = modelConfig?.context_config ?? modelConfig?.contextConfig; + if (!Array.isArray(list)) return []; + const byCount = new Map(); + for (const entry of list) { + if (!entry || typeof entry !== "object") continue; + const tokenCount = parseTierTokenCount( + entry.tokenCount ?? entry.token_count ?? entry.max_input_tokens ?? entry.maxInputTokens ?? entry.contextLength ?? entry.context_length, + ); + if (!tokenCount) continue; + const isDefault = entry.isDefault === true || entry.is_default === true || entry.default === true; + const prev = byCount.get(tokenCount); + byCount.set(tokenCount, { + name: tierName(entry, tokenCount), + tokenCount, + isDefault: (prev?.isDefault || false) || isDefault, + }); + } + return [...byCount.values()].sort((a, b) => a.tokenCount - b.tokenCount); +} + +const CJK_RE = /[\u1100-\u11ff\u2e80-\u9fff\uac00-\ud7af\uf900-\ufaff\uff00-\uffef]/g; + +/** + * Rough prompt-size estimate in tokens. CJK characters count ~1 token each, everything + * else ~4 chars/token — the plain chars/4 rule underestimates Chinese/Japanese by up to + * 4x, which is exactly when a tier decision matters. + */ +export function estimateQoderPromptTokens({ system, messages, tools } = {}) { + let text = ""; + try { + text = JSON.stringify({ system: system || "", messages: messages || [], tools: tools || [] }) || ""; + } catch { + return 0; + } + const cjk = (text.match(CJK_RE) || []).length; + return Math.ceil(cjk + (text.length - cjk) / 4); +} + +function normalizeMode(preference) { + const p = String(preference ?? "").trim(); + return p ? p : QODER_CONTEXT_TIER_MODES.AUTO; +} + +function findNamedTier(tiers, name) { + const wanted = name.replace(/\s+/g, "").toUpperCase(); + const asCount = parseTierTokenCount(wanted); + return tiers.find((t) => t.name.replace(/\s+/g, "").toUpperCase() === wanted || (asCount && t.tokenCount === asCount)) || null; +} + +/** + * Decide which tier a request should run under. + * + * @param {object} modelConfig raw Qoder model_config (has context_config + max_input_tokens) + * @param {{system?: string, messages?: any[], tools?: any[]}} prompt what will be sent + * @param {{preference?: string, headroom?: number}} [options] + * @returns {{ tier: {name, tokenCount, isDefault}, estimatedTokens: number, reason: string } | null} + * null → leave the payload exactly as before (no tiers, or the default already fits). + */ +export function resolveQoderContextTier(modelConfig, prompt, options = {}) { + const tiers = getQoderContextTiers(modelConfig); + if (!tiers.length) return null; + + const mode = normalizeMode(options.preference); + const largest = tiers[tiers.length - 1]; + const defaultTier = tiers.find((t) => t.isDefault) || tiers[0]; + const estimatedTokens = estimateQoderPromptTokens(prompt); + const headroom = typeof options.headroom === "number" ? options.headroom : QODER_CONTEXT_TIER_HEADROOM; + const need = Math.ceil(estimatedTokens * (1 + headroom)); + + if (mode.toLowerCase() === QODER_CONTEXT_TIER_MODES.MAX) { + return { tier: largest, estimatedTokens, reason: "forced:max" }; + } + if (mode.toLowerCase() === QODER_CONTEXT_TIER_MODES.DEFAULT) { + return { tier: defaultTier, estimatedTokens, reason: "forced:default" }; + } + if (mode.toLowerCase() !== QODER_CONTEXT_TIER_MODES.AUTO) { + const named = findNamedTier(tiers, mode); + if (named) return { tier: named, estimatedTokens, reason: `forced:${named.name}` }; + // Unknown tier name → fall through to auto rather than silently breaking requests. + } + + // auto: keep the upstream default (current behaviour) while the prompt fits in it. + const currentMax = parseTierTokenCount(modelConfig?.max_input_tokens ?? modelConfig?.maxInputTokens); + const currentLimit = currentMax || defaultTier.tokenCount; + if (need <= currentLimit) return null; + + const fits = tiers.find((t) => t.tokenCount >= need && t.tokenCount > currentLimit); + const tier = fits || largest; + if (tier.tokenCount <= currentLimit) return null; // nothing bigger to escalate to + return { tier, estimatedTokens, reason: fits ? "auto:fits" : "auto:largest" }; +} + +/** + * Write the chosen tier into a Qoder chat payload (mutates + returns it). + * Mirrors the IDE: parameters.context_length, ideModelConfigOverride, model_config. + */ +export function applyQoderContextTier(payload, tier) { + if (!payload || !tier?.tokenCount) return payload; + payload.parameters = { ...(payload.parameters || {}), context_length: tier.tokenCount }; + payload.chat_context = payload.chat_context || {}; + payload.chat_context.extra = { + ...(payload.chat_context.extra || {}), + ideModelConfigOverride: { + ...(payload.chat_context.extra?.ideModelConfigOverride || {}), + max_input_tokens: tier.tokenCount, + }, + }; + if (payload.model_config && typeof payload.model_config === "object") { + payload.model_config = { ...payload.model_config, max_input_tokens: tier.tokenCount }; + } + return payload; +} diff --git a/open-sse/shared/qoder/sse.js b/open-sse/shared/qoder/sse.js new file mode 100644 index 00000000..ad30785e --- /dev/null +++ b/open-sse/shared/qoder/sse.js @@ -0,0 +1,208 @@ +/** + * Qoder SSE is OpenAI-shaped inside `{statusCodeValue, body}` envelopes, but + * usage arrives on a later `choices: []` frame — after finish_reason, which + * itself often lives on `delta.finish_reason` rather than the choice. + * + * Downstream (Claude translator, OpenAI clients, Claude Code) look for usage + * on the finish chunk or drop `choices: []` entirely. 9router's own dashboard + * still sees tokens because extractUsage runs on every forwarded frame. + * + * Coalesce: hold empty finish + usage-only frames, then emit one OpenAI + * include_usage-style chunk: `{choices:[{delta:{}, finish_reason}], usage}`. + */ + +function num(v) { + const n = Number(v); + return Number.isFinite(n) ? n : null; +} + +/** + * Normalize Qoder/OpenAI usage into the shape stream.js + Claude translation + * already understand (prompt_tokens + prompt_tokens_details.cached_tokens). + */ +export function canonicalizeQoderUsage(usage) { + if (!usage || typeof usage !== "object" || Array.isArray(usage)) return null; + + const prompt = num(usage.prompt_tokens ?? usage.input_tokens); + const completion = num(usage.completion_tokens ?? usage.output_tokens); + if (prompt == null && completion == null) return null; + + const details = (usage.prompt_tokens_details && typeof usage.prompt_tokens_details === "object") + ? { ...usage.prompt_tokens_details } + : {}; + const cached = num( + details.cached_tokens ?? + usage.cached_tokens ?? + usage.prompt_cache_hit_tokens ?? + usage.cache_read_input_tokens, + ); + const cacheCreation = num( + details.cache_creation_tokens ?? + usage.cache_creation_input_tokens, + ); + + const promptTokens = prompt || 0; + const completionTokens = completion || 0; + const out = { + prompt_tokens: promptTokens, + completion_tokens: completionTokens, + total_tokens: num(usage.total_tokens) ?? (promptTokens + completionTokens), + }; + + if (cached != null) { + out.cached_tokens = cached; + details.cached_tokens = cached; + } + if (cacheCreation != null) { + details.cache_creation_tokens = cacheCreation; + } + if (Object.keys(details).length) out.prompt_tokens_details = details; + + if (usage.completion_tokens_details && typeof usage.completion_tokens_details === "object") { + out.completion_tokens_details = usage.completion_tokens_details; + } + const reasoning = num(usage.reasoning_tokens ?? usage.completion_tokens_details?.reasoning_tokens); + if (reasoning != null) out.reasoning_tokens = reasoning; + + return out; +} + +function finishReasonOf(parsed) { + const choice = parsed?.choices?.[0]; + return choice?.finish_reason || choice?.delta?.finish_reason || parsed?.finish_reason || null; +} + +function hasValuableDelta(parsed) { + const delta = parsed?.choices?.[0]?.delta; + if (!delta || typeof delta !== "object") return false; + if (typeof delta.content === "string" && delta.content.length > 0) return true; + if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0) return true; + if (Array.isArray(delta.tool_calls) && delta.tool_calls.length > 0) return true; + if (delta.role) return true; + return false; +} + +function parseInner(inner) { + if (inner == null || inner === "") return { raw: false, parsed: null }; + if (inner === "[DONE]") return { done: true }; + if (typeof inner !== "string") { + if (typeof inner === "object") return { parsed: inner }; + return { raw: true, text: String(inner) }; + } + try { + return { parsed: JSON.parse(inner) }; + } catch { + return { raw: true, text: inner }; + } +} + +/** + * @param {object} opts + * @param {string} opts.model + * @param {TextEncoder} opts.encoder + * @param {string} opts.sseDone "data: [DONE]\\n\\n" + */ +export function createQoderSseCoalescer({ model, encoder, sseDone }) { + let pendingFinish = null; + let pendingUsage = null; + let lastMeta = { id: null, created: null, model }; + let doneEmitted = false; + let finishAlreadyForwarded = false; + + const emitJson = (controller, obj) => { + const sanitized = JSON.stringify(obj).replace(/\r?\n/g, ""); + controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`)); + }; + + const emitRaw = (controller, text) => { + controller.enqueue(encoder.encode(`data: ${String(text).replace(/\r?\n/g, "")}\n\n`)); + }; + + const emitDone = (controller) => { + if (doneEmitted) return; + controller.enqueue(encoder.encode(sseDone)); + doneEmitted = true; + }; + + const emitTerminal = (controller) => { + if (!pendingFinish && !pendingUsage) return; + emitJson(controller, { + id: lastMeta.id || `qoder-${Date.now()}`, + object: "chat.completion.chunk", + created: lastMeta.created || Math.floor(Date.now() / 1000), + model: lastMeta.model || model, + choices: [{ index: 0, delta: {}, finish_reason: pendingFinish || "stop" }], + ...(pendingUsage ? { usage: pendingUsage } : {}), + }); + pendingFinish = null; + pendingUsage = null; + }; + + const flush = (controller) => { + if (doneEmitted) return; + if (pendingUsage || (pendingFinish && !finishAlreadyForwarded)) { + emitTerminal(controller); + } + emitDone(controller); + }; + + const handleInner = (inner, controller) => { + if (doneEmitted) return { terminal: true }; + + const parsedInner = parseInner(inner); + if (parsedInner.done) { + flush(controller); + return { terminal: true }; + } + if (parsedInner.raw) { + emitRaw(controller, parsedInner.text); + return {}; + } + const parsed = parsedInner.parsed; + if (!parsed || typeof parsed !== "object") return {}; + + if (typeof parsed.id === "string" && parsed.id) lastMeta.id = parsed.id; + if (typeof parsed.created === "number") lastMeta.created = parsed.created; + if (typeof parsed.model === "string" && parsed.model) lastMeta.model = parsed.model; + + const usage = canonicalizeQoderUsage(parsed.usage); + if (usage) pendingUsage = usage; + + const finish = finishReasonOf(parsed); + if (hasValuableDelta(parsed)) { + // Stream content as-is (preserves upstream JSON for tests/clients). + emitRaw(controller, typeof inner === "string" ? inner : JSON.stringify(parsed)); + if (finish) { + finishAlreadyForwarded = true; + // Keep finish around only if we still need a usage trailer. + pendingFinish = pendingUsage ? finish : null; + } + if (pendingFinish && pendingUsage) { + emitTerminal(controller); + emitDone(controller); + return { terminal: true }; + } + return {}; + } + + if (finish) pendingFinish = finish; + + // Empty finish and/or usage-only: emit as soon as we have both (Qoder + // order is finish then usage). Don't wait for the later [DONE]/keepalive. + if ((pendingFinish || finishAlreadyForwarded) && pendingUsage) { + if (!pendingFinish) pendingFinish = "stop"; + emitTerminal(controller); + emitDone(controller); + return { terminal: true }; + } + return {}; + }; + + return { + handleInner, + flush, + get doneEmitted() { + return doneEmitted; + }, + }; +} diff --git a/open-sse/transformer/responsesTransformer.js b/open-sse/transformer/responsesTransformer.js index ac84db20..6558a3f1 100644 --- a/open-sse/transformer/responsesTransformer.js +++ b/open-sse/transformer/responsesTransformer.js @@ -6,6 +6,7 @@ import fs from "fs"; import path from "path"; +import { toResponsesUsage } from "../translator/concerns/usage.js"; // Create log directory for responses (Node.js only) export function createResponsesLogger(model, logsDir = null) { @@ -73,6 +74,7 @@ export function createResponsesApiTransformStream(logger = null) { funcArgsDone: {}, funcItemDone: {}, buffer: "", + usage: null, completedSent: false }; @@ -225,17 +227,17 @@ export function createResponsesApiTransformStream(logger = null) { const sendCompleted = (controller) => { if (!state.completedSent) { state.completedSent = true; - emit(controller, "response.completed", { - type: "response.completed", - response: { - id: state.responseId, - object: "response", - created_at: state.created, - status: "completed", - background: false, - error: null - } - }); + const response = { + id: state.responseId, + object: "response", + created_at: state.created, + status: "completed", + background: false, + error: null + }; + const usage = toResponsesUsage(state.usage); + if (usage) response.usage = usage; + emit(controller, "response.completed", { type: "response.completed", response }); } }; @@ -264,6 +266,9 @@ export function createResponsesApiTransformStream(logger = null) { continue; } + // Remember usage (finish chunk or trailing include_usage frame) for response.completed + if (parsed.usage && typeof parsed.usage === "object") state.usage = parsed.usage; + if (!parsed.choices?.length) continue; const choice = parsed.choices[0]; diff --git a/open-sse/translator/concerns/usage.js b/open-sse/translator/concerns/usage.js index 3ace3062..1ee28489 100644 --- a/open-sse/translator/concerns/usage.js +++ b/open-sse/translator/concerns/usage.js @@ -67,3 +67,34 @@ export function toOpenAIUsage(raw, kind) { if (!extract || !raw || typeof raw !== "object") return null; return buildUsage(extract(raw)); } + +// Convert an OpenAI-shaped (or already-canonical / Claude-shaped) usage object into the +// Responses API shape emitted by `response.completed`. Details objects are always present +// (like the real API) so proxies that read `input_tokens_details.cached_tokens` never see undefined. +// Returns null when there is nothing countable. +export function toResponsesUsage(usage) { + if (!usage || typeof usage !== "object") return null; + const input = n(usage.prompt_tokens ?? usage.input_tokens); + const output = n(usage.completion_tokens ?? usage.output_tokens); + if (input === 0 && output === 0) return null; + const cached = n( + usage.input_tokens_details?.cached_tokens ?? + usage.prompt_tokens_details?.cached_tokens ?? + usage.cached_tokens ?? + usage.cache_read_input_tokens + ); + const reasoning = n( + usage.output_tokens_details?.reasoning_tokens ?? + usage.completion_tokens_details?.reasoning_tokens ?? + usage.reasoning_tokens + ); + const out = { + input_tokens: input, + output_tokens: output, + total_tokens: typeof usage.total_tokens === "number" ? usage.total_tokens : input + output, + input_tokens_details: { cached_tokens: cached }, + output_tokens_details: { reasoning_tokens: reasoning }, + }; + if (usage.estimated) out.estimated = true; + return out; +} diff --git a/open-sse/translator/response/openai-responses.js b/open-sse/translator/response/openai-responses.js index bd435f9c..1d9a7dec 100644 --- a/open-sse/translator/response/openai-responses.js +++ b/open-sse/translator/response/openai-responses.js @@ -5,7 +5,7 @@ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; import { buildChunk } from "../concerns/chunk.js"; -import { buildUsage } from "../concerns/usage.js"; +import { buildUsage, toResponsesUsage } from "../concerns/usage.js"; import { fallbackToolCallId } from "../concerns/toolCall.js"; import { reasoningDelta, extractReasoningText } from "../concerns/reasoning.js"; import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } from "../schema/index.js"; @@ -18,7 +18,13 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { if (!chunk) { return flushEvents(state); } - + + // Usage riding on the finish chunk (include_usage style, e.g. coalesced Qoder frames): + // remember it so response.completed can report tokens even outside stream.js. + if (chunk.usage && typeof chunk.usage === "object" && !state.usage) { + state.usage = chunk.usage; + } + if (!chunk.choices?.length) return []; const events = []; @@ -368,17 +374,19 @@ function closeToolCall(state, emit, idx) { function sendCompleted(state, emit) { if (!state.completedSent) { state.completedSent = true; - emit("response.completed", { - type: "response.completed", - response: { - id: state.responseId, - object: "response", - created_at: state.created, - status: "completed", - background: false, - error: null - } - }); + const response = { + id: state.responseId, + object: "response", + created_at: state.created, + status: "completed", + background: false, + error: null + }; + // Carry provider usage (recorded by stream.js or from the finish chunk itself) in the + // Responses shape; proxies such as sub2api/Codex read tokens only from here. + const usage = toResponsesUsage(state.usage); + if (usage) response.usage = usage; + emit("response.completed", { type: "response.completed", response }); } } diff --git a/open-sse/translator/response/openai-to-claude.js b/open-sse/translator/response/openai-to-claude.js index 3998cc84..f291e83a 100644 --- a/open-sse/translator/response/openai-to-claude.js +++ b/open-sse/translator/response/openai-to-claude.js @@ -67,48 +67,50 @@ function stopTextBlock(state, results) { state.textBlockStarted = false; } +function recordOpenAIUsage(chunk, state) { + if (!chunk?.usage || typeof chunk.usage !== "object") return; + + const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0; + const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0; + + // Extract cache tokens from prompt_tokens_details + const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens; + const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens; + const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0; + const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0; + + // input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens + // Because OpenAI's prompt_tokens includes all prompt-side tokens + const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens; + + state.usage = { + input_tokens: inputTokens, + output_tokens: outputTokens + }; + + if (cacheReadTokens > 0) { + state.usage.cache_read_input_tokens = cacheReadTokens; + } + if (cacheCreateTokens > 0) { + state.usage.cache_creation_input_tokens = cacheCreateTokens; + } +} + // Convert OpenAI stream chunk to Claude format export function openaiToClaudeResponse(chunk, state) { - if (!chunk || !chunk.choices?.[0]) return null; + if (!chunk) return null; + + // Track usage from OpenAI chunk if available + if (chunk.usage && typeof chunk.usage === "object") { + recordOpenAIUsage(chunk, state); + } + + if (!chunk.choices?.[0]) return null; const results = []; const choice = chunk.choices[0]; const delta = choice.delta; - // Track usage from OpenAI chunk if available - if (chunk.usage && typeof chunk.usage === "object") { - const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0; - const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0; - - // Extract cache tokens from prompt_tokens_details - const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens; - const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens; - const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0; - const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0; - - // input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens - // Because OpenAI's prompt_tokens includes all prompt-side tokens - const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens; - - state.usage = { - input_tokens: inputTokens, - output_tokens: outputTokens - }; - - // Add cache_read_input_tokens if present - if (cacheReadTokens > 0) { - state.usage.cache_read_input_tokens = cacheReadTokens; - } - - // Add cache_creation_input_tokens if present - if (cacheCreateTokens > 0) { - state.usage.cache_creation_input_tokens = cacheCreateTokens; - } - - // Note: completion_tokens_details.reasoning_tokens is already included in output_tokens - // No need to add separately as Claude expects total output_tokens - } - // First chunk - ALWAYS send message_start first if (!state.messageStartSent) { state.messageStartSent = true; @@ -221,8 +223,9 @@ export function openaiToClaudeResponse(chunk, state) { } } - // Finish - if (choice.finish_reason) { + // Finish (OpenAI puts this on the choice; Qoder often puts it on delta) + const finishReason = choice.finish_reason || delta?.finish_reason; + if (finishReason) { stopThinkingBlock(state, results); stopTextBlock(state, results); @@ -244,13 +247,13 @@ export function openaiToClaudeResponse(chunk, state) { } // Mark finish for later usage injection in stream.js - state.finishReason = choice.finish_reason; + state.finishReason = finishReason; // Use tracked usage (will be estimated in stream.js if not valid) const finalUsage = state.usage || { input_tokens: 0, output_tokens: 0 }; results.push({ type: "message_delta", - delta: { stop_reason: convertFinishReason(choice.finish_reason) }, + delta: { stop_reason: convertFinishReason(finishReason) }, usage: finalUsage }); results.push({ type: "message_stop" }); diff --git a/open-sse/utils/stream.js b/open-sse/utils/stream.js index 15ee37d8..f61fd193 100644 --- a/open-sse/utils/stream.js +++ b/open-sse/utils/stream.js @@ -3,6 +3,7 @@ import { FORMATS } from "../translator/formats.js"; import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb.js"; import { extractUsage, mergeUsage, hasValidUsage, estimateUsage, logUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js"; import { parseSSELine, hasValuableContent, fixInvalidId, formatSSE } from "./streamHelpers.js"; +import { toResponsesUsage } from "../translator/concerns/usage.js"; import { getOpenAIResponsesEventName, isOpenAIResponsesTerminalEvent, formatIncompleteOpenAIResponsesStreamFailure } from "./responsesStreamHelpers.js"; import { dbg, isDebugEnabled } from "./debugLog.js"; @@ -201,7 +202,8 @@ export function createSSEStream(options = {}) { responsesTerminal = isOpenAIResponsesTerminalEvent(currentOpenAIResponsesEvent, parsed); - const isFinishChunk = parsed.choices?.[0]?.finish_reason; + const isFinishChunk = parsed.choices?.[0]?.finish_reason + || parsed.choices?.[0]?.delta?.finish_reason; if (isFinishChunk && !hasValidUsage(parsed.usage)) { const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI); parsed.usage = filterUsageForFormat(estimated, FORMATS.OPENAI); @@ -365,6 +367,19 @@ export function createSSEStream(options = {}) { item.usage = filterUsageForFormat(buffered, sourceFormat); } + // Responses API clients (Codex, sub2api /v1/responses): usage lives on + // response.completed → response.usage. Same buffer/estimate policy as above. + const completedResponse = item.event === "response.completed" ? item.data?.response : null; + if (completedResponse && typeof completedResponse === "object") { + if (state.usage) { + completedResponse.usage = toResponsesUsage(addBufferToUsage(state.usage)) ?? completedResponse.usage; + } else if (!completedResponse.usage && totalContentLength > 0) { + const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI); + completedResponse.usage = toResponsesUsage(estimated); + state.usage = estimated; + } + } + const output = formatSSE(item, sourceFormat); reqLogger?.appendConvertedChunk?.(output); controller.enqueue(sharedEncoder.encode(output)); diff --git a/src/app/api/v1/models/route.js b/src/app/api/v1/models/route.js index 40adc696..173d91c2 100644 --- a/src/app/api/v1/models/route.js +++ b/src/app/api/v1/models/route.js @@ -9,7 +9,7 @@ import { getProviderConnections, getCombos, getCustomModels, getModelAliases } f import { getDisabledModels } from "@/lib/disabledModelsDb"; import { resolveKiroModels } from "open-sse/services/kiroModels.js"; import { resolveKimchiModels } from "open-sse/services/kimchiModels.js"; -import { resolveQoderModels } from "open-sse/services/qoderModels.js"; +import { resolveQoderModels, routableQoderModels } from "open-sse/services/qoderModels.js"; import { resolveCopilotModels } from "open-sse/services/copilotModels.js"; import { resolveClinepassModels } from "open-sse/services/clinepassModels.js"; import { resolveGrokCliModels } from "open-sse/services/grokCliModels.js"; @@ -34,15 +34,18 @@ const LIVE_MODEL_RESOLVERS = { qoder: async (conn) => { const result = await resolveQoderModels({ accessToken: conn.accessToken, + // PAT (pt-...) connections keep the token in apiKey; without it the live + // catalog silently fails and /v1/models falls back to the static list. + apiKey: conn.apiKey, refreshToken: conn.refreshToken, email: conn.email, displayName: conn.displayName, providerSpecificData: conn.providerSpecificData || {} }); - if (!result?.models?.length) return null; - return { - models: result.models.map((m) => ({ id: m.id, name: m.name })), - }; + // Visible + hidden (enable:false) catalog keys — chat routes all of them. + const models = routableQoderModels(result); + if (!models.length) return null; + return { models: models.map((m) => ({ id: m.id, name: m.name })) }; }, kimchi: async (conn) => { const result = await resolveKimchiModels({ diff --git a/tests/unit/openai-responses-usage.test.js b/tests/unit/openai-responses-usage.test.js new file mode 100644 index 00000000..38ec581d --- /dev/null +++ b/tests/unit/openai-responses-usage.test.js @@ -0,0 +1,182 @@ +/** + * Responses API clients (Codex, sub2api /v1/responses) read token usage only from + * `response.completed → response.usage`. For chat-native upstreams (Qoder, most + * OpenAI-compatible providers) the translator used to emit that event without usage, + * so proxies logged 0 input / 0 output / 0 cached tokens. + */ +import { describe, expect, it, vi } from "vitest"; + +vi.mock("@/lib/usageDb.js", () => ({ + appendRequestLog: vi.fn(async () => {}), + saveRequestDetail: vi.fn(async () => {}), + saveRequestUsage: vi.fn(async () => {}), + trackPendingRequest: vi.fn(() => {}), +})); + +const { FORMATS } = await import("../../open-sse/translator/formats.js"); +const { initState } = await import("../../open-sse/translator/index.js"); +const { toResponsesUsage } = await import("../../open-sse/translator/concerns/usage.js"); +const { openaiToOpenAIResponsesResponse } = await import("../../open-sse/translator/response/openai-responses.js"); +const { createSSETransformStreamWithLogger } = await import("../../open-sse/utils/stream.js"); +const { createResponsesApiTransformStream } = await import("../../open-sse/transformer/responsesTransformer.js"); +const { addBufferToUsage } = await import("../../open-sse/utils/usageTracking.js"); +// stream.js adds the same context-safety buffer it applies to chat/claude clients +const BUFFER_TOKENS = addBufferToUsage({ prompt_tokens: 0 }).prompt_tokens; + +const QODER_FINISH_CHUNK = { + id: "chatcmpl-qoder-1", + object: "chat.completion.chunk", + created: 1_700_000_000, + model: "qmodel_38max", + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + usage: { + prompt_tokens: 27_339, + completion_tokens: 437, + total_tokens: 27_776, + prompt_tokens_details: { cached_tokens: 27_200 }, + }, +}; + +function sse(chunks) { + return chunks.map((c) => `data: ${typeof c === "string" ? c : JSON.stringify(c)}\n\n`).join(""); +} + +async function pipe(input, transform) { + const encoder = new TextEncoder(); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(input)); + controller.close(); + }, + }); + const reader = stream.pipeThrough(transform).getReader(); + const decoder = new TextDecoder(); + let text = ""; + for (;;) { + const { value, done } = await reader.read(); + if (done) break; + text += decoder.decode(value, { stream: true }); + } + return text + decoder.decode(); +} + +function completedEvent(text) { + const m = text.match(/event: response\.completed\ndata: (.+)\n/); + return m ? JSON.parse(m[1]) : null; +} + +describe("toResponsesUsage", () => { + it("maps OpenAI usage (nested cached_tokens) to the Responses shape", () => { + expect(toResponsesUsage(QODER_FINISH_CHUNK.usage)).toEqual({ + input_tokens: 27_339, + output_tokens: 437, + total_tokens: 27_776, + input_tokens_details: { cached_tokens: 27_200 }, + output_tokens_details: { reasoning_tokens: 0 }, + }); + }); + + it("accepts canonical flat fields and Claude-style cache fields", () => { + expect(toResponsesUsage({ prompt_tokens: 10, completion_tokens: 2, cached_tokens: 4, reasoning_tokens: 1 })).toMatchObject({ + input_tokens: 10, + output_tokens: 2, + total_tokens: 12, + input_tokens_details: { cached_tokens: 4 }, + output_tokens_details: { reasoning_tokens: 1 }, + }); + expect(toResponsesUsage({ input_tokens: 5, output_tokens: 1, cache_read_input_tokens: 3 }).input_tokens_details.cached_tokens).toBe(3); + }); + + it("keeps the estimated marker and returns null for empty usage", () => { + expect(toResponsesUsage({ prompt_tokens: 1, completion_tokens: 1, estimated: true }).estimated).toBe(true); + expect(toResponsesUsage({})).toBeNull(); + expect(toResponsesUsage(null)).toBeNull(); + }); +}); + +describe("openai → openai-responses translator", () => { + it("puts usage from the finish chunk on response.completed", () => { + const state = initState(FORMATS.OPENAI_RESPONSES); + const events = openaiToOpenAIResponsesResponse(QODER_FINISH_CHUNK, state); + const completed = events.find((e) => e.event === "response.completed"); + expect(completed).toBeTruthy(); + expect(completed.data.response.usage).toEqual({ + input_tokens: 27_339, + output_tokens: 437, + total_tokens: 27_776, + input_tokens_details: { cached_tokens: 27_200 }, + output_tokens_details: { reasoning_tokens: 0 }, + }); + }); + + it("omits usage when the upstream never reported any", () => { + const state = initState(FORMATS.OPENAI_RESPONSES); + const events = openaiToOpenAIResponsesResponse({ ...QODER_FINISH_CHUNK, usage: undefined }, state); + const completed = events.find((e) => e.event === "response.completed"); + expect(completed.data.response.usage).toBeUndefined(); + }); +}); + +describe("stream.js translate mode: chat upstream → Responses client", () => { + const transform = () => createSSETransformStreamWithLogger( + FORMATS.OPENAI, // provider (Qoder executor emits OpenAI chunks) + FORMATS.OPENAI_RESPONSES, // client + "qoder", + null, + null, + "qmodel_38max", + null, + { model: "qd/qmodel_38max", messages: [{ role: "user", content: "hi" }] }, + ); + + it("emits provider usage (+buffer) with cached tokens on response.completed", async () => { + const out = await pipe(sse([ + { ...QODER_FINISH_CHUNK, choices: [{ index: 0, delta: { role: "assistant", content: "Hello" }, finish_reason: null }], usage: undefined }, + QODER_FINISH_CHUNK, + "[DONE]", + ]), transform()); + + const completed = completedEvent(out); + expect(completed).toBeTruthy(); + expect(completed.response.usage).toEqual({ + input_tokens: 27_339 + BUFFER_TOKENS, + output_tokens: 437, + total_tokens: 27_776 + BUFFER_TOKENS, + input_tokens_details: { cached_tokens: 27_200 }, + output_tokens_details: { reasoning_tokens: 0 }, + }); + // Responses clients terminate on response.completed (no [DONE] sentinel in translate mode) + expect(out.indexOf("event: response.completed")).toBeGreaterThan(out.indexOf("event: response.output_item.done")); + }); + + it("injects estimated usage when the upstream reports none", async () => { + const out = await pipe(sse([ + { ...QODER_FINISH_CHUNK, choices: [{ index: 0, delta: { role: "assistant", content: "Hello world" }, finish_reason: null }], usage: undefined }, + { ...QODER_FINISH_CHUNK, usage: undefined }, + "[DONE]", + ]), transform()); + + const completed = completedEvent(out); + expect(completed.response.usage).toBeTruthy(); + expect(completed.response.usage.estimated).toBe(true); + expect(completed.response.usage.input_tokens).toBeGreaterThan(0); + expect(completed.response.usage.output_tokens).toBeGreaterThan(0); + }); +}); + +describe("responsesTransformer (Chat SSE → Codex Responses SSE)", () => { + it("forwards finish-chunk usage on response.completed", async () => { + const out = await pipe(sse([ + { ...QODER_FINISH_CHUNK, choices: [{ index: 0, delta: { role: "assistant", content: "Hello" }, finish_reason: null }], usage: undefined }, + QODER_FINISH_CHUNK, + "[DONE]", + ]), createResponsesApiTransformStream()); + + const completed = completedEvent(out); + expect(completed.response.usage).toMatchObject({ + input_tokens: 27_339, + output_tokens: 437, + input_tokens_details: { cached_tokens: 27_200 }, + }); + }); +}); diff --git a/tests/unit/openai-to-claude.test.js b/tests/unit/openai-to-claude.test.js index 45b67fb2..0c5768d7 100644 --- a/tests/unit/openai-to-claude.test.js +++ b/tests/unit/openai-to-claude.test.js @@ -203,4 +203,31 @@ describe("openaiToClaudeResponse", () => { limit: 120 }); }); + + it("records usage from a choices:[] frame so the finish chunk can emit it", () => { + const state = { toolCalls: new Map() }; + expect(openaiToClaudeResponse({ + usage: { + prompt_tokens: 90, + completion_tokens: 7, + prompt_tokens_details: { cached_tokens: 30 }, + }, + choices: [], + }, state)).toBeNull(); + expect(state.usage).toEqual({ + input_tokens: 60, + output_tokens: 7, + cache_read_input_tokens: 30, + }); + + const events = openaiToClaudeResponse({ + id: "chatcmpl-qoder-finish", + model: "qoder/auto", + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + }, state); + const delta = events.find((e) => e.type === "message_delta"); + expect(delta.usage.input_tokens).toBe(60); + expect(delta.usage.output_tokens).toBe(7); + expect(delta.usage.cache_read_input_tokens).toBe(30); + }); }); diff --git a/tests/unit/qoder-context-tier.test.js b/tests/unit/qoder-context-tier.test.js new file mode 100644 index 00000000..8850fd0d --- /dev/null +++ b/tests/unit/qoder-context-tier.test.js @@ -0,0 +1,193 @@ +/** + * Qoder context-window tiers + routable model listing. + * + * The Qoder IDE lets a user pick 200K / 400K / 1M for a model; qodercli-style + * requests (what 9router sends) only carry the default max_input_tokens. These + * tests pin the escalation policy and the payload fields the IDE writes. + */ +import { describe, it, expect } from "vitest"; + +import { + parseTierTokenCount, + getQoderContextTiers, + estimateQoderPromptTokens, + resolveQoderContextTier, + applyQoderContextTier, +} from "../../open-sse/shared/qoder/contextTier.js"; +import { routableQoderModels } from "../../open-sse/services/qoderModels.js"; + +// Shape mirrors the live /algo/api/v2/model/list entry for qmodel_38max. +const MODEL_CONFIG = { + key: "qmodel_38max", + display_name: "Qwen3.8-Max", + is_reasoning: true, + max_input_tokens: 180_000, + max_output_tokens: 32_768, + context_config: [ + { name: "200K", tokenCount: 200_000, isDefault: true }, + { name: "400K", tokenCount: 400_000, isDefault: false }, + { name: "1M", tokenCount: 1_000_000, isDefault: false }, + ], +}; + +function promptOfTokens(n) { + // ~4 ASCII chars per token + return { system: "", messages: [{ role: "user", content: "abcd".repeat(n) }], tools: [] }; +} + +describe("parseTierTokenCount", () => { + it("accepts numbers and K/M suffixed strings", () => { + expect(parseTierTokenCount(204800)).toBe(204800); + expect(parseTierTokenCount("200K")).toBe(200_000); + expect(parseTierTokenCount("1M")).toBe(1_000_000); + expect(parseTierTokenCount("1.5m")).toBe(1_500_000); + expect(parseTierTokenCount("131072")).toBe(131072); + }); + + it("returns 0 for garbage", () => { + expect(parseTierTokenCount(null)).toBe(0); + expect(parseTierTokenCount("big")).toBe(0); + expect(parseTierTokenCount(-5)).toBe(0); + }); +}); + +describe("getQoderContextTiers", () => { + it("sorts tiers ascending and keeps the default flag", () => { + const tiers = getQoderContextTiers({ + context_config: [ + { name: "1M", tokenCount: 1_000_000 }, + { name: "200K", tokenCount: 200_000, isDefault: true }, + ], + }); + expect(tiers.map((t) => t.tokenCount)).toEqual([200_000, 1_000_000]); + expect(tiers[0].isDefault).toBe(true); + expect(tiers[1].isDefault).toBe(false); + }); + + it("understands camelCase / snake_case variants and derives names", () => { + const tiers = getQoderContextTiers({ + contextConfig: [{ token_count: "400K", is_default: true }, { max_input_tokens: 1_000_000 }], + }); + expect(tiers).toEqual([ + { name: "400K", tokenCount: 400_000, isDefault: true }, + { name: "1M", tokenCount: 1_000_000, isDefault: false }, + ]); + }); + + it("returns [] when the model has no tiers", () => { + expect(getQoderContextTiers({ max_input_tokens: 131072 })).toEqual([]); + expect(getQoderContextTiers(null)).toEqual([]); + }); +}); + +describe("estimateQoderPromptTokens", () => { + it("counts CJK characters as ~1 token each instead of chars/4", () => { + const ascii = estimateQoderPromptTokens({ messages: [{ role: "user", content: "a".repeat(4000) }] }); + const cjk = estimateQoderPromptTokens({ messages: [{ role: "user", content: "中".repeat(4000) }] }); + expect(ascii).toBeLessThan(1_200); + expect(cjk).toBeGreaterThan(4_000); + }); +}); + +describe("resolveQoderContextTier (auto)", () => { + it("leaves the payload untouched while the prompt fits the current max_input_tokens", () => { + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(50_000))).toBeNull(); + }); + + it("escalates to the smallest tier that fits once the prompt outgrows the default", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(250_000)); + expect(choice).not.toBeNull(); + expect(choice.tier.name).toBe("400K"); + expect(choice.reason).toBe("auto:fits"); + expect(choice.estimatedTokens).toBeGreaterThan(240_000); + }); + + it("falls back to the largest tier when nothing fits (upstream decides)", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(1_200_000)); + expect(choice.tier.name).toBe("1M"); + expect(choice.reason).toBe("auto:largest"); + }); + + it("applies headroom so a prompt just under the limit still escalates", () => { + // 170K estimated * 1.15 = 195.5K > 180K current → smallest tier above the current limit (200K) + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(170_000))?.tier.name).toBe("200K"); + // 190K * 1.15 = 218.5K → 200K no longer fits → 400K + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(190_000))?.tier.name).toBe("400K"); + }); + + it("returns null for models without context_config", () => { + expect(resolveQoderContextTier({ max_input_tokens: 131072 }, promptOfTokens(500_000))).toBeNull(); + }); + + it("never escalates when the current limit is already the largest tier", () => { + const cfg = { ...MODEL_CONFIG, max_input_tokens: 1_000_000 }; + expect(resolveQoderContextTier(cfg, promptOfTokens(1_500_000))).toBeNull(); + }); +}); + +describe("resolveQoderContextTier (forced via QODER_CONTEXT_TIER)", () => { + it("max picks the largest tier regardless of prompt size", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "max" }); + expect(choice.tier.name).toBe("1M"); + expect(choice.reason).toBe("forced:max"); + }); + + it("default picks the isDefault tier", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "default" }); + expect(choice.tier.name).toBe("200K"); + }); + + it("a tier name or token count selects that tier", () => { + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "400k" }).tier.tokenCount).toBe(400_000); + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "1000000" }).tier.name).toBe("1M"); + }); + + it("an unknown tier name falls back to auto", () => { + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "9M" })).toBeNull(); + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(250_000), { preference: "9M" }).tier.name).toBe("400K"); + }); +}); + +describe("applyQoderContextTier", () => { + it("mirrors the tier into the three places the IDE writes", () => { + const payload = { + parameters: { max_tokens: 32_768 }, + chat_context: { extra: { context: [], modelConfig: { key: "qmodel_38max" } } }, + model_config: { ...MODEL_CONFIG }, + }; + applyQoderContextTier(payload, { name: "1M", tokenCount: 1_000_000 }); + expect(payload.parameters).toEqual({ max_tokens: 32_768, context_length: 1_000_000 }); + expect(payload.chat_context.extra.ideModelConfigOverride).toEqual({ max_input_tokens: 1_000_000 }); + expect(payload.chat_context.extra.modelConfig).toEqual({ key: "qmodel_38max" }); + expect(payload.model_config.max_input_tokens).toBe(1_000_000); + expect(payload.model_config.context_config).toHaveLength(3); + }); + + it("is a no-op without a tier", () => { + const payload = { parameters: { max_tokens: 1 } }; + expect(applyQoderContextTier(payload, null)).toBe(payload); + expect(payload).toEqual({ parameters: { max_tokens: 1 } }); + }); +}); + +describe("routableQoderModels", () => { + it("lists visible models first, then hidden (enable:false) catalog keys", () => { + const catalog = { + models: [{ id: "qmodel_38max", name: "Qwen3.8-Max" }], + rawConfigs: new Map([ + ["qmodel_38max", { key: "qmodel_38max", enable: true }], + ["qfmodel", { key: "qfmodel", enable: false, display_name: "Qwen Fast" }], + ["dmodel", { key: "dmodel", enable: false }], + ]), + }; + expect(routableQoderModels(catalog)).toEqual([ + { id: "qmodel_38max", name: "Qwen3.8-Max", hidden: false }, + { id: "qfmodel", name: "Qwen Fast", hidden: true }, + { id: "dmodel", name: "dmodel", hidden: true }, + ]); + }); + + it("returns [] for a failed catalog fetch", () => { + expect(routableQoderModels(null)).toEqual([]); + }); +}); diff --git a/tests/unit/qoder.test.js b/tests/unit/qoder.test.js index 04ce1c45..ea45ac61 100644 --- a/tests/unit/qoder.test.js +++ b/tests/unit/qoder.test.js @@ -9,7 +9,7 @@ * - device flow URL construction */ -import { describe, it, expect } from "vitest"; +import { describe, it, expect, beforeEach } from "vitest"; import crypto from "crypto"; import { qoderEncodeBody } from "../../src/lib/qoder/encoding.js"; @@ -22,6 +22,13 @@ import { } from "../../src/lib/qoder/constants.js"; import { PROVIDER_MODELS } from "../../open-sse/config/providerModels.js"; import { __test__ as qoderExecutorInternals } from "../../open-sse/executors/qoder.js"; +import { canonicalizeQoderUsage } from "../../open-sse/shared/qoder/sse.js"; +import { + rewriteQoderMessageAttachments, + clearQoderUploadCache, + buildMultipartFile, +} from "../../open-sse/shared/qoder/attachments.js"; +import { qoderInferenceBase } from "../../open-sse/shared/qoder/constants.js"; // Convenience aliases — tests were originally written against module-level // helpers; the QoderService class wraps them so each test creates its own @@ -431,6 +438,21 @@ describe("normalizeMessages", () => { ]); expect(result.messages[0].content).toBe("hi"); }); + + it("turns leftover file/document blocks into short stubs instead of dropping them", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "see" }, + { type: "file", file: { filename: "big.pdf", file_data: "data:application/pdf;base64,AAA" } }, + ], + }, + ]); + expect(result.messages[0].content).toContain("see"); + expect(result.messages[0].content).toContain("big.pdf"); + expect(result.messages[0].content).not.toContain("AAA"); + }); }); describe("wrapQoderSSE", () => { @@ -530,4 +552,190 @@ describe("wrapQoderSSE", () => { const wrapped = await wrapQoderSSE(r, "qoder/auto"); expect(wrapped).toBe(r); }); + + function envelope(body) { + return `data: ${JSON.stringify({ statusCodeValue: 200, body })}\n\n`; + } + + function parseForwardedChunks(out) { + return out + .split("\n\n") + .map((block) => block.trim()) + .filter((block) => block.startsWith("data:") && !block.includes("[DONE]")) + .map((block) => JSON.parse(block.slice("data:".length).trim())); + } + + it("coalesces empty finish-in-delta + usage-only into one OpenAI usage chunk", async () => { + const content = JSON.stringify({ + id: "chatcmpl-qoder-1", + created: 1700000000, + model: "auto", + choices: [{ index: 0, delta: { content: "hi" } }], + }); + const finish = JSON.stringify({ + id: "chatcmpl-qoder-1", + choices: [{ index: 0, delta: { content: "", finish_reason: "stop" } }], + }); + const usage = JSON.stringify({ + id: "chatcmpl-qoder-1", + choices: [], + usage: { + prompt_tokens: 100, + completion_tokens: 20, + total_tokens: 120, + prompt_tokens_details: { cached_tokens: 40 }, + }, + }); + const wrapped = await wrapQoderSSE( + makeResponse([envelope(content) + envelope(finish) + envelope(usage) + envelope("[DONE]")]), + "qoder/auto", + ); + const out = await drain(wrapped); + expect(out).toContain(`data: ${content}\n\n`); + const chunks = parseForwardedChunks(out); + const usageChunk = chunks.find((c) => c.usage); + expect(usageChunk).toBeDefined(); + expect(usageChunk.choices[0].finish_reason).toBe("stop"); + expect(usageChunk.usage.prompt_tokens).toBe(100); + expect(usageChunk.usage.completion_tokens).toBe(20); + expect(usageChunk.usage.prompt_tokens_details.cached_tokens).toBe(40); + expect(chunks.some((c) => Array.isArray(c.choices) && c.choices.length === 0)).toBe(false); + expect((out.match(/data: \[DONE\]/g) || []).length).toBe(1); + }); + + it("maps Qoder input_tokens aliases onto prompt_tokens in the coalesced usage chunk", async () => { + const finish = JSON.stringify({ + choices: [{ index: 0, delta: { finish_reason: "stop" } }], + }); + const usage = JSON.stringify({ + choices: [], + usage: { + input_tokens: 80, + output_tokens: 10, + cache_read_input_tokens: 25, + }, + }); + const wrapped = await wrapQoderSSE( + makeResponse([envelope(finish) + envelope(usage)]), + "qoder/lite", + ); + const chunks = parseForwardedChunks(await drain(wrapped)); + const usageChunk = chunks.find((c) => c.usage); + expect(usageChunk.usage.prompt_tokens).toBe(80); + expect(usageChunk.usage.completion_tokens).toBe(10); + expect(usageChunk.usage.prompt_tokens_details.cached_tokens).toBe(25); + }); +}); + +describe("canonicalizeQoderUsage", () => { + it("returns null for missing or empty usage", () => { + expect(canonicalizeQoderUsage(null)).toBeNull(); + expect(canonicalizeQoderUsage({})).toBeNull(); + }); + + it("copies prompt_tokens_details.cached_tokens through", () => { + const out = canonicalizeQoderUsage({ + prompt_tokens: 50, + completion_tokens: 5, + prompt_tokens_details: { cached_tokens: 12 }, + }); + expect(out.prompt_tokens).toBe(50); + expect(out.cached_tokens).toBe(12); + expect(out.prompt_tokens_details.cached_tokens).toBe(12); + expect(out.total_tokens).toBe(55); + }); +}); + +describe("qoderInferenceBase", () => { + it("sends job tokens to api2 and device tokens to api3", () => { + expect(qoderInferenceBase({ accessToken: "jt-abc" })).toContain("api2.qoder.sh"); + expect(qoderInferenceBase({ accessToken: "dt-abc" })).toContain("api3.qoder.sh"); + }); +}); + +describe("rewriteQoderMessageAttachments", () => { + beforeEach(() => clearQoderUploadCache()); + + it("uploads data-URI images and keeps only the OSS URL in the message", async () => { + const messages = [{ + role: "user", + content: [ + { type: "text", text: "see this" }, + { type: "image_url", image_url: { url: "data:image/png;base64,AAAA" } }, + ], + }]; + const stats = await rewriteQoderMessageAttachments(messages, { + uploadFn: async ({ buffer, mediaType }) => { + expect(Buffer.isBuffer(buffer)).toBe(true); + expect(mediaType).toBe("image/png"); + return "https://cdn.qoder.example/img.png"; + }, + }); + expect(messages[0].content).toEqual([ + { type: "text", text: "see this" }, + { type: "image_url", image_url: { url: "https://cdn.qoder.example/img.png" } }, + ]); + expect(JSON.stringify(messages)).not.toContain("AAAA"); + expect(stats.imageUrls).toEqual(["https://cdn.qoder.example/img.png"]); + }); + + it("does not re-upload already-hosted http(s) image URLs", async () => { + const messages = [{ + role: "user", + content: [{ type: "image_url", image_url: { url: "https://example.com/a.png" } }], + }]; + await rewriteQoderMessageAttachments(messages, { + uploadFn: async () => { + throw new Error("should not upload remote URLs"); + }, + }); + expect(messages[0].content[0].image_url.url).toBe("https://example.com/a.png"); + }); + + it("stubs non-image file blocks instead of inlining bytes", async () => { + const pdfB64 = "A".repeat(200); + const messages = [{ + role: "user", + content: [ + { type: "text", text: "read this" }, + { type: "file", file: { filename: "big.pdf", file_data: `data:application/pdf;base64,${pdfB64}` } }, + ], + }]; + await rewriteQoderMessageAttachments(messages, { + uploadFn: async () => { + throw new Error("should not upload PDFs as images"); + }, + }); + const wire = JSON.stringify(messages); + expect(wire).not.toContain(pdfB64); + expect(wire).toContain("[file omitted: big.pdf"); + }); + + it("stubs oversized images when OSS upload fails instead of keeping a huge data URI", async () => { + const big = "A".repeat(700_000); + const messages = [{ + role: "user", + content: [{ type: "image_url", image_url: { url: `data:image/png;base64,${big}` } }], + }]; + await rewriteQoderMessageAttachments(messages, { + uploadFn: async () => { + throw new Error("upstream 413"); + }, + }); + const wire = JSON.stringify(messages); + expect(wire).not.toContain(big); + expect(wire).toContain("[file omitted:"); + expect(Buffer.byteLength(wire, "utf8")).toBeLessThan(4096); + }); + + it("buildMultipartFile uses the file field name qodercli sends", () => { + const { boundary, body } = buildMultipartFile(Buffer.from("hi"), { + fileName: "image.png", + mediaType: "image/png", + }); + const text = body.toString("latin1"); + expect(text).toContain(`name="file"`); + expect(text).toContain("filename=\"image.png\""); + expect(text).toContain(`--${boundary}`); + }); }); From 1892ed77c842d0be7db85e3958215ab699640dfc Mon Sep 17 00:00:00 2001 From: Federico Liva Date: Thu, 10 Sep 2026 22:23:01 +0700 Subject: [PATCH 34/78] fix(kiro): never send a top-level systemPrompt (400 REQUEST_BODY_INVALID) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit kiro.dev rejects any body carrying a top-level systemPrompt with 400 REQUEST_BODY_INVALID. The translators stopped emitting the field in v0.5.59 (the prompt travels in the first user turn via contentPrefix), but two paths kept writing it back downstream of the translator: - rtk/systemInject.js::injectKiroSystem() appended the RTK prompt to body.systemPrompt, so every kr/ model failed whenever an RTK injector (caveman, ponytail) was active. It now appends to the first history user turn's content (else currentMessage), reusing dedupStringAppend/hasPrompt so retries stay idempotent. - executors/kiro.js::appendRepairInstruction() wrote the tool-call repair instruction to systemPrompt on the retry, turning every repair into a hard failure. It now appends to currentMessage.userInputMessage.content. isKiroBody() no longer requires a string body.systemPrompt — that marker is gone from the wire shape — and sniffs the conversation turn shape instead, keeping the stray-conversationState guard intact. Stale comments in both kiro translators corrected: the systemPrompt local is only a session-replay cache key, not a wire field. Also drops the mirror/rollback repair heuristic the injector no longer needs: net -52 lines. Fixes #3641, #3845, #2890, #2901, #2939, #3109, #3459, #3749 --- open-sse/executors/kiro.js | 12 +- open-sse/rtk/systemInject.js | 100 ++++--------- open-sse/translator/request/claude-to-kiro.js | 6 +- open-sse/translator/request/openai-to-kiro.js | 6 +- tests/unit/kiro-terminal-integrity.test.js | 10 +- tests/unit/system-inject.test.js | 136 ++++++++---------- 6 files changed, 109 insertions(+), 161 deletions(-) diff --git a/open-sse/executors/kiro.js b/open-sse/executors/kiro.js index e12aced6..9d5d754f 100644 --- a/open-sse/executors/kiro.js +++ b/open-sse/executors/kiro.js @@ -127,12 +127,18 @@ async function readResponsePrefix(response, signal, maxBytes, timeoutMs) { return decoder.decode(concatChunks(chunks, totalBytes)); } +// The instruction goes into the current user turn, never into a top-level +// `systemPrompt`: kiro.dev answers any body carrying that field with +// 400 REQUEST_BODY_INVALID, so writing it here turned every repair retry into +// a hard failure. function appendRepairInstruction(body, kind) { const repaired = structuredClone(body || {}); const instruction = REPAIR_INSTRUCTIONS[kind] || "Retry the previous incomplete Kiro response."; - repaired.systemPrompt = repaired.systemPrompt - ? `${repaired.systemPrompt}\n\n${instruction}` - : instruction; + const msg = repaired?.conversationState?.currentMessage?.userInputMessage; + if (msg) { + const content = typeof msg.content === "string" ? msg.content : ""; + msg.content = content ? `${content}\n\n${instruction}` : instruction; + } return repaired; } diff --git a/open-sse/rtk/systemInject.js b/open-sse/rtk/systemInject.js index b60e15d3..b4e744f8 100644 --- a/open-sse/rtk/systemInject.js +++ b/open-sse/rtk/systemInject.js @@ -13,7 +13,7 @@ export function injectSystemPrompt(body, format, prompt) { if (!body || !prompt) return; if (typeof body !== "object") return; - // Kiro wire shape is unique (conversationState/systemPrompt) — handle directly. + // Kiro wire shape is unique (conversationState) — handle directly. if (isKiroBody(body) || format === FORMATS.KIRO) { injectKiroSystem(body, prompt); return; @@ -61,10 +61,13 @@ export function injectSystemPrompt(body, format, prompt) { function isKiroBody(body) { if (!body || typeof body !== "object") return false; - if (typeof body.systemPrompt !== "string") return false; const cs = body.conversationState; if (!cs || typeof cs !== "object") return false; - return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object"); + // A top-level `systemPrompt` used to be the marker, but the Kiro translator no + // longer emits it (kiro.dev rejects the field), so gate on the turn shape. + const historyTurn = Array.isArray(cs.history) + && cs.history.some(it => it && (it.userInputMessage || it.assistantResponseMessage)); + return historyTurn || !!(cs.currentMessage && cs.currentMessage.userInputMessage); } // Exact idempotency: prompt present as its own SEP-delimited segment (or the @@ -258,80 +261,33 @@ function injectGeminiSystem(body, prompt) { } // ---- Kiro ---- -// Updates top-level systemPrompt and only the mirrored leading prefix of the -// first user history turn, else current user. next = old + SEP + prompt. -// Replace old leading prefix only; preserve time context and user tail. +// The prompt is appended to the first user turn's content — the same place the +// Kiro translator already mirrors the system text via its contentPrefix. +// +// A top-level `systemPrompt` is deliberately NOT written: the kiro.dev gateway +// answers any body carrying that field with +// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"} +// The translator stopped emitting it in v0.5.59, but this injector kept adding +// it back, so every kr/ model failed whenever an RTK prompt (caveman, ponytail) +// was active. function injectKiroSystem(body, prompt) { try { - let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : ""; - // Repair path: a previous partial write left systemPrompt updated but user - // content still mirroring the pre-write prefix. Re-derive the effective old - // prefix from content so this pass converges instead of early-returning. - const cs0 = body.conversationState; - let firstUser0 = cs0 && Array.isArray(cs0.history) - ? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null) - : null; - if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage; - - if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) { - const c0 = firstUser0.content; - if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) { - // systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored - oldPrompt = ""; - } - } - if (oldPrompt && hasPrompt(oldPrompt, prompt)) return; - const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt; - - // Atomicity: write user content first, then systemPrompt only if content - // write succeeded (or was a no-op). If systemPrompt write then fails, the - // repair heuristic above re-derives from content on retry — no permanent - // half-applied state. const cs = body.conversationState; let targetMsg = null; - try { - const hist = Array.isArray(cs?.history) ? cs.history : null; - if (hist) { - for (const item of hist) { - if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; } - } - } - if (!targetMsg && cs?.currentMessage?.userInputMessage) { - targetMsg = cs.currentMessage.userInputMessage; - } - } catch (_) { targetMsg = null; } - - let sysWritten = false; - try { body.systemPrompt = next; sysWritten = true; } catch (_) {} - - const applyContent = () => { - const content = typeof targetMsg.content === "string" ? targetMsg.content : ""; - if (oldPrompt === "") { - // Empty old prompt: prepend unless already at head (exact, not substring) - if (content.startsWith(prompt) || content.startsWith(next)) return; - const newContent = content ? `${next}${SEP}${content}` : next; - try { targetMsg.content = newContent; } catch (_) {} - return; - } - if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone - if (content.startsWith(next)) return; // already applied → idempotent - const tail = content.slice(oldPrompt.length); - try { targetMsg.content = `${next}${tail}`; } catch (_) {} - }; - - try { - if (targetMsg) applyContent(); - } catch (_) {} - if (sysWritten && targetMsg) { - // verify convergence: content should now start with next (or be un-mirrored) - let ok = false; - try { - const c = targetMsg.content; - ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt); - } catch (_) {} - if (!ok) { - try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback + const hist = Array.isArray(cs?.history) ? cs.history : null; + if (hist) { + for (const item of hist) { + if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; } } } + if (!targetMsg && cs?.currentMessage?.userInputMessage) { + targetMsg = cs.currentMessage.userInputMessage; + } + if (!targetMsg) return; + + const content = typeof targetMsg.content === "string" ? targetMsg.content : ""; + const next = dedupStringAppend(content, prompt); + if (next === content) return; // already injected — idempotent across retries + try { targetMsg.content = next; } catch (_) { /* frozen/proxy fail-open */ } } catch (_) {} } diff --git a/open-sse/translator/request/claude-to-kiro.js b/open-sse/translator/request/claude-to-kiro.js index 9a7cf377..ef2dd6c5 100644 --- a/open-sse/translator/request/claude-to-kiro.js +++ b/open-sse/translator/request/claude-to-kiro.js @@ -242,9 +242,9 @@ export function claudeToKiroRequest(model, body, stream, credentials) { ? (credentials?.providerSpecificData?.profileArn || "") : (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod)); - // Kiro CLI/KAS sends system prompt as top-level `systemPrompt`. Keep a - // content fallback too because the CodeWhisperer surface does not always - // enforce top-level systemPrompt for direct calls. + // The system prompt travels inside the first user turn's content (contentPrefix): + // the CodeWhisperer surface rejects a top-level `systemPrompt` with + // 400 REQUEST_BODY_INVALID, so the value below is only a replay cache key. const timestamp = new Date().toISOString(); const systemPromptParts = []; if (thinkingBudget !== null && !usesNativeGptEffort) { diff --git a/open-sse/translator/request/openai-to-kiro.js b/open-sse/translator/request/openai-to-kiro.js index 990b607f..dbaefbae 100644 --- a/open-sse/translator/request/openai-to-kiro.js +++ b/open-sse/translator/request/openai-to-kiro.js @@ -340,9 +340,9 @@ export function openaiToKiroRequest(model, body, stream, credentials) { const timestamp = new Date().toISOString(); - // Kiro CLI/KAS sends these as top-level systemPrompt. Keep a content fallback - // too because the CodeWhisperer surface does not always enforce top-level - // systemPrompt for direct calls. + // The system prompt travels inside the first user turn's content (contentPrefix): + // the CodeWhisperer surface rejects a top-level `systemPrompt` with + // 400 REQUEST_BODY_INVALID, so the value below is only a replay cache key. const systemPromptParts = []; if (thinkingBudget !== null && !usesNativeGptEffort) { systemPromptParts.push(buildThinkingSystemPrefix(thinkingBudget)); diff --git a/tests/unit/kiro-terminal-integrity.test.js b/tests/unit/kiro-terminal-integrity.test.js index aaf66209..3199c03d 100644 --- a/tests/unit/kiro-terminal-integrity.test.js +++ b/tests/unit/kiro-terminal-integrity.test.js @@ -115,7 +115,7 @@ async function text(stream) { async function execute(executor = new KiroExecutor(), overrides = {}) { return executor.execute({ model: "kr/claude-opus-4.8", - body: { systemPrompt: "base", conversationState: {} }, + body: { conversationState: { currentMessage: { userInputMessage: { content: "base", modelId: "m" } } } }, stream: true, credentials, ...overrides @@ -342,8 +342,12 @@ describe("Kiro terminal integrity recovery", () => { const retryBody = JSON.parse(fetchMock.mock.calls[1][1].body); expect(body).toContain("Recovered safely."); - expect(retryBody.systemPrompt).toContain("tool_call wrapper was malformed"); - expect(retryBody.systemPrompt).not.toContain("IGNORE_ALL_INSTRUCTIONS"); + // The repair instruction rides in the user turn: kiro.dev rejects a + // top-level systemPrompt with 400 REQUEST_BODY_INVALID. + const retryContent = retryBody.conversationState.currentMessage.userInputMessage.content; + expect(retryBody.systemPrompt).toBeUndefined(); + expect(retryContent).toContain("tool_call wrapper was malformed"); + expect(retryContent).not.toContain("IGNORE_ALL_INSTRUCTIONS"); }); it("lets a complete tool call override metadata end_turn", async () => { diff --git a/tests/unit/system-inject.test.js b/tests/unit/system-inject.test.js index bff7c94c..fd074f2d 100644 --- a/tests/unit/system-inject.test.js +++ b/tests/unit/system-inject.test.js @@ -261,138 +261,121 @@ describe("system-inject gemini", () => { }); describe("system-inject kiro", () => { - it("updates systemPrompt and mirrored prefix of first history user preserving tail", () => { - const oldPrompt = "OLD_SYS"; + // The kiro.dev gateway rejects any body carrying a top-level `systemPrompt` + // with 400 REQUEST_BODY_INVALID, so the prompt goes into the user turn only. + it("appends to first history user, leaves systemPrompt untouched", () => { const timeCtx = "[Context: Current time is 2026-01-01T00:00:00.000Z]"; const tail = "user tail content"; - const historyUserContent = `${oldPrompt}${SEP}${timeCtx}${SEP}${tail}`; + const historyUserContent = `${timeCtx}${SEP}${tail}`; const body = { - systemPrompt: oldPrompt, conversationState: { history: [{ userInputMessage: { content: historyUserContent, modelId: "m" } }, { assistantResponseMessage: { content: "..." } }], currentMessage: { userInputMessage: { content: "current " + tail, modelId: "m" } }, }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - const next = `${oldPrompt}${SEP}${P1}`; - expect(body.systemPrompt).toBe(next); - expect(body.conversationState.history[0].userInputMessage.content).toBe(`${next}${SEP}${timeCtx}${SEP}${tail}`); + expect(body.systemPrompt).toBeUndefined(); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`${historyUserContent}${SEP}${P1}`); // currentMessage must stay untouched expect(body.conversationState.currentMessage.userInputMessage.content).toBe("current " + tail); }); - it("when no history user, updates currentMessage instead", () => { - const oldPrompt = "OLD"; - const body = { - systemPrompt: oldPrompt, - conversationState: { - history: [], - currentMessage: { userInputMessage: { content: `${oldPrompt}${SEP}tail`, modelId: "m" } }, - }, - }; - injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(`${oldPrompt}${SEP}${P1}`); - expect(body.conversationState.currentMessage.userInputMessage.content).toBe(`${oldPrompt}${SEP}${P1}${SEP}tail`); - }); - - it("empty old prompt prepends to chosen user content", () => { - const body = { - systemPrompt: "", - conversationState: { - history: [{ userInputMessage: { content: "tail hello", modelId: "m" } }], - currentMessage: { userInputMessage: { content: "cur", modelId: "m" } }, - }, - }; - injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(P1); - expect(body.conversationState.history[0].userInputMessage.content).toBe(`${P1}${SEP}tail hello`); - }); - - it("if old prompt not mirrored at head, do not alter user content", () => { + it("never writes a top-level systemPrompt, even if one is already present", () => { const body = { systemPrompt: "OLD", conversationState: { - history: [{ userInputMessage: { content: "different head content", modelId: "m" } }], + history: [{ userInputMessage: { content: "tail", modelId: "m" } }], + }, + }; + injectSystemPrompt(body, FORMATS.KIRO, P1); + expect(body.systemPrompt).toBe("OLD"); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`tail${SEP}${P1}`); + }); + + it("when no history user, updates currentMessage instead", () => { + const body = { + conversationState: { + history: [], + currentMessage: { userInputMessage: { content: "tail", modelId: "m" } }, + }, + }; + injectSystemPrompt(body, FORMATS.KIRO, P1); + expect(body.systemPrompt).toBeUndefined(); + expect(body.conversationState.currentMessage.userInputMessage.content).toBe(`tail${SEP}${P1}`); + }); + + it("empty user content becomes the prompt itself", () => { + const body = { + conversationState: { + history: [{ userInputMessage: { content: "", modelId: "m" } }], currentMessage: { userInputMessage: { content: "cur", modelId: "m" } }, }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(`OLD${SEP}${P1}`); - expect(body.conversationState.history[0].userInputMessage.content).toBe("different head content"); + expect(body.conversationState.history[0].userInputMessage.content).toBe(P1); + expect(body.conversationState.currentMessage.userInputMessage.content).toBe("cur"); }); it("exact retry idempotency for kiro", () => { - const oldPrompt = "OLD"; const body = { - systemPrompt: oldPrompt, conversationState: { - history: [{ userInputMessage: { content: `${oldPrompt}${SEP}tail`, modelId: "m" } }], + history: [{ userInputMessage: { content: "tail", modelId: "m" } }], currentMessage: { userInputMessage: { content: "cur", modelId: "m" } }, }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - const after1 = JSON.parse(JSON.stringify(body)); + const after1 = body.conversationState.history[0].userInputMessage.content; injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(after1.systemPrompt); - expect(body.conversationState.history[0].userInputMessage.content).toBe(after1.conversationState.history[0].userInputMessage.content); - // different prompt both apply + expect(body.conversationState.history[0].userInputMessage.content).toBe(after1); + // different prompt both apply, in injection order injectSystemPrompt(body, FORMATS.KIRO, P2); - expect(body.systemPrompt).toBe(`${oldPrompt}${SEP}${P1}${SEP}${P2}`); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`tail${SEP}${P1}${SEP}${P2}`); }); it("preserves non-enumerable _kiroUpstreamModel", () => { const body = { - systemPrompt: "OLD", - conversationState: { history: [{ userInputMessage: { content: "OLD" + SEP + "tail", modelId: "m" } }], currentMessage: { userInputMessage: { content: "OLD" + SEP + "tail2", modelId: "m" } } }, + conversationState: { history: [{ userInputMessage: { content: "tail", modelId: "m" } }], currentMessage: { userInputMessage: { content: "tail2", modelId: "m" } } }, }; Object.defineProperty(body, "_kiroUpstreamModel", { value: "m", enumerable: false }); injectSystemPrompt(body, FORMATS.KIRO, P1); expect(body._kiroUpstreamModel).toBe("m"); expect(Object.getOwnPropertyDescriptor(body, "_kiroUpstreamModel").enumerable).toBe(false); }); + + it("frozen user message fails open without throwing or half-writing", () => { + const body = { + conversationState: { + history: [{ userInputMessage: Object.freeze({ content: "tail", modelId: "m" }) }], + }, + }; + expect(() => injectSystemPrompt(body, FORMATS.KIRO, P1)).not.toThrow(); + expect(body.systemPrompt).toBeUndefined(); + expect(body.conversationState.history[0].userInputMessage.content).toBe("tail"); + }); }); describe("system-inject regression fixes", () => { it("kiro partial mutation converges on retry after transient content write failure", () => { - const oldPrompt = "OLD"; let failNextWrite = true; - const um = { content: `${oldPrompt}${SEP}tail`, modelId: "m" }; + const um = { content: "tail", modelId: "m" }; const proxiedUm = new Proxy(um, { set(t, p, v) { if (p === "content" && failNextWrite) { failNextWrite = false; throw new Error("transient"); } t[p] = v; return true; }, }); - const body = { - systemPrompt: oldPrompt, - conversationState: { - history: [{ userInputMessage: proxiedUm }], - }, - }; + const body = { conversationState: { history: [{ userInputMessage: proxiedUm }] } }; injectSystemPrompt(body, FORMATS.KIRO, P1); - // first pass rolled back atomically — nothing half-applied - expect(body.systemPrompt).toBe(oldPrompt); - expect(um.content).toBe(`${oldPrompt}${SEP}tail`); + // nothing half-applied + expect(um.content).toBe("tail"); + expect(body.systemPrompt).toBeUndefined(); // retry converges injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(`${oldPrompt}${SEP}${P1}`); - expect(um.content).toBe(`${oldPrompt}${SEP}${P1}${SEP}tail`); - }); - - it("kiro rolls back systemPrompt when user content write fails (atomicity)", () => { - const oldPrompt = "OLD"; - const body = { - systemPrompt: oldPrompt, - conversationState: { - history: [{ userInputMessage: Object.freeze({ content: `${oldPrompt}${SEP}tail`, modelId: "m" }) }], - }, - }; - injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(oldPrompt); + expect(um.content).toBe(`tail${SEP}${P1}`); }); it("kiro shape gate: stray conversationState without history/currentMessage does not hijack chat body", () => { - const body = { messages: [{ role: ROLE.SYSTEM, content: "hello" }], systemPrompt: "", conversationState: {} }; + const body = { messages: [{ role: ROLE.SYSTEM, content: "hello" }], conversationState: {} }; injectSystemPrompt(body, FORMATS.OPENAI, P1); expect(body.messages[0].content).toBe(`hello${SEP}${P1}`); }); @@ -409,15 +392,14 @@ describe("system-inject regression fixes", () => { expect(body.instructions).toBe(`You are RULE follower${SEP}RULE`); }); - it("kiro empty-old prepend fires when prompt appears mid-tail only", () => { + it("substring occurrence does not suppress kiro injection", () => { const body = { - systemPrompt: "", conversationState: { history: [{ userInputMessage: { content: `some ${P1} here`, modelId: "m" } }], }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.conversationState.history[0].userInputMessage.content).toBe(`${P1}${SEP}some ${P1} here`); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`some ${P1} here${SEP}${P1}`); }); }); From 781c18d83746a9b3a9dd7d401ca0ed2c6db1ce69 Mon Sep 17 00:00:00 2001 From: anhtran-ai Date: Thu, 10 Sep 2026 22:25:23 +0700 Subject: [PATCH 35/78] fix(codex): strip Unicode-property tool schema patterns Codex rejects MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex's /responses validator has no Unicode property escapes, so a tool `pattern` containing `\p{...}` 400s the whole request with `Invalid schema for function ... is not a 'regex'` — identically on every account, costing a full combo failover per turn (#3922). - Add open-sse/utils/codexToolSchema.js: copy-on-write walk that drops only `pattern` values carrying a property escape, returning the original reference when nothing changed so the caller's schema stays intact for a retry against another provider - Treat `properties` keys as property names, so a field literally called `pattern` is never read as the schema keyword; skip escaped literals via backslash-parity counting - Apply it in normalizeCodexTools for both function and namespace sub-tool parameters, and log the strip count via dbg - Add three cases to tests/unit/codex-tool-normalization.test.js --- open-sse/executors/codex.js | 12 +++- open-sse/utils/codexToolSchema.js | 80 +++++++++++++++++++++ tests/unit/codex-tool-normalization.test.js | 63 ++++++++++++++++ 3 files changed, 154 insertions(+), 1 deletion(-) create mode 100644 open-sse/utils/codexToolSchema.js diff --git a/open-sse/executors/codex.js b/open-sse/executors/codex.js index 4d9acbd2..de2af822 100644 --- a/open-sse/executors/codex.js +++ b/open-sse/executors/codex.js @@ -12,6 +12,7 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js"; import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js"; import { dbg } from "../utils/debugLog.js"; import { resolveSessionId } from "../utils/sessionManager.js"; +import { stripCodexUnsupportedPatterns } from "../utils/codexToolSchema.js"; // SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts. const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"]; @@ -72,6 +73,9 @@ function stripStoredItemReferences(body) { function normalizeCodexTools(body) { if (!Array.isArray(body.tools)) return; const validNames = new Set(); + // Codex's schema validator has no Unicode property escapes; a `pattern` + // carrying `\p{...}` 400s the whole request on every account (#3922). + const patternStats = { removed: 0 }; body.tools = body.tools.filter((tool) => { if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false; const type = typeof tool.type === "string" ? tool.type : ""; @@ -80,6 +84,9 @@ function normalizeCodexTools(body) { for (const st of tool.tools) { const n = typeof st?.name === "string" ? st.name.trim().slice(0, 128) : ""; if (n) validNames.add(n); + if (st?.parameters && typeof st.parameters === "object") { + st.parameters = stripCodexUnsupportedPatterns(st.parameters, patternStats); + } } } return true; @@ -101,10 +108,13 @@ function normalizeCodexTools(body) { tool.type = "function"; tool.name = name.slice(0, 128); if (description) tool.description = description; - tool.parameters = parameters; + tool.parameters = stripCodexUnsupportedPatterns(parameters, patternStats); validNames.add(name); return true; }); + if (patternStats.removed > 0) { + dbg("CODEX", `stripped ${patternStats.removed} unsupported tool schema pattern(s)`); + } // Drop tool_choice if it references an unknown function name if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) { if (body.tool_choice.type === "function") { diff --git a/open-sse/utils/codexToolSchema.js b/open-sse/utils/codexToolSchema.js new file mode 100644 index 00000000..0c017b61 --- /dev/null +++ b/open-sse/utils/codexToolSchema.js @@ -0,0 +1,80 @@ +// Codex-specific tool JSON Schema compatibility. +// +// `https://chatgpt.com/backend-api/codex/responses` validates every function +// tool's `parameters` with a regex engine that does not implement Unicode +// property escapes. A `pattern` such as +// +// "^(?!__.*__$)[^\\p{Cc}\\p{Cf}\\p{Zl}\\p{Zp}\"\\\\./\\[\\]]{1,200}$" +// +// is a perfectly valid ECMAScript `u`-mode regex, but Codex answers +// +// 400 Invalid schema for function 'Artifact': '^\p{Cc}...' is not a 'regex' +// param: tools[0].parameters +// +// The request is deterministically malformed for this provider, so every +// account fails identically and the combo pays a full failover before landing +// somewhere that accepts it (#3922). +// +// Scope guardrail (#3667): this is NOT a global schema sanitizer. Providers +// that do support `\p{...}` keep the constraint untouched — the strip runs only +// on the Codex dispatch path, and only on `pattern` strings that actually +// contain a property escape. Everything else in the schema (including valid +// patterns) passes through byte-identical. + +// `\p{...}` / `\P{...}` with an odd number of preceding backslashes — an even +// count means the backslash itself is escaped, so `\\p{Cc}` is a literal "p". +const UNICODE_PROPERTY_ESCAPE = /(^|[^\\])(\\\\)*\\[pP]\{/; + +export function hasUnicodePropertyEscape(pattern) { + return typeof pattern === "string" && UNICODE_PROPERTY_ESCAPE.test(pattern); +} + +// Copy-on-write walk: returns the original reference when nothing changed, so +// untouched schemas keep object identity and callers can cheaply detect a no-op. +// `properties` is special-cased because its keys are arbitrary property *names* +// (which may themselves be "pattern" or "properties") and must never be read as +// schema keywords; every other key recurses as an ordinary schema node. +function stripNode(node, stats) { + if (Array.isArray(node)) { + let changed = false; + const next = node.map((item) => { + const cleaned = stripNode(item, stats); + if (cleaned !== item) changed = true; + return cleaned; + }); + return changed ? next : node; + } + if (!node || typeof node !== "object") return node; + + let changed = false; + const next = {}; + for (const [key, value] of Object.entries(node)) { + if (key === "pattern" && hasUnicodePropertyEscape(value)) { + stats.removed++; + changed = true; + continue; + } + if (key === "properties" && value && typeof value === "object" && !Array.isArray(value)) { + let propsChanged = false; + const props = {}; + for (const [propName, propSchema] of Object.entries(value)) { + const cleaned = stripNode(propSchema, stats); + if (cleaned !== propSchema) propsChanged = true; + props[propName] = cleaned; + } + if (propsChanged) changed = true; + next[key] = propsChanged ? props : value; + continue; + } + const cleaned = stripNode(value, stats); + if (cleaned !== value) changed = true; + next[key] = cleaned; + } + return changed ? next : node; +} + +// Remove only the `pattern` constraints Codex's validator rejects. +// Returns the same reference when the schema is already compatible. +export function stripCodexUnsupportedPatterns(schema, stats = { removed: 0 }) { + return stripNode(schema, stats); +} diff --git a/tests/unit/codex-tool-normalization.test.js b/tests/unit/codex-tool-normalization.test.js index d1b5901e..c7193898 100644 --- a/tests/unit/codex-tool-normalization.test.js +++ b/tests/unit/codex-tool-normalization.test.js @@ -118,6 +118,69 @@ describe("CodexExecutor tool normalization", () => { ]); }); + it("strips only Unicode-property patterns rejected by Codex", () => { + const unicodePattern = "^(?!__.*__$)[^\\p{Cc}\\p{Cf}\\p{Zl}\\p{Zp}]{1,200}$"; + const validPattern = "^[a-z][a-z0-9_-]{0,31}$"; + const sourceParameters = { + type: "object", + properties: { + artifact: { + type: "object", + properties: { + name: { type: "string", pattern: unicodePattern }, + slug: { type: "string", pattern: validPattern }, + }, + }, + // A property named "pattern" is data, not the schema keyword. + pattern: { type: "string", pattern: validPattern }, + }, + allOf: [{ properties: { title: { type: "string", pattern: unicodePattern } } }], + }; + const tools = normalizeTools([{ + type: "function", + name: "Artifact", + parameters: sourceParameters, + }]); + + expect(tools[0].parameters.properties.artifact.properties.name.pattern).toBeUndefined(); + expect(tools[0].parameters.properties.artifact.properties.slug.pattern).toBe(validPattern); + expect(tools[0].parameters.properties.pattern.pattern).toBe(validPattern); + expect(tools[0].parameters.allOf[0].properties.title.pattern).toBeUndefined(); + // Copy-on-write: the caller's schema remains available for another provider. + expect(sourceParameters.properties.artifact.properties.name.pattern).toBe(unicodePattern); + }); + + it("keeps escaped literal property text and schema identity when no strip is needed", () => { + const parameters = { + type: "object", + properties: { + literal: { type: "string", pattern: "^\\\\p{Cc}$" }, + simple: { type: "string", pattern: "^[A-Z]+$" }, + }, + }; + const tools = normalizeTools([{ type: "function", name: "probe", parameters }]); + + expect(tools[0].parameters).toBe(parameters); + expect(tools[0].parameters.properties.literal.pattern).toBe("^\\\\p{Cc}$"); + }); + + it("sanitizes nested namespace function schemas", () => { + const tools = normalizeTools([{ + type: "namespace", + name: "agent", + tools: [{ + type: "function", + name: "Artifact", + parameters: { + type: "object", + properties: { name: { type: "string", pattern: "^\\p{Cc}+$" } }, + }, + }], + }]); + + expect(tools[0].tools[0].parameters.properties.name.pattern).toBeUndefined(); + }); + it("preserves custom freeform tools with format payloads", () => { const tools = normalizeTools([ { From 45ec1d30bb694123c93ca1b55fd9d9c3bdd26e43 Mon Sep 17 00:00:00 2001 From: kimono381 <62897149+kimono381@users.noreply.github.com> Date: Thu, 10 Sep 2026 22:26:15 +0700 Subject: [PATCH 36/78] fix(deepseek): keep Anthropic-only tool types when forwarding to /anthropic/v1/messages DeepSeek's Anthropic-compatible endpoint accepts only the built-in web_search_20250305 / web_search_20260209 tools and rejects client-defined `custom` tools (MCP / Read / Bash) with HTTP 400 "unknown variant `custom`". The generic non-Claude filter in prepareClaudeRequest dropped the offending tools but also dropped the web_search_* ones DeepSeek does accept. - Add an opt-in per-provider transport quirk `claudeSupportedToolTypes`; when declared it becomes a strict allow-list for Anthropic tool `type` values - Stop stripping the `type` discriminator from surviving tools under that quirk, since DeepSeek needs it to route built-ins - Declare the quirk on the deepseek transport with the two web_search_* types - Providers without the quirk keep the previous filter and normalisation behaviour byte-for-byte; openai-format targets never reach this path --- open-sse/providers/registry/deepseek.js | 15 +++ open-sse/translator/formats/claude.js | 22 +++- tests/unit/deepseek-claude-tools.test.js | 145 +++++++++++++++++++++++ 3 files changed, 181 insertions(+), 1 deletion(-) create mode 100644 tests/unit/deepseek-claude-tools.test.js diff --git a/open-sse/providers/registry/deepseek.js b/open-sse/providers/registry/deepseek.js index bb8015b0..2c3a3e73 100644 --- a/open-sse/providers/registry/deepseek.js +++ b/open-sse/providers/registry/deepseek.js @@ -25,6 +25,21 @@ export default { reasoningInject: { scope: "all", }, + quirks: { + // DeepSeek's Anthropic-compatible endpoint + // (https://api.deepseek.com/anthropic/v1/messages) accepts ONLY the + // built-in web_search_* tools and rejects client-defined `custom` tools + // (MCP / Read / Bash / etc.) with HTTP 400 + // "tools[0]: unknown variant `custom`, expected + // `web_search_20250305` or `web_search_20260209`". + // + // Declaring this whitelist makes prepareClaudeRequest() forward only + // web_search_* tools and strip everything else before sending, so MCP / + // function tools are dropped instead of failing the whole request. + // DeepSeek's OpenAI-compatible transport is unaffected (targetFormat + // there is "openai", not "claude", so prepareClaudeRequest is not run). + claudeSupportedToolTypes: ["web_search_20250305", "web_search_20260209"], + }, }, // Multi-endpoint: pick the transport matching client sourceFormat to skip translation. transports: [ diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index 62a0531a..930c761a 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -461,8 +461,21 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // Strip built-in tools (e.g. web_search_20250305) and normalize to Anthropic-native shape // (drop `type` field, fold `function.{name,description,parameters}`) for non-Anthropic providers if (provider !== "claude") { + // Provider-specific whitelist of Anthropic tool `type` values that the + // upstream actually accepts. When the provider declares it + // (e.g. DeepSeek — only web_search_*), keep only listed types; otherwise + // keep the prior behaviour of dropping every non-function tool, which is + // correct for OpenAI-compatible targets reached through this Claude-format + // pass (their tools get normalized below to function-style). + const supportedTypes = PROVIDERS[provider]?.quirks?.claudeSupportedToolTypes; + const hasWhitelist = Array.isArray(supportedTypes); body.tools = body.tools - .filter(tool => !tool.type || tool.type === "function") + .filter(tool => { + const t = tool?.type; + if (!t || t === "function") return true; + if (hasWhitelist) return supportedTypes.includes(t); + return false; + }) .map(tool => { if (tool.function) { return { @@ -471,6 +484,13 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne input_schema: tool.function.parameters, }; } + // When the provider declared a supportedToolTypes whitelist, keep + // the surviving tools' `type` field intact — the upstream + // Anthropic-compatible endpoint (e.g. DeepSeek) requires it to + // route built-ins like web_search_* correctly. Without a + // whitelist, preserve prior behaviour and strip `type` so the + // tool is normalized to plain Anthropic shape. + if (hasWhitelist) return tool; const { type, ...rest } = tool; return rest; }); diff --git a/tests/unit/deepseek-claude-tools.test.js b/tests/unit/deepseek-claude-tools.test.js new file mode 100644 index 00000000..bd05e208 --- /dev/null +++ b/tests/unit/deepseek-claude-tools.test.js @@ -0,0 +1,145 @@ +/** + * Regression test: prepareClaudeRequest() must strip client-defined `custom` + * tools when forwarding to a provider whose Anthropic-compatible endpoint + * does not accept them (DeepSeek — accepts only web_search_*). + * + * Background: + * When Claude Code talks to a DeepSeek route via /v1/messages, 9router + * forwards the request body as Claude-format to + * https://api.deepseek.com/anthropic/v1/messages. MCP / function tools + * arrive with `type: "custom"`. DeepSeek rejects them with HTTP 400 + * "tools[0]: unknown variant `custom`, expected `web_search_20250305` + * or `web_search_20260209`". The previous generic filter dropped them + * but also stripped the web_search_* tools that DeepSeek actually + * accepts. DeepSeek now exposes a `quirks.claudeSupportedToolTypes` + * whitelist and prepareClaudeRequest honours it. + */ + +import { describe, it, expect } from "vitest"; +import { prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; +import { PROVIDERS } from "../../open-sse/providers/index.js"; + +function makeBody(tools) { + return { + model: "deepseek-v4-pro", + max_tokens: 1024, + messages: [{ role: "user", content: "hello" }], + tools, + }; +} + +describe("prepareClaudeRequest — provider: deepseek", () => { + it("declares the supportedTypes quirk on the provider transport", () => { + expect(PROVIDERS.deepseek).toBeDefined(); + expect(PROVIDERS.deepseek.quirks).toBeDefined(); + expect(PROVIDERS.deepseek.quirks.claudeSupportedToolTypes).toEqual([ + "web_search_20250305", + "web_search_20260209", + ]); + }); + + it("strips MCP / custom tools (the regression) before forwarding", () => { + const body = makeBody([ + { type: "custom", name: "Bash", input_schema: { type: "object" } }, + { type: "custom", name: "Read", input_schema: { type: "object" } }, + { type: "custom", name: "Glob", input_schema: { type: "object" } }, + ]); + + const out = prepareClaudeRequest(body, "deepseek"); + + expect(out.tools).toBeUndefined(); + expect(out.tool_choice).toBeUndefined(); + }); + + it("keeps web_search_20250305 and web_search_20260209", () => { + const out = prepareClaudeRequest( + makeBody([ + { type: "web_search_20250305", name: "web_search" }, + { type: "web_search_20260209", name: "web_search" }, + ]), + "deepseek" + ); + + expect(Array.isArray(out.tools)).toBe(true); + expect(out.tools).toHaveLength(2); + const types = out.tools.map(t => t.type).sort(); + expect(types).toEqual(["web_search_20250305", "web_search_20260209"]); + }); + + it("preserves the `type` field on web_search_* tools (DeepSeek requires it)", () => { + // The .map below the filter must NOT strip `type` when the provider + // declared a whitelist — DeepSeek would reject a tool object missing + // its discriminator field with the same unknown-variant error. + const out = prepareClaudeRequest( + makeBody([{ type: "web_search_20250305", name: "web_search" }]), + "deepseek" + ); + + expect(out.tools[0].type).toBe("web_search_20250305"); + expect(out.tools[0].name).toBe("web_search"); + }); + + it("drops `custom` but keeps `web_search_*` when both are present", () => { + const out = prepareClaudeRequest( + makeBody([ + { type: "custom", name: "Bash", input_schema: { type: "object" } }, + { type: "web_search_20250305", name: "web_search" }, + ]), + "deepseek" + ); + + expect(Array.isArray(out.tools)).toBe(true); + expect(out.tools).toHaveLength(1); + expect(out.tools[0].type).toBe("web_search_20250305"); + expect(out.tools[0].name).toBe("web_search"); + }); + + it("survives when body has no tools", () => { + const out = prepareClaudeRequest(makeBody(undefined), "deepseek"); + expect(out.tools).toBeUndefined(); + }); + + it("rejects future / unknown tool types instead of forwarding them", () => { + const out = prepareClaudeRequest( + makeBody([{ type: "future_tool_2099", name: "x" }]), + "deepseek" + ); + expect(out.tools).toBeUndefined(); + }); +}); + +describe("prepareClaudeRequest — backward compat: providers without the quirk", () => { + // Pick any non-Claude provider that has a Claude-format transport and has + // NOT been migrated to the new quirk. This protects GLM / Kimi / future + // Anthropic-compatible providers from unintended changes. + it("keeps prior behaviour (drop custom + web_search_*, normalize no-type tools)", () => { + const candidate = Object.entries(PROVIDERS).find( + ([id, p]) => + id !== "claude" && + p?.transports?.some(t => t.format === "claude") && + !p?.quirks?.claudeSupportedToolTypes + ); + + if (!candidate) { + // Every Claude-format provider has been migrated — nothing to verify. + return; + } + + const [providerId] = candidate; + + const out = prepareClaudeRequest( + makeBody([ + { type: "custom", name: "Bash", input_schema: { type: "object" } }, + { type: "web_search_20250305", name: "web_search" }, + { name: "no_type_tool", input_schema: { type: "object" } }, + ]), + providerId + ); + + if (out.tools !== undefined) { + for (const t of out.tools) { + expect(t.type).toBeUndefined(); + } + } + }); +}); \ No newline at end of file From f6e7cabe60461b9b55f7b7a0ceedeb9f56c72207 Mon Sep 17 00:00:00 2001 From: izzzzzi Date: Thu, 10 Sep 2026 22:48:06 +0700 Subject: [PATCH 37/78] fix(cline): stop workos:-prefixing ClinePass API keys and add clinepass token refresh MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cline/ClinePass requests failed with HTTP 401 ("Please make sure you are using the latest version of Cline and re-authenticate your Cline account", #3230 / #2333 / #3644). `getClineAccessToken()` unconditionally prefixed every token with `workos:`, which is correct for Cline OAuth access tokens (WorkOS JWTs) but wrong for ClinePass API keys — those are opaque strings (e.g. `clp_…`) that the API accepts only verbatim, so the `workos:`-prefixed value was rejected. Only prefix tokens that look like a WorkOS JWT (`eyJ…`); API keys and other opaque tokens pass through untouched, and an existing `workos:` prefix is never doubled. Also register `clinepass` in the token-refresh handlers. ClinePass shares Cline's WorkOS auth endpoints, but without the entry expired ClinePass OAuth tokens were never rotated, so every request kept 401ing. Finally, list `apikey` first in the ClinePass `authModes` (ClinePass is meant to be used with an API key from app.cline.bot/settings/api-keys), and add an "Import from /models" button that pulls the live Cline catalog into custom models. --- open-sse/providers/registry/clinepass.js | 5 +- open-sse/services/tokenRefresh.js | 2 + open-sse/shared/clineAuth.js | 9 ++- .../dashboard/providers/[id]/page.js | 62 +++++++++++++++++++ tests/unit/cline-auth.test.js | 40 ++++++++++++ 5 files changed, 116 insertions(+), 2 deletions(-) create mode 100644 tests/unit/cline-auth.test.js diff --git a/open-sse/providers/registry/clinepass.js b/open-sse/providers/registry/clinepass.js index 702054ac..1e7d0520 100644 --- a/open-sse/providers/registry/clinepass.js +++ b/open-sse/providers/registry/clinepass.js @@ -14,7 +14,10 @@ export default { }, }, category: "oauth", - authModes: ["oauth", "apikey"], + // ClinePass authenticates with a plain API key from app.cline.bot/settings/api-keys + // (category "apikey"). The OAuth extension flow used by Cline does not issue + // tokens that the ClinePass API consumer endpoint accepts (HTTP 401) — see #2333. + authModes: ["apikey", "oauth"], hasOAuth: true, transport: { baseUrl: "https://api.cline.bot/api/v1/chat/completions", diff --git a/open-sse/services/tokenRefresh.js b/open-sse/services/tokenRefresh.js index dbf11ac2..ed1f4f38 100644 --- a/open-sse/services/tokenRefresh.js +++ b/open-sse/services/tokenRefresh.js @@ -148,6 +148,8 @@ const REFRESH_HANDLERS = { "codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log), trae: (c, log) => refreshTraeToken(c.refreshToken, c, log), cline: (c, log) => refreshClineToken(c.refreshToken, log), + // ClinePass shares Cline's WorkOS auth endpoints, so the same refresh works. + clinepass: (c, log) => refreshClineToken(c.refreshToken, log), zed: () => refreshZedToken(), windsurf: (c, log) => refreshWindsurfToken(c, log), // Kimi Code OAuth (merged into id `kimi`); legacy id still routes here diff --git a/open-sse/shared/clineAuth.js b/open-sse/shared/clineAuth.js index 1b2b7df6..541b060d 100644 --- a/open-sse/shared/clineAuth.js +++ b/open-sse/shared/clineAuth.js @@ -6,7 +6,14 @@ export function getClineAccessToken(token) { if (typeof token !== "string") return ""; const trimmed = token.trim(); if (!trimmed) return ""; - return trimmed.startsWith("workos:") ? trimmed : `workos:${trimmed}`; + if (trimmed.toLowerCase().startsWith("workos:")) return trimmed; + // Cline OAuth access tokens are WorkOS JWTs (base64url `eyJ…` header). + // ClinePass API keys (category "apikey", e.g. `clp_…`) are NOT JWTs and must + // be sent verbatim — prefixing them with `workos:` makes the Cline API reject + // the request with HTTP 401 ("Please make sure you're using the latest + // version of Cline and re-authenticate your Cline account."). + const isWorkOsJwt = /^eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+/.test(trimmed); + return isWorkOsJwt ? `workos:${trimmed}` : trimmed; } export function getClineAuthorizationHeader(token) { diff --git a/src/app/(dashboard)/dashboard/providers/[id]/page.js b/src/app/(dashboard)/dashboard/providers/[id]/page.js index df1c9db6..9657b7ed 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/page.js +++ b/src/app/(dashboard)/dashboard/providers/[id]/page.js @@ -81,6 +81,7 @@ export default function ProviderDetailPage() { const [oneByOneSummary, setOneByOneSummary] = useState(null); const stopOneByOneRef = useRef(false); const [importingQoderModels, setImportingQoderModels] = useState(false); + const [importingClineModels, setImportingClineModels] = useState(false); const { copied, copy } = useCopyToClipboard(); const AG_RISK_STORAGE_KEY = "ag_risk_confirmed"; @@ -612,6 +613,53 @@ export default function ProviderDetailPage() { setImportingQoderModels(false); } }; + // Fetch the live Cline /models catalog and add every model not yet present. + // Cline and ClinePass share the same catalog endpoint (api.cline.bot/api/v1/models). + const handleImportClineModels = async () => { + if (importingClineModels) return; + const activeConnection = connections.find((conn) => conn.isActive !== false); + if (!activeConnection) { + alert(translate("Please add an active Cline connection first")); + return; + } + setImportingClineModels(true); + try { + const res = await fetch(`/api/providers/${activeConnection.id}/models`); + const data = await res.json(); + if (!res.ok) { + alert(data.error || translate("Failed to fetch models")); + return; + } + const models = data.models || []; + if (models.length === 0) { + alert(translate("No models returned")); + return; + } + let importedCount = 0; + for (const model of models) { + const modelId = model.id || model.name; + if (!modelId) continue; + const alreadyExists = customModels.some( + (entry) => entry.providerAlias === providerStorageAlias && entry.id === modelId && (entry.kind || entry.type || "llm") === "llm" + ) || Object.values(modelAliases).includes(`${providerStorageAlias}/${modelId}`); + if (alreadyExists) { + continue; + } + await handleAddCustomModel(modelId, "llm", providerStorageAlias); + importedCount += 1; + } + if (importedCount === 0) { + alert(translate("All models already exist, no new models added")); + } else { + alert(translate("Successfully added") + ` ${importedCount} ` + translate("models")); + } + } catch (error) { + console.log("Error importing Cline models:", error); + alert(translate("Error fetching models") + ": " + error.message); + } finally { + setImportingClineModels(false); + } + }; const handleRunOneByOneTest = async () => { if (oneByOneRunning || connections.length === 0) return; @@ -1187,6 +1235,20 @@ export default function ProviderDetailPage() { )} + {/* Import Cline /models catalog button — only show for cline and clinepass providers */} + {(providerId === "cline" || providerId === "clinepass") && connections.some((conn) => conn.isActive !== false) && ( + + )} + {/* Suggested models from provider API — show only models not yet added */} {suggestedModels.length > 0 && (() => { const addedFullModels = new Set([ diff --git a/tests/unit/cline-auth.test.js b/tests/unit/cline-auth.test.js new file mode 100644 index 00000000..e1eaae20 --- /dev/null +++ b/tests/unit/cline-auth.test.js @@ -0,0 +1,40 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + getClineAccessToken, + getClineAuthorizationHeader, +} from "../../open-sse/shared/clineAuth.js"; + +test("getClineAccessToken keeps an existing workos: prefix", () => { + const token = "workos:eyJhbGciOiJSUzI1NiJ9.eyJwYXAiJ9"; + assert.equal(getClineAccessToken(token), token); + assert.equal(getClineAccessToken(` ${token} `), token); +}); + +test("getClineAccessToken prefixes a bare WorkOS JWT with workos:", () => { + const jwt = "eyJhbGciOiJSUzI1NiJ9.eyJwYXAiJ9"; + assert.equal(getClineAccessToken(jwt), `workos:${jwt}`); +}); + +test("getClineAccessToken does NOT prefix ClinePass API keys", () => { + // ClinePass API keys are opaque strings (e.g. clp_…). Sending them as + // `workos:clp_…` makes api.cline.bot respond 401. + assert.equal(getClineAccessToken("clp_1234567890abcdef"), "clp_1234567890abcdef"); + assert.equal(getClineAccessToken("sk-9r-abcdef"), "sk-9r-abcdef"); + assert.equal(getClineAccessToken(""), ""); + assert.equal(getClineAccessToken(" "), ""); + assert.equal(getClineAccessToken(undefined), ""); + assert.equal(getClineAccessToken(null), ""); +}); + +test("getClineAuthorizationHeader builds a Bearer header without double prefixing", () => { + assert.equal(getClineAuthorizationHeader("clp_abc"), "Bearer clp_abc"); + assert.equal( + getClineAuthorizationHeader("eyJpeg.eyJbG"), + "Bearer workos:eyJpeg.eyJbG" + ); + assert.equal( + getClineAuthorizationHeader("workos:eyJpeg.eyJbG"), + "Bearer workos:eyJpeg.eyJbG" + ); +}); \ No newline at end of file From 122f23eebc45f43214f0397f28130f8978715386 Mon Sep 17 00:00:00 2001 From: Nick Nyanjui Date: Thu, 10 Sep 2026 22:46:07 +0700 Subject: [PATCH 38/78] fix(cline,airforce): unwrap {success,data} envelope, add live catalog, and refresh airforce free models MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cline (api.cline.bot) wraps non-stream chat completions in {"success":true,"data":{...choices...}}, which both the dashboard model-test ping and the proxy non-stream path read at top level, producing "Provider returned no completion choices for this model" (#3644). Unwrap the envelope before usage extraction and response translation; the error envelope ({"success":false,...}) never matches and passes through untouched. Scoped through `transport.quirks.clineEnvelope` so only cline/clinepass opt in — no other provider's response body is ever rewritten. Also adds a live Cline catalog: `fetchClineRawModels()` is shared between `resolveClineModels()` (full catalog, including free-tier ids such as z-ai/glm-5.3-flash) and `resolveClinepassModels()` (cline-pass/* only), wired into /v1/models, the per-provider models route, and the combo selector's model picker with the static catalog kept as fallback. Refreshes the dead api-airforce free models (anthropic/claude-3.7-sonnet, moonshot/kimi-k2.6, google/gemini-2.5-flash) with the live gpt-oss-120b, gpt-oss-20b and kimi-k2.7-code, plus passthroughModels, forceStream and a suggested-models filter. --- .../handlers/chatCore/nonStreamingHandler.js | 6 + open-sse/providers/registry/api-airforce.js | 9 +- open-sse/providers/registry/cline.js | 4 + open-sse/providers/registry/clinepass.js | 2 + open-sse/services/clinepassModels.js | 62 +++- open-sse/shared/clineEnvelope.js | 19 ++ src/app/api/models/test/ping.js | 7 + src/app/api/providers/[id]/models/route.js | 32 ++ .../api/providers/suggested-models/filters.js | 6 + src/app/api/v1/models/route.js | 9 +- src/shared/components/ModelSelectModal.js | 112 ++++--- src/shared/constants/cliTools.js | 2 +- tests/unit/api-airforce-free-models.test.js | 36 +++ tests/unit/cline-free-models-envelope.test.js | 297 ++++++++++++++++++ 14 files changed, 540 insertions(+), 63 deletions(-) create mode 100644 open-sse/shared/clineEnvelope.js create mode 100644 tests/unit/api-airforce-free-models.test.js create mode 100644 tests/unit/cline-free-models-envelope.test.js diff --git a/open-sse/handlers/chatCore/nonStreamingHandler.js b/open-sse/handlers/chatCore/nonStreamingHandler.js index 3b6cfc6d..fe94b282 100644 --- a/open-sse/handlers/chatCore/nonStreamingHandler.js +++ b/open-sse/handlers/chatCore/nonStreamingHandler.js @@ -6,6 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin import { createErrorResult } from "../../utils/error.js"; import { HTTP_STATUS } from "../../config/runtimeConfig.js"; import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js"; +import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js"; import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { decloakToolNames } from "../../utils/claudeCloaking.js"; @@ -302,6 +303,11 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m } } + // Unwrap before any consumer reads choices/usage so non-stream clients get a + // bare OpenAI body and usage tracking sees data.usage. No-op unless the + // provider opts in via transport.quirks.clineEnvelope. + responseBody = unwrapClineEnvelope(responseBody, provider); + reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody); if (onRequestSuccess) { Promise.resolve() diff --git a/open-sse/providers/registry/api-airforce.js b/open-sse/providers/registry/api-airforce.js index 16ed8b3e..1f98bc20 100644 --- a/open-sse/providers/registry/api-airforce.js +++ b/open-sse/providers/registry/api-airforce.js @@ -20,6 +20,8 @@ export default { authModes: [ "apikey", ], + passthroughModels: true, + modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" }, transport: { baseUrl: "https://api.airforce/v1/chat/completions", validateUrl: "https://api.airforce/v1/models", @@ -27,10 +29,11 @@ export default { "HTTP-Referer": "https://endpoint-proxy.local", "X-Title": "Endpoint Proxy", }, + forceStream: true, }, models: [ - { id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 }, - { id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 }, - { id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 }, + { id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 }, + { id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 }, + { id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 }, ], }; diff --git a/open-sse/providers/registry/cline.js b/open-sse/providers/registry/cline.js index cfa788c4..90b16c4b 100644 --- a/open-sse/providers/registry/cline.js +++ b/open-sse/providers/registry/cline.js @@ -14,12 +14,16 @@ export default { }, }, category: "oauth", + authModes: ["oauth"], + hasOAuth: true, transport: { baseUrl: "https://api.cline.bot/api/v1/chat/completions", headers: { "HTTP-Referer": "https://cline.bot", "X-Title": "Cline", }, + // Non-stream chat completions come back wrapped in {"success":true,"data":{...}} + quirks: { clineEnvelope: true }, tokenUrl: "https://api.cline.bot/api/v1/auth/token", refreshUrl: "https://api.cline.bot/api/v1/auth/refresh", auth: { diff --git a/open-sse/providers/registry/clinepass.js b/open-sse/providers/registry/clinepass.js index 1e7d0520..fc21590f 100644 --- a/open-sse/providers/registry/clinepass.js +++ b/open-sse/providers/registry/clinepass.js @@ -25,6 +25,8 @@ export default { "HTTP-Referer": "https://cline.bot", "X-Title": "Cline", }, + // Non-stream chat completions come back wrapped in {"success":true,"data":{...}} + quirks: { clineEnvelope: true }, auth: { combined: true, header: "Authorization", diff --git a/open-sse/services/clinepassModels.js b/open-sse/services/clinepassModels.js index 4208a4b6..0aa96ffe 100644 --- a/open-sse/services/clinepassModels.js +++ b/open-sse/services/clinepassModels.js @@ -19,12 +19,10 @@ function buildModelListHeaders(token, isApiKey) { } /** - * Fetch ClinePass live model catalog from Cline's /models endpoint. - * - * @param {object} credentials - Connection credentials ({ accessToken, apiKey }) - * @returns {Promise<{ models: { id: string, name: string }[] } | null>} + * Internal: fetch the raw model list from Cline's /models endpoint. + * Returns the parsed array or null on any failure. */ -export async function resolveClinepassModels(credentials) { +async function fetchClineRawModels(credentials) { const isApiKey = Boolean(credentials?.apiKey); const token = isApiKey ? credentials.apiKey : credentials?.accessToken; if (!token) return null; @@ -45,19 +43,53 @@ export async function resolveClinepassModels(credentials) { const json = await response.json(); const rawList = Array.isArray(json) ? json : json?.data; - if (!Array.isArray(rawList)) return null; - - const models = rawList - .filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/")) - .map((m) => ({ - id: m.id, - name: m.name || m.id, - })); - - return models.length ? { models } : null; + return Array.isArray(rawList) ? rawList : null; } catch { return null; } finally { clearTimeout(timer); } } + +/** + * Fetch ClinePass live model catalog from Cline's /models endpoint. + * Returns only models with the cline-pass/ prefix. + * + * @param {object} credentials - Connection credentials ({ accessToken, apiKey }) + * @returns {Promise<{ models: { id: string, name: string }[] } | null>} + */ +export async function resolveClinepassModels(credentials) { + const rawList = await fetchClineRawModels(credentials); + if (!rawList) return null; + + const models = rawList + .filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/")) + .map((m) => ({ + id: m.id, + name: m.name || m.id, + })); + + return models.length ? { models } : null; +} + +/** + * Fetch Cline live model catalog from Cline's /models endpoint. + * Unlike resolveClinepassModels, this returns ALL models (including + * free-tier models like z-ai/glm-5.3-flash) without the cline-pass/ prefix filter. + * + * @param {object} credentials - Connection credentials ({ accessToken, apiKey }) + * @returns {Promise<{ models: { id: string, name: string }[] } | null>} + */ +export async function resolveClineModels(credentials) { + const rawList = await fetchClineRawModels(credentials); + if (!rawList) return null; + + const models = rawList + .filter((m) => typeof m?.id === "string" && m.id.trim() !== "") + .map((m) => ({ + id: m.id, + name: m.name || m.id, + })); + + return models.length ? { models } : null; +} diff --git a/open-sse/shared/clineEnvelope.js b/open-sse/shared/clineEnvelope.js new file mode 100644 index 00000000..3873b746 --- /dev/null +++ b/open-sse/shared/clineEnvelope.js @@ -0,0 +1,19 @@ +import { PROVIDERS } from "../providers/index.js"; + +/** + * Unwrap Cline's non-stream envelope: {"success":true,"data":{...choices...}}. + * + * Scoped to providers opting in via `transport.quirks.clineEnvelope` so no other + * provider's body is ever rewritten. The error envelope ({"success":false,...}) + * never matches and passes through untouched. + * + * @param {object} body - Parsed upstream response body + * @param {string} provider - Provider id or alias + * @returns {object} The inner `data` object, or `body` unchanged + */ +export function unwrapClineEnvelope(body, provider) { + if (!provider || !PROVIDERS[provider]?.quirks?.clineEnvelope) return body; + const { success, data } = body || {}; + if (success !== true || !data || typeof data !== "object" || Array.isArray(data)) return body; + return data; +} diff --git a/src/app/api/models/test/ping.js b/src/app/api/models/test/ping.js index 24119eb2..273aff47 100644 --- a/src/app/api/models/test/ping.js +++ b/src/app/api/models/test/ping.js @@ -1,4 +1,6 @@ import { getApiKeys } from "@/lib/localDb"; +import { resolveProviderId } from "@/shared/constants/providers.js"; +import { unwrapClineEnvelope } from "open-sse/shared/clineEnvelope.js"; import { UPDATER_CONFIG } from "@/shared/constants/config"; import { getConsistentMachineId } from "@/shared/utils/machineId"; @@ -151,6 +153,11 @@ export async function pingModelByKind(model, kind, baseUrl = `http://127.0.0.1:$ let parsed = null; try { parsed = rawText ? JSON.parse(rawText) : null; } catch {} + // Unwrap before the choices checks below. No-op for providers that do not + // opt in via transport.quirks.clineEnvelope. + const providerId = resolveProviderId(String(model).split("/")[0]); + parsed = unwrapClineEnvelope(parsed, providerId); + if (!res.ok) { const detail = parsed?.error?.message || parsed?.msg || parsed?.message || parsed?.error || rawText; return { ok: false, latencyMs, error: `HTTP ${res.status}${detail ? `: ${String(detail).slice(0, 240)}` : ""}`, status: res.status }; diff --git a/src/app/api/providers/[id]/models/route.js b/src/app/api/providers/[id]/models/route.js index d73ec11d..605bc5ca 100644 --- a/src/app/api/providers/[id]/models/route.js +++ b/src/app/api/providers/[id]/models/route.js @@ -11,6 +11,7 @@ import { resolveQoderModels } from "open-sse/services/qoderModels.js"; import { resolveGrokCliModels } from "open-sse/services/grokCliModels.js"; import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy"; import { resolveCursorModels } from "open-sse/services/cursorModels.js"; +import { resolveClineModels, resolveClinepassModels } from "open-sse/services/clinepassModels.js"; const GEMINI_CLI_MODELS_URL = "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels"; @@ -287,6 +288,37 @@ const PROVIDER_MODELS_CONFIG = { }, }, + // Cline/ClinePass share api.cline.bot/api/v1/models. The service layer already + // handles Bearer-vs-`workos:` auth and swallows failures into null, so these follow + // the cursor direct pattern (no refreshFn) and only differ in filtering: + // cline returns the whole catalog verbatim, clinepass keeps cline-pass/* only. + cline: { + customResolver: async (connection) => { + const result = await resolveClineModels({ + accessToken: connection.accessToken, + apiKey: connection.apiKey, + }); + if (result?.models?.length) return { models: result.models }; + return { + models: getStaticProviderModels("cline"), + warning: "Cline returned no live models; falling back to static catalog.", + }; + }, + }, + clinepass: { + customResolver: async (connection) => { + const result = await resolveClinepassModels({ + accessToken: connection.accessToken, + apiKey: connection.apiKey, + }); + if (result?.models?.length) return { models: result.models }; + return { + models: getStaticProviderModels("clinepass"), + warning: "ClinePass returned no live models; falling back to static catalog.", + }; + }, + }, + // Custom resolvers (non-OpenAI-shaped APIs / token-refresh flows) kiro: { customResolver: async (connection) => { diff --git a/src/app/api/providers/suggested-models/filters.js b/src/app/api/providers/suggested-models/filters.js index 13454757..8babbb9b 100644 --- a/src/app/api/providers/suggested-models/filters.js +++ b/src/app/api/providers/suggested-models/filters.js @@ -26,4 +26,10 @@ export const FILTERS = { (Array.isArray(models) ? models : []) .filter((m) => m.id?.startsWith("mimo") || m.name?.toLowerCase().includes("mimo")) .map((m) => ({ id: m.id, name: m.name || m.id })), + + "airforce-free": (models) => + (Array.isArray(models) ? models : []) + .filter((m) => (m.tier === "free" || m.id?.endsWith(":free")) && m.supports_chat === true && (!m.media_type || m.media_type === "chat" || m.media_type === "text")) + .map((m) => ({ id: m.id, name: m.name || m.id, contextLength: m.context_length })) + .sort((a, b) => String(a.id).localeCompare(String(b.id))), }; diff --git a/src/app/api/v1/models/route.js b/src/app/api/v1/models/route.js index 173d91c2..9ce571b4 100644 --- a/src/app/api/v1/models/route.js +++ b/src/app/api/v1/models/route.js @@ -11,7 +11,7 @@ import { resolveKiroModels } from "open-sse/services/kiroModels.js"; import { resolveKimchiModels } from "open-sse/services/kimchiModels.js"; import { resolveQoderModels, routableQoderModels } from "open-sse/services/qoderModels.js"; import { resolveCopilotModels } from "open-sse/services/copilotModels.js"; -import { resolveClinepassModels } from "open-sse/services/clinepassModels.js"; +import { resolveClinepassModels, resolveClineModels } from "open-sse/services/clinepassModels.js"; import { resolveGrokCliModels } from "open-sse/services/grokCliModels.js"; import { resolveCursorModels } from "open-sse/services/cursorModels.js"; import { resolveZedModels } from "open-sse/shared/zedAuth.js"; @@ -79,6 +79,13 @@ const LIVE_MODEL_RESOLVERS = { }); return result?.models?.length ? { models: result.models } : null; }, + cline: async (conn) => { + const result = await resolveClineModels({ + accessToken: conn.accessToken, + apiKey: conn.apiKey, + }); + return result?.models?.length ? { models: result.models } : null; + }, "grok-cli": async (conn) => { const proxy = await resolveConnectionProxyConfig(conn.providerSpecificData || {}); const result = await resolveGrokCliModels({ diff --git a/src/shared/components/ModelSelectModal.js b/src/shared/components/ModelSelectModal.js index 2cbc11a2..20e17d95 100644 --- a/src/shared/components/ModelSelectModal.js +++ b/src/shared/components/ModelSelectModal.js @@ -20,6 +20,54 @@ const PROVIDER_ORDER = [ // Providers that need no auth — always show in model selector const NO_AUTH_PROVIDER_IDS = Object.keys(FREE_PROVIDERS).filter(id => FREE_PROVIDERS[id].noAuth); +// Providers with per-account live catalogs via /api/providers/[id]/models. +// Static registry stays as fallback when live fetch fails or is empty. +const LIVE_CATALOG_PROVIDERS = ["cursor", "cline", "clinepass"]; + +// Fetch a provider's account-scoped catalog for every active connection and merge +// the results. Entries collapse by model id on purpose: two connections of the +// same provider produce the same picker value (`alias/id`), so keeping the first +// avoids duplicate rows. There is no per-connection metadata to preserve beyond +// {id,name}. Empty array means "nothing live" so callers keep the static fallback. +function useLiveProviderModels(isOpen, connectionIds, label) { + const [models, setModels] = useState([]); + const idsKey = (connectionIds ?? []).join("|"); + + useEffect(() => { + const ids = idsKey ? idsKey.split("|") : []; + if (!isOpen || ids.length === 0) { + setModels([]); + return undefined; + } + + let cancelled = false; + Promise.all(ids.map(async (connectionId) => { + const response = await fetch(`/api/providers/${connectionId}/models`, { cache: "no-store" }); + if (!response.ok) return []; + const data = await response.json(); + return Array.isArray(data.models) ? data.models : []; + })) + .then((modelLists) => { + if (cancelled) return; + const seen = new Set(); + setModels(modelLists.flat().filter((model) => { + if (!model?.id || seen.has(model.id)) return false; + seen.add(model.id); + return true; + })); + }) + .catch((error) => { + // Do not hide the static fallback when the account catalog is unavailable. + console.warn(`Unable to load ${label} models for selector:`, error); + if (!cancelled) setModels([]); + }); + + return () => { cancelled = true; }; + }, [isOpen, idsKey, label]); + + return models; +} + export default function ModelSelectModal({ isOpen, onClose, @@ -49,48 +97,25 @@ export default function ModelSelectModal({ const [providerNodes, setProviderNodes] = useState([]); const [customModels, setCustomModels] = useState([]); const [disabledModels, setDisabledModels] = useState({}); - const [cursorModels, setCursorModels] = useState([]); - - // Cursor exposes the usable catalog per account. Keep the static catalog only - // as a fallback, since it quickly becomes stale and different accounts can - // have different model entitlements. - const cursorConnectionIds = useMemo( - () => activeProviders - .filter((provider) => provider.provider === "cursor" && provider.id) - .map((provider) => provider.id), - [activeProviders], - ); - - useEffect(() => { - if (!isOpen || cursorConnectionIds.length === 0) { - setCursorModels([]); - return undefined; + // Cursor and Cline expose the usable catalog per account, so the static catalog is + // kept only as a fallback: it goes stale quickly and entitlements differ per account. + // Single map driven by LIVE_CATALOG_PROVIDERS so the constant cannot drift + // from the memos below; per-provider arrays stay referentially stable unless + // activeProviders itself changes. + const liveConnectionIdsByProvider = useMemo(() => { + const map = Object.fromEntries(LIVE_CATALOG_PROVIDERS.map((id) => [id, []])); + for (const p of activeProviders) { + if (p?.id && Object.prototype.hasOwnProperty.call(map, p.provider)) map[p.provider].push(p.id); } + return map; + }, [activeProviders]); + const cursorConnectionIds = liveConnectionIdsByProvider.cursor; + const clineConnectionIds = liveConnectionIdsByProvider.cline; + const clinepassConnectionIds = liveConnectionIdsByProvider.clinepass; - let cancelled = false; - Promise.all(cursorConnectionIds.map(async (connectionId) => { - const response = await fetch(`/api/providers/${connectionId}/models`, { cache: "no-store" }); - if (!response.ok) return []; - const data = await response.json(); - return Array.isArray(data.models) ? data.models : []; - })) - .then((modelLists) => { - if (cancelled) return; - const seen = new Set(); - setCursorModels(modelLists.flat().filter((model) => { - if (!model?.id || seen.has(model.id)) return false; - seen.add(model.id); - return true; - })); - }) - .catch((error) => { - // Do not hide the static fallback when the account catalog is unavailable. - console.warn("Unable to load Cursor models for selector:", error); - if (!cancelled) setCursorModels([]); - }); - - return () => { cancelled = true; }; - }, [isOpen, cursorConnectionIds]); + const cursorModels = useLiveProviderModels(isOpen, cursorConnectionIds, "Cursor"); + const clineModels = useLiveProviderModels(isOpen, clineConnectionIds, "Cline"); + const clinepassModels = useLiveProviderModels(isOpen, clinepassConnectionIds, "ClinePass"); const fetchCombos = async () => { try { @@ -323,8 +348,9 @@ export default function ModelSelectModal({ hasModels: mergedModels.length > 0, }; } else { - const hardcodedModels = providerId === "cursor" && cursorModels.length > 0 - ? cursorModels + const liveModels = providerId === "cursor" ? cursorModels : providerId === "cline" ? clineModels : providerId === "clinepass" ? clinepassModels : []; + const hardcodedModels = liveModels.length > 0 + ? liveModels : getModelsByProviderId(providerId); const hardcodedIds = new Set(hardcodedModels.map((m) => m.id)); @@ -394,7 +420,7 @@ export default function ModelSelectModal({ }); return groups; - }, [filteredActiveProviders, modelAliases, allProviders, providerNodes, customModels, disabledModels, kindFilter, activeProviders, cursorModels]); + }, [filteredActiveProviders, modelAliases, allProviders, providerNodes, customModels, disabledModels, kindFilter, activeProviders, cursorModels, clineModels, clinepassModels]); // Filter combos by search query (and hide combos when kindFilter is set — combos are LLM-only by design) const filteredCombos = useMemo(() => { diff --git a/src/shared/constants/cliTools.js b/src/shared/constants/cliTools.js index 0f7c0236..a6bb4685 100644 --- a/src/shared/constants/cliTools.js +++ b/src/shared/constants/cliTools.js @@ -199,7 +199,7 @@ export const CLI_TOOLS = { id: "cline", name: "Cline", image: "/providers/cline.png", - color: "#00D1B2", + color: "#5B9BD5", description: "Cline AI Coding Assistant", configType: "custom", }, diff --git a/tests/unit/api-airforce-free-models.test.js b/tests/unit/api-airforce-free-models.test.js new file mode 100644 index 00000000..0b435981 --- /dev/null +++ b/tests/unit/api-airforce-free-models.test.js @@ -0,0 +1,36 @@ +import { describe, it, expect } from "vitest"; +import airforce from "../../open-sse/providers/registry/api-airforce.js"; +import { PROVIDERS, PROVIDER_MODELS } from "../../open-sse/providers/index.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; + +describe("api-airforce free models", () => { + const ids = airforce.models.map((m) => m.id); + + it("registers the three live free models", () => { + expect(ids).toContain("gpt-oss-120b"); + expect(ids).toContain("gpt-oss-20b"); + expect(ids).toContain("kimi-k2.7-code"); + }); + + it("drops the dead catalog ids", () => { + expect(ids).not.toContain("anthropic/claude-3.7-sonnet"); + expect(ids).not.toContain("moonshot/kimi-k2.6"); + expect(ids).not.toContain("google/gemini-2.5-flash"); + }); + + it("is passthrough so any live id resolves", () => { + expect(airforce.passthroughModels).toBe(true); + expect(PROVIDERS["api-airforce"].forceStream).toBe(true); + }); + + it("PROVIDER_MODELS['af'] exposes the new ids", () => { + expect(PROVIDER_MODELS.af.map((m) => m.id)).toEqual(expect.arrayContaining([ + "gpt-oss-120b", "gpt-oss-20b", "kimi-k2.7-code", + ])); + }); + + it("caps resolve for the free ids", () => { + expect(getCapabilitiesForModel("api-airforce", "kimi-k2.7-code").reasoning).toBe(true); + expect(getCapabilitiesForModel("api-airforce", "gpt-oss-120b").reasoning).toBe(true); + }); +}); diff --git a/tests/unit/cline-free-models-envelope.test.js b/tests/unit/cline-free-models-envelope.test.js new file mode 100644 index 00000000..8ae0fa89 --- /dev/null +++ b/tests/unit/cline-free-models-envelope.test.js @@ -0,0 +1,297 @@ +// Cline free models (z-ai/glm-5.3-flash, deepseek-v4-flash) wrap non-stream +// chat completions in {"success":true,"data":{...choices...}} on +// https://api.cline.bot/api/v1/chat/completions. Both the UI model-test ping +// (src/app/api/models/test/ping.js) and the proxy non-stream path +// (open-sse/handlers/chatCore/nonStreamingHandler.js) read `choices` at the +// top level, so enveloped choices are invisible ("Provider returned no +// completion choices for this model"). These tests pin the unwrap behavior. + +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +// Mock the heavy Next.js-dependent imports BEFORE importing ping.js +// (same pattern as tests/unit/ping-reasoning-models-3010.test.js). +vi.mock("@/lib/localDb", () => ({ getApiKeys: vi.fn(async () => [{ key: "test-key", isActive: true }]) })); +vi.mock("@/shared/constants/config", () => ({ UPDATER_CONFIG: { appPort: 20127 } })); +vi.mock("@/shared/utils/machineId", () => ({ getConsistentMachineId: vi.fn(async () => "cli-token") })); +// requestDetail.js imports from @/lib/usageDb.js too, so one mock covers both +// the handler and its usage/detail helpers. +vi.mock("@/lib/usageDb.js", () => ({ + appendRequestLog: vi.fn(async () => {}), + saveRequestDetail: vi.fn(async () => {}), + saveRequestUsage: vi.fn(async () => {}), +})); + +const { pingModelByKind } = await import("../../src/app/api/models/test/ping.js"); +const { handleNonStreamingResponse } = await import("../../open-sse/handlers/chatCore/nonStreamingHandler.js"); + +// The proxy adds a 2000-token headroom buffer to usage before returning it +// to the client (addBufferToUsage), so response-body usage is input + 2000. +// The usage recorded via saveRequestUsage is the unbuffered extraction — +// asserting on it proves the unwrap ran before usage extraction. +const { saveRequestUsage } = await import("@/lib/usageDb.js"); + +describe("cline free-models {success,data} envelope", () => { + let fetchMock; + + beforeEach(() => { + fetchMock = vi.fn(); + vi.stubGlobal("fetch", fetchMock); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + }); + + function jsonResponse(obj) { + return { + ok: true, + status: 200, + text: async () => JSON.stringify(obj), + json: async () => obj, + }; + } + + it("ping: enveloped success unwraps to ok:true (regression for reported error)", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ success: true, data: { choices: [{ message: { content: "OK" } }] } }) + ); + const result = await pingModelByKind("cl/z-ai/glm-5.3-flash", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(true); + }); + + it("ping: enveloped reasoning-only response still soft-passes with note", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ + success: true, + data: { + choices: [ + { + finish_reason: "length", + message: { content: "", reasoning: "The user said hi" }, + }, + ], + }, + }) + ); + const result = await pingModelByKind("cl/z-ai/glm-5.3-flash", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(true); + expect(result.note).toMatch(/reasoning-only/); + }); + + it("ping: error envelope passes through without unwrap", async () => { + const body = { error: "empty response content", success: false }; + fetchMock.mockResolvedValue({ + ok: false, + status: 500, + text: async () => JSON.stringify(body), + json: async () => body, + }); + const result = await pingModelByKind("cl/z-ai/glm-5.3-flash", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(false); + expect(result.error).toMatch(/empty response content/); + }); + + it("ping: bare (un-enveloped) body still passes", async () => { + fetchMock.mockResolvedValue(jsonResponse({ choices: [{ message: { content: "Hello!" } }] })); + const result = await pingModelByKind("openai/gpt-4o", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(true); + }); + + it("ping: does not unwrap for a provider that did not opt in", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ success: true, data: { choices: [{ message: { content: "OK" } }] } }) + ); + const result = await pingModelByKind("openai/gpt-4o", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(false); + expect(result.error).toMatch(/no completion choices/i); + }); +}); + +describe("cline free-models envelope in nonStreamingHandler", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + function stubLogger() { + return { logProviderResponse() {}, logConvertedResponse() {} }; + } + + function callHandler(providerResponse, provider = "cline") { + return handleNonStreamingResponse({ + providerResponse, + provider, + model: "z-ai/glm-5.3-flash", + sourceFormat: "openai", + targetFormat: "openai", + body: { stream: false }, + stream: false, + translatedBody: null, + finalBody: null, + requestStartTime: Date.now(), + connectionId: "c1", + apiKey: "k", + clientRawRequest: null, + onRequestSuccess: () => {}, + reqLogger: stubLogger(), + toolNameMap: null, + customToolNames: null, + trackDone: () => {}, + appendLog: () => {}, + pxpipe: null, + reqTag: "t", + log: null, + }); + } + + it("unwraps the {success,data} envelope before usage extraction and translation", async () => { + const providerResponse = new Response( + JSON.stringify({ + success: true, + data: { + choices: [{ message: { content: "Hi" } }], + usage: { prompt_tokens: 5, completion_tokens: 2 }, + }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + const result = await callHandler(providerResponse); + expect(result.success).toBe(true); + const body = await result.response.json(); + expect(body.choices).toBeDefined(); + expect(body.choices[0].message.content).toBe("Hi"); + expect(body.success).toBeUndefined(); + expect(saveRequestUsage).toHaveBeenCalledTimes(1); + expect(saveRequestUsage.mock.calls[0][0].tokens).toMatchObject({ + prompt_tokens: 5, + completion_tokens: 2, + }); + expect(body.usage.prompt_tokens).toBe(2005); + }); + + it("passes a bare (non-enveloped) body through unchanged", async () => { + const providerResponse = new Response( + JSON.stringify({ + choices: [{ message: { content: "Hi" } }], + usage: { prompt_tokens: 3, completion_tokens: 1 }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + const result = await callHandler(providerResponse); + expect(result.success).toBe(true); + const body = await result.response.json(); + expect(body.choices[0].message.content).toBe("Hi"); + expect(saveRequestUsage).toHaveBeenCalledTimes(1); + expect(saveRequestUsage.mock.calls[0][0].tokens).toMatchObject({ + prompt_tokens: 3, + completion_tokens: 1, + }); + expect(body.usage.prompt_tokens).toBe(2003); + }); + + // The unwrap is opt-in via transport.quirks.clineEnvelope so it can never + // rewrite another provider's body — including one that happens to return + // {"success":true,"data":...} for its own reasons. + it("leaves an enveloped body untouched for a provider that did not opt in", async () => { + const providerResponse = new Response( + JSON.stringify({ + success: true, + data: { choices: [{ message: { content: "Hi" } }] }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + const result = await callHandler(providerResponse, "openai"); + const body = await result.response.json(); + expect(body.success).toBe(true); + expect(body.data.choices[0].message.content).toBe("Hi"); + expect(body.choices).toBeUndefined(); + }); +}); + +describe("cline /api/v1/models aggregation (resolveClineModels vs resolveClinepassModels)", () => { + const API_MODELS_URL = "https://api.cline.bot/api/v1/models"; + + const API_RESPONSE = [ + { id: "cline-pass/deepseek-v4-flash", name: "DeepSeek V4 Flash" }, + { id: "cline-pass/glm-5.2", name: "GLM-5.2" }, + { id: "z-ai/glm-5.3-flash", name: "GLM-5.3 Flash" }, + { id: "z-ai/deepseek-v4-flash", name: "DeepSeek V4 Flash (Free)" }, + ]; + + let fetchMock; + + beforeEach(() => { + fetchMock = vi.fn(); + vi.stubGlobal("fetch", fetchMock); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it("resolveClineModels returns all models (including free-tier)", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => API_RESPONSE, + }); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result).not.toBeNull(); + expect(result.models).toHaveLength(4); + const ids = result.models.map((m) => m.id); + expect(ids).toContain("cline-pass/deepseek-v4-flash"); + expect(ids).toContain("z-ai/glm-5.3-flash"); + expect(ids).toContain("z-ai/deepseek-v4-flash"); + }); + + it("resolveClinepassModels returns only cline-pass/ models", async () => { + const { resolveClinepassModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => API_RESPONSE, + }); + const result = await resolveClinepassModels({ accessToken: "test-token" }); + expect(result).not.toBeNull(); + expect(result.models).toHaveLength(2); + const ids = result.models.map((m) => m.id); + expect(ids).toContain("cline-pass/deepseek-v4-flash"); + expect(ids).toContain("cline-pass/glm-5.2"); + expect(ids).not.toContain("z-ai/glm-5.3-flash"); + }); + + it("resolveClineModels unwraps {success,data} envelope", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => ({ success: true, data: API_RESPONSE }), + }); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result).not.toBeNull(); + expect(result.models).toHaveLength(4); + }); + + it("resolveClineModels returns null when no token", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({}); + expect(result).toBeNull(); + }); + + it("resolveClineModels returns null on fetch error", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockRejectedValue(new Error("network error")); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result).toBeNull(); + }); + + it("resolveClineModels returns {id,name} shape", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => API_RESPONSE, + }); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result.models[0]).toHaveProperty("id"); + expect(result.models[0]).toHaveProperty("name"); + expect(typeof result.models[0].id).toBe("string"); + expect(typeof result.models[0].name).toBe("string"); + }); +}); From 8a81085a72dd4624cc958de165da3e8bb2ba358e Mon Sep 17 00:00:00 2001 From: Nick Nyanjui Date: Thu, 10 Sep 2026 22:53:57 +0700 Subject: [PATCH 39/78] fix(claude): cap re-anchored cache_control at the 4-marker budget and keep single-object content turns Anthropic accepts at most 4 blocks carrying cache_control per request. When the client had already spent that budget, the re-anchor added a 5th marker and the request was rejected with a non-retryable 400 that the failure path treated as an account problem, retrying the same malformed body across the whole pool until every account locked. anchorClaudeCache now normalizes bare-object content, strips the invalid cache_control carried by defer_loading tools, pins the 1h head anchors on the last system block and last cacheable tool, then trims an over-budget body to 4 markers. The trim holds those head anchors and fills the remaining slots with the tail-most message markers: a plain "keep the last four in document order" rule drops the anchors first even though they lead document order, and skipping the re-anchor at a spent budget left system/tools on the 5m default instead of 1h. Some clients send content as a single block object rather than a one-element array. Such a turn was dropped or zeroed on every leg that reads messages, silently losing conversation history. normalizeMessageContent wraps it as a one-block array on all four paths, and hasValidContent keeps it. --- open-sse/translator/formats/claude.js | 106 ++++++++- .../translator/request/claude-to-openai.js | 7 + .../claude-cache-budget-single-object.test.js | 222 ++++++++++++++++++ 3 files changed, 330 insertions(+), 5 deletions(-) create mode 100644 tests/unit/claude-cache-budget-single-object.test.js diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index 930c761a..f57972da 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -27,6 +27,14 @@ export function lastCacheableToolIndex(tools) { // Check if message has valid non-empty content export function hasValidContent(msg) { if (typeof msg.content === "string" && msg.content.trim()) return true; + if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) { + const block = msg.content; + return !!((block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) || + block.type === CLAUDE_BLOCK.TOOL_USE || + block.type === CLAUDE_BLOCK.TOOL_RESULT || + block.type === CLAUDE_BLOCK.IMAGE || + block.type === CLAUDE_BLOCK.DOCUMENT); + } if (Array.isArray(msg.content)) { return msg.content.some(block => (block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) || @@ -38,6 +46,60 @@ export function hasValidContent(msg) { } return false; } +// Content may arrive as a single content block object (spec allows string | array; +// some clients send the bare object). Wrap it as a one-block array and strip any +// client-placed cache_control: a bare-object marker must never survive +// normalization, on any path, guard or no guard. +function normalizeMessageContent(msg) { + const c = msg?.content; + if (c && typeof c === "object" && !Array.isArray(c)) { + delete c.cache_control; + msg.content = [c]; + } + return msg; +} + +// Total blocks carrying cache_control across system, tools, and messages — the +// upstream Messages API allows at most 4 markers per request. +function countCacheControlBlocks(body) { + let n = 0; + if (Array.isArray(body?.system)) for (const b of body.system) if (b?.cache_control) n++; + if (Array.isArray(body?.tools)) for (const t of body.tools) if (t?.cache_control) n++; + if (Array.isArray(body?.messages)) { + for (const m of body.messages) { + if (Array.isArray(m?.content)) { + for (const b of m.content) if (b?.cache_control) n++; + } else if (m?.content && typeof m.content === "object" && m.content.cache_control) n++; + } + } + return n; +} +// Trim every marker past the 4-marker budget. The head anchors (last system +// block, last cacheable tool) are held; the remaining slots go to the tail-most +// of the other markers in document order. A plain "keep the last 4 in document +// order" rule would drop the head anchors first — they lead document order, yet +// they are exactly what re-anchoring exists to pin. +function capCacheControlBlocks(body) { + const isHead = (b) => { + const sys = Array.isArray(body?.system) ? body.system : []; + if (sys.length && sys[sys.length - 1] === b) return true; + const tools = Array.isArray(body?.tools) ? body.tools : []; + const lastTool = lastCacheableToolIndex(tools); + return lastTool >= 0 && tools[lastTool] === b; + }; + const marked = []; + if (Array.isArray(body?.system)) for (const b of body.system) if (b?.cache_control) marked.push(b); + if (Array.isArray(body?.tools)) for (const t of body.tools) if (t?.cache_control) marked.push(t); + if (Array.isArray(body?.messages)) { + for (const m of body.messages) { + if (Array.isArray(m?.content)) for (const b of m.content) if (b?.cache_control) marked.push(b); + } + } + const head = marked.filter(isHead); + const rest = marked.filter(b => !isHead(b)); + const keep = Math.max(0, 4 - head.length); + for (const b of rest.slice(0, Math.max(0, rest.length - keep))) delete b.cache_control; +} // Fix tool_use/tool_result ordering for Claude API // 1. Assistant message with tool_use: remove text AFTER tool_use (Claude doesn't allow) @@ -136,8 +198,9 @@ function hasForeignServerToolUseId(block) { // Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject: // 1. thinking.type "adaptive" → unsupported on Haiku // 2. output_config.effort → unsupported on Haiku -// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed -// 4. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright +// 3. bare content-block objects (content: {block} instead of [{block}]) → wrapped first +// 4. role "system" messages (mid-conversation-system beta) → only top-level system is allowed +// 5. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright export function normalizeClaudePassthrough(body, model = "") { if (!body || typeof body !== "object") return body; @@ -152,7 +215,15 @@ export function normalizeClaudePassthrough(body, model = "") { if (Object.keys(body.output_config).length === 0) delete body.output_config; } - // 2. Fold mid-conversation system messages into the neighbouring turn. + // 3. Wrap bare content-block objects as one-element arrays before folding. + // Some clients send content: {block} instead of content: [{block}]; the + // mid-conversation-system fold below assumes the array shape, so it must + // run first — a bare-object neighbor would otherwise be zeroed to []. + if (Array.isArray(body.messages)) { + for (const msg of body.messages) normalizeMessageContent(msg); + } + + // 4. Fold mid-conversation system messages into the neighbouring turn. // Hoisting them into body.system would insert volatile content (token counters, // reminders) ahead of the whole conversation and invalidate the prefix cache on // every request. Folding in place keeps the cached prefix stable. @@ -186,7 +257,7 @@ export function normalizeClaudePassthrough(body, model = "") { body.messages = messages; } - // 3. Drop thinking blocks whose signature is not Claude's (combo mixes models, + // 5. Drop thinking blocks whose signature is not Claude's (combo mixes models, // so foreign signatures leak into history and Anthropic rejects them). const thinkingEnabled = body.thinking?.type === "enabled"; const droppedServerToolUseIds = new Set(); @@ -233,7 +304,7 @@ export function normalizeClaudePassthrough(body, model = "") { } } - // 5. Drop empty text blocks and any message left with no content at all. + // 6. Drop empty text blocks and any message left with no content at all. // Anthropic rejects `messages.N.content` blocks with empty text (400 // "text content blocks must be non-empty"); a message whose blocks were all // stripped above must be dropped, not padded with an empty placeholder. @@ -271,7 +342,22 @@ function markLastCacheableBlock(msg) { // (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail. export function anchorClaudeCache(body) { if (!body || typeof body !== "object") return body; + if (Array.isArray(body.messages)) { + for (const msg of body.messages) normalizeMessageContent(msg); + } + // Invalid markers first, whatever the budget: Anthropic rejects a tool that + // carries BOTH defer_loading and cache_control (#3567). The re-anchor path + // below strips them anyway; the over-budget early return used to forward + // them untouched. + if (Array.isArray(body.tools)) { + for (const t of body.tools) { + if (t?.defer_loading === true) delete t.cache_control; + } + } + // Head anchors first, before any budget guard: the 1h TTL on system/tools is + // the point of re-anchoring, and skipping it because the client spent its + // budget would silently downgrade a cache hit to the 5m default. if (Array.isArray(body.system)) { const last = body.system.length - 1; body.system.forEach((block, i) => { @@ -289,6 +375,15 @@ export function anchorClaudeCache(body) { }); } + // Budget guard AFTER the head anchors: with the last system block and last + // tool pinned, at most 2 slots remain. At >= 4 markers the client has spent + // the rest of the budget and every remaining marker is itself a valid + // breakpoint — re-anchoring the tail could only exceed 4, so trim instead. + if (countCacheControlBlocks(body) >= 4) { + capCacheControlBlocks(body); + return body; + } + if (Array.isArray(body.messages)) { let anchored = null; for (let i = body.messages.length - 1; i >= 0; i--) { @@ -368,6 +463,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // Pass 1: remove cache_control + filter empty messages for (let i = 0; i < len; i++) { const msg = body.messages[i]; + normalizeMessageContent(msg); // Remove cache_control from content blocks if (Array.isArray(msg.content)) { diff --git a/open-sse/translator/request/claude-to-openai.js b/open-sse/translator/request/claude-to-openai.js index 3956f828..38976226 100644 --- a/open-sse/translator/request/claude-to-openai.js +++ b/open-sse/translator/request/claude-to-openai.js @@ -142,6 +142,13 @@ function systemReminderText(content) { // Convert single Claude message - returns single message or array of messages function convertClaudeMessage(msg) { + // Some clients send content as a single block object; normalize to the + // one-element array every branch below (the system-reminder fold included) + // expects. Must run BEFORE the role branch: systemReminderText only reads + // arrays and strings, so a bare-object system turn was dropped outright. + if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) { + msg.content = [msg.content]; + } // Mid-conversation system message -> user (per Anthropic placement rules) if (msg.role === ROLE.SYSTEM) { const text = systemReminderText(msg.content); diff --git a/tests/unit/claude-cache-budget-single-object.test.js b/tests/unit/claude-cache-budget-single-object.test.js new file mode 100644 index 00000000..e2680723 --- /dev/null +++ b/tests/unit/claude-cache-budget-single-object.test.js @@ -0,0 +1,222 @@ +// #3795 — a proxy must not dispatch more cache_control blocks than Anthropic +// accepts, and must not lose turns that use single-object content (#3567 interplay). +import { describe, it, expect } from "vitest"; +import { + anchorClaudeCache, + normalizeClaudePassthrough, + prepareClaudeRequest, +} from "../../open-sse/translator/formats/claude.js"; +import { claudeToOpenAIRequest } from "../../open-sse/translator/request/claude-to-openai.js"; + +const CC = { type: "ephemeral" }; +const text = (t, extra = {}) => ({ type: "text", text: t, ...extra }); +const tool = (name, extra = {}) => ({ name, description: "d", input_schema: {}, ...extra }); + +// counts markers incl. single-object content — mirrors the upstream contract +function countMarkers(body) { + let n = 0; + if (Array.isArray(body.system)) for (const b of body.system) if (b?.cache_control) n++; + if (Array.isArray(body.tools)) for (const t of body.tools) if (t?.cache_control) n++; + if (Array.isArray(body.messages)) for (const m of body.messages) { + if (Array.isArray(m?.content)) { + for (const b of m.content) if (b?.cache_control) n++; + } else if (m?.content && typeof m.content === "object" && m.content.cache_control) n++; + } + return n; +} + +describe("cache marker budget and single-block content", () => { + it("never emits more than four markers when the client already spent its budget", () => { + const out = anchorClaudeCache({ + system: [text("s1"), text("s2", { cache_control: CC })], + tools: [tool("t1"), tool("t2", { cache_control: CC })], + messages: [ + { role: "user", content: text("u1", { cache_control: CC }) }, + { role: "assistant", content: text("a1", { cache_control: CC }) }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBeLessThanOrEqual(4); // base: 5 + }); + + it("normalizes a single-object turn in passthrough and anchors it", () => { + const body = { + messages: [ + { role: "user", content: [text("u1")] }, + { role: "assistant", content: text("a1") }, // single object, no marker + { role: "user", content: [text("q")] }, + ], + }; + normalizeClaudePassthrough(body); + const assistant = body.messages.find(m => m.role === "assistant"); + expect(assistant).toBeDefined(); + expect(Array.isArray(assistant.content)).toBe(true); // base: still bare object + expect(assistant.content).toHaveLength(1); + const out = anchorClaudeCache(body); + expect(countMarkers(out)).toBe(1); + }); + + it("keeps a turn whose content is a single object and strips its marker", () => { + const out = prepareClaudeRequest({ + model: "claude-sonnet-5", max_tokens: 100, + system: [text("s1")], + messages: [ + { role: "user", content: text("u1", { cache_control: CC }) }, + { role: "assistant", content: [text("a1")] }, + { role: "user", content: [text("q")] }, + ], + }, "claude"); + const kept = out.messages.filter(m => JSON.stringify(m.content).includes("u1")); + expect(kept.length).toBe(1); // base: 0 (dropped) + expect(kept[0].content).toHaveLength(1); // normalized to array + expect(kept[0].content[0].cache_control).toBeUndefined(); + }); + + it("drops no conversation turn when content is a single text object", () => { + const out = prepareClaudeRequest({ + model: "claude-sonnet-5", max_tokens: 100, + messages: [ + { role: "user", content: text("u1") }, + { role: "assistant", content: text("a1") }, + { role: "user", content: [text("q")] }, + ], + }, "claude"); + expect(out.messages.length).toBe(3); // base: 1 + expect(Array.isArray(out.messages[0].content)).toBe(true); + }); + + it("re-anchors the last assistant turn even when it uses single-object content", () => { + const out = prepareClaudeRequest({ + model: "claude-sonnet-5", max_tokens: 100, + messages: [ + { role: "user", content: [text("u1")] }, + { role: "assistant", content: text("a1") }, + { role: "user", content: [text("q")] }, + ], + }, "claude"); + expect(countMarkers(out)).toBe(1); // base: 0 + }); + + it("keeps a marked single-object turn when the marker budget is spent", () => { + const body = { + system: [text("s1", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC })], + messages: [ + { role: "user", content: [text("c1"), text("c2")] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: text("u1", { cache_control: CC }) }, + { role: "user", content: [text("q")] }, + ], + }; + normalizeClaudePassthrough(body); + const out = anchorClaudeCache(body); + const kept = out.messages.filter(m => JSON.stringify(m.content).includes("u1")); + expect(kept.length).toBe(1); + expect(Array.isArray(kept[0].content)).toBe(true); // base: bare object survives + expect(kept[0].content).toHaveLength(1); + expect(kept[0].content[0].cache_control).toBeUndefined(); + const ctx = out.messages.find(m => JSON.stringify(m.content).includes("c1")); + expect(ctx.content).toEqual([text("c1"), text("c2")]); + expect(countMarkers(out)).toBeLessThanOrEqual(4); // fixed: 3 + }); + + it("keeps single-object turns on the claude-to-openai leg", () => { + const out = claudeToOpenAIRequest("m", { + messages: [ + { role: "user", content: text("u1") }, + { role: "assistant", content: { type: "image", source: { type: "base64", media_type: "image/png", data: "iVBORw0KGgo=" } } }, + ], + }, false); + expect(out.messages.some(m => JSON.stringify(m.content).includes("u1"))).toBe(true); // base: dropped + const img = out.messages.find(m => m.role === "assistant"); + expect(JSON.stringify(img.content)).toContain("image_url"); // base: dropped + }); + + it("keeps a bare-object user turn folded with a mid-conversation system message", () => { + const body = { + messages: [ + { role: "user", content: text("u1") }, + { role: "system", content: [text("reminder")] }, + { role: "user", content: [text("q")] }, + ], + }; + normalizeClaudePassthrough(body); + const first = body.messages[0]; + expect(Array.isArray(first.content)).toBe(true); + expect(JSON.stringify(first.content).includes("u1")).toBe(true); // pre-hoist: fold zeroes bare-object content + expect(JSON.stringify(first.content).includes("reminder")).toBe(true); + }); + it("prunes a client body that already carries five markers down to four", () => { + const out = anchorClaudeCache({ + system: [text("s1"), text("s2", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC }), tool("t2", { cache_control: CC })], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBe(4); // pre-fix: 5 forwarded unchanged + expect(out.system[0].cache_control).toBeUndefined(); // earliest marker pruned + }); + + // A spent budget must not cost the head anchors their 1h TTL: system/tools are + // the whole point of re-anchoring, and a 5m fallback silently halves the cache + // lifetime on exactly the requests that already cached aggressively. + it("keeps the 1h head anchors when the client spent the whole budget", () => { + const out = anchorClaudeCache({ + system: [text("s1"), text("s2", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC }), tool("t2")], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBeLessThanOrEqual(4); + expect(out.system.at(-1).cache_control?.ttl).toBe("1h"); // pre-fix: fell back to 5m + expect(out.tools.at(-1).cache_control?.ttl).toBe("1h"); // pre-fix: fell back to 5m + }); + + it("keeps the 1h head anchors on an over-budget body", () => { + const out = anchorClaudeCache({ + system: [text("s1", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC }), tool("t2")], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("u2", { cache_control: CC })] }, + { role: "assistant", content: [text("a2", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBe(4); + expect(out.system.at(-1).cache_control?.ttl).toBe("1h"); + expect(out.tools.at(-1).cache_control?.ttl).toBe("1h"); + }); + + it("strips a marker from a deferred tool even when the budget is spent", () => { + const out = anchorClaudeCache({ + system: [text("s1", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC, defer_loading: true })], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBeLessThanOrEqual(4); + const deferred = out.tools.find(t => t.defer_loading); + expect(deferred?.cache_control).toBeUndefined(); // pre-fix: invalid marker forwarded + }); + + it("keeps a bare-object system reminder on the claude-to-openai leg", () => { + const out = claudeToOpenAIRequest("m", { + messages: [ + { role: "user", content: "hi" }, + { role: "system", content: text("be brief") }, + ], + }, false); + expect(JSON.stringify(out.messages)).toContain("be brief"); // pre-fix: turn dropped + }); +}); From 998bb3d97593ac668c74f506e54c48a74a4e9cd1 Mon Sep 17 00:00:00 2001 From: galiehneh Date: Thu, 10 Sep 2026 23:03:42 +0700 Subject: [PATCH 40/78] fix(tools): scope Claude tool type defaulting to gateways that need it (#3905) defaultClaudeToolType() stamped tools[].type = "custom" onto every Claude-format request carrying tools since e08ac6da. That satisfied MiniMax (error 2013) but broke Anthropic-compatible endpoints that only accept the legacy typeless tool shape. DeepSeek's endpoint (api.deepseek.com/anthropic/v1/messages) whitelists its tool `type` enum to the web_search_* variants and answers HTTP 400 "unknown variant `custom`", so every Claude Code request routed to a DeepSeek connection failed and surfaced as a persistent 503. Run the defaulting only when the target provider declares the new requireClaudeToolType quirk (MiniMax, MiniMax-CN). Add shouldDefaultClaudeToolType(provider, finalFormat, tools, PROVIDERS) in translator/concerns/toolCall.js so the gate is unit-testable, and cover MiniMax keeping the explicit type, DeepSeek/Anthropic staying typeless, non-Claude formats and tool-less requests never defaulting. Effectively a no-op for MiniMax and a restore of the pre-e08ac6da behaviour everywhere else. Another strict gateway now only needs the same one-line quirk instead of a global behavioural change. --- open-sse/handlers/chatCore.js | 8 +++-- open-sse/providers/registry/minimax-cn.js | 1 + open-sse/providers/registry/minimax.js | 1 + open-sse/translator/concerns/toolCall.js | 15 +++++++++ .../bugs-3905-deepseek-tool-type.test.js | 33 +++++++++++++++++++ 5 files changed, 56 insertions(+), 2 deletions(-) create mode 100644 tests/translator/bugs-3905-deepseek-tool-type.test.js diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index d5dcf2d8..85c6fc4a 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -28,7 +28,7 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js"; import { getCapabilitiesForModel } from "../providers/capabilities.js"; import { stripUnsupportedModalities } from "../translator/concerns/modality.js"; import { prefetchRemoteImages } from "../translator/concerns/prefetch.js"; -import { defaultClaudeToolType } from "../translator/concerns/toolCall.js"; +import { defaultClaudeToolType, shouldDefaultClaudeToolType } from "../translator/concerns/toolCall.js"; import { resolveSessionId } from "../utils/sessionManager.js"; /** @@ -243,7 +243,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred // Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax) // reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing. - if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) { + // Provider-scoped via quirks (shouldDefaultClaudeToolType): only gateways that declare + // requireClaudeToolType get the explicit type. Applying it unconditionally breaks + // Claude-format endpoints that only accept the legacy typeless tool shape — DeepSeek's + // Anthropic-compatible endpoint 400s with "unknown variant `custom`" (#3905). + if (shouldDefaultClaudeToolType(provider, finalFormat, translatedBody.tools, PROVIDERS)) { translatedBody.tools = defaultClaudeToolType(translatedBody.tools); } diff --git a/open-sse/providers/registry/minimax-cn.js b/open-sse/providers/registry/minimax-cn.js index 19aa3127..268a7344 100644 --- a/open-sse/providers/registry/minimax-cn.js +++ b/open-sse/providers/registry/minimax-cn.js @@ -22,6 +22,7 @@ export default { headers: { ...CLAUDE_API_HEADERS }, quirks: { dropOutputConfig: true, + requireClaudeToolType: true, }, reasoningInject: { scope: "all", diff --git a/open-sse/providers/registry/minimax.js b/open-sse/providers/registry/minimax.js index 47b82c89..66c00958 100644 --- a/open-sse/providers/registry/minimax.js +++ b/open-sse/providers/registry/minimax.js @@ -22,6 +22,7 @@ export default { headers: { ...CLAUDE_API_HEADERS }, quirks: { dropOutputConfig: true, + requireClaudeToolType: true, }, reasoningInject: { scope: "all", diff --git a/open-sse/translator/concerns/toolCall.js b/open-sse/translator/concerns/toolCall.js index 958764dd..251850c1 100644 --- a/open-sse/translator/concerns/toolCall.js +++ b/open-sse/translator/concerns/toolCall.js @@ -1,5 +1,7 @@ // Tool call helper functions for translator +import { FORMATS } from "../formats.js"; + // Anthropic tool_use.id must match: ^[a-zA-Z0-9_-]+$ const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/; @@ -165,3 +167,16 @@ export function defaultClaudeToolType(tools) { return tools.map(tool => tool?.type ? tool : { ...tool, type: "custom" }); } +// Whether Claude-format tools need explicit `type` defaulting before dispatch. +// Only gateways that declare the `requireClaudeToolType` quirk (MiniMax) reject typeless +// tools. Applying the default globally breaks Claude-format endpoints that only accept the +// legacy typeless tool shape — DeepSeek's Anthropic-compatible endpoint answers HTTP 400 +// "unknown variant `custom`" and every Claude Code request routed there fails (#3905). +export function shouldDefaultClaudeToolType(provider, finalFormat, tools, PROVIDERS) { + return ( + finalFormat === FORMATS.CLAUDE + && Array.isArray(tools) + && PROVIDERS?.[provider]?.quirks?.requireClaudeToolType === true + ); +} + diff --git a/tests/translator/bugs-3905-deepseek-tool-type.test.js b/tests/translator/bugs-3905-deepseek-tool-type.test.js new file mode 100644 index 00000000..fd561d0b --- /dev/null +++ b/tests/translator/bugs-3905-deepseek-tool-type.test.js @@ -0,0 +1,33 @@ +// Regression for #3905: defaultClaudeToolType() (type:"custom") must only run for +// gateways that declare the requireClaudeToolType quirk (MiniMax). Claude-format +// endpoints that only accept the legacy typeless tool shape — e.g. DeepSeek's +// Anthropic-compatible endpoint, which answers HTTP 400 "unknown variant `custom`" — +// must never receive tools[].type = "custom". +import { describe, it, expect } from "vitest"; +import { PROVIDERS } from "../../open-sse/providers/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { shouldDefaultClaudeToolType } from "../../open-sse/translator/concerns/toolCall.js"; + +const tools = [{ name: "get_weather", description: "weather", input_schema: { type: "object" } }]; + +describe("Claude tool `type` defaulting is provider-scoped (#3905)", () => { + it("runs only for providers declaring requireClaudeToolType", () => { + expect(shouldDefaultClaudeToolType("minimax", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(true); + expect(shouldDefaultClaudeToolType("minimax-cn", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(true); + // Endpoints accepting only the legacy typeless shape must NOT get type:"custom". + expect(shouldDefaultClaudeToolType("deepseek", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(false); + expect(shouldDefaultClaudeToolType("claude", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(false); + }); + + it("never applies outside Claude-format requests or without tools", () => { + expect(shouldDefaultClaudeToolType("minimax", FORMATS.OPENAI, tools, PROVIDERS)).toBe(false); + expect(shouldDefaultClaudeToolType("minimax", FORMATS.CLAUDE, undefined, PROVIDERS)).toBe(false); + expect(shouldDefaultClaudeToolType("minimax", FORMATS.CLAUDE, null, PROVIDERS)).toBe(false); + }); + + it("declares the quirk only on the MiniMax providers (registry tripwire)", () => { + expect(PROVIDERS.minimax?.quirks?.requireClaudeToolType).toBe(true); + expect(PROVIDERS["minimax-cn"]?.quirks?.requireClaudeToolType).toBe(true); + expect(PROVIDERS.deepseek?.quirks?.requireClaudeToolType).toBeUndefined(); + }); +}); From 248d7da01c44180dcac466631e24a17b585b95b0 Mon Sep 17 00:00:00 2001 From: LLL <2798142644@qq.com> Date: Thu, 10 Sep 2026 22:52:29 +0700 Subject: [PATCH 41/78] revert(qoder): drop the Responses usage plumbing from shared code The merged Qoder work also rewrote shared translator/handler code so that /v1/responses clients got token usage on response.completed. That changed behaviour for every provider, not just Qoder: proxies saw input tokens rise by the 2000-token context buffer, and the plain token mapping was replaced by one that always adds input_tokens_details. A probe confirms the Qoder benefit does not depend on those edits: the executor's coalescer already emits one include_usage-style finish chunk, so a Claude client receives input_tokens and cache_read_input_tokens with every shared file at its original state. Only the Responses path relies on the shared translator, and that path has no Qoder-owned seam to put it in. Reverts the shared files to their pre-PR state and drops the Responses usage test. The Cline envelope unwrap in nonStreamingHandler.js, which landed after the PR in the same file, is kept. --- .../handlers/chatCore/nonStreamingHandler.js | 8 +- .../handlers/chatCore/sseToJsonHandler.js | 8 +- open-sse/transformer/responsesTransformer.js | 27 ++- open-sse/translator/concerns/usage.js | 31 --- .../translator/response/openai-responses.js | 34 ++-- .../translator/response/openai-to-claude.js | 81 ++++---- open-sse/utils/stream.js | 17 +- tests/unit/openai-responses-usage.test.js | 182 ------------------ tests/unit/openai-to-claude.test.js | 27 --- 9 files changed, 74 insertions(+), 341 deletions(-) delete mode 100644 tests/unit/openai-responses-usage.test.js diff --git a/open-sse/handlers/chatCore/nonStreamingHandler.js b/open-sse/handlers/chatCore/nonStreamingHandler.js index fe94b282..d7f4f3fe 100644 --- a/open-sse/handlers/chatCore/nonStreamingHandler.js +++ b/open-sse/handlers/chatCore/nonStreamingHandler.js @@ -11,7 +11,6 @@ import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, sav import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { decloakToolNames } from "../../utils/claudeCloaking.js"; import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js"; -import { toResponsesUsage } from "../../translator/concerns/usage.js"; function parseToolArguments(value) { if (!value) return {}; @@ -132,8 +131,11 @@ function openAICompletionToResponses(responseBody, customToolNames = null) { background: false, error: null, output, - // Keep cached/reasoning details (input_tokens_details) — proxies bill cache hits from them - usage: toResponsesUsage(usage) || { input_tokens: 0, output_tokens: 0, total_tokens: 0 }, + usage: { + input_tokens: usage.prompt_tokens || usage.input_tokens || 0, + output_tokens: usage.completion_tokens || usage.output_tokens || 0, + total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0), + }, }; } diff --git a/open-sse/handlers/chatCore/sseToJsonHandler.js b/open-sse/handlers/chatCore/sseToJsonHandler.js index 0da5c69f..6801be89 100644 --- a/open-sse/handlers/chatCore/sseToJsonHandler.js +++ b/open-sse/handlers/chatCore/sseToJsonHandler.js @@ -5,7 +5,6 @@ import { FORMATS } from "../../translator/formats.js"; import { PROVIDERS } from "../../config/providers.js"; import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js"; -import { toResponsesUsage } from "../../translator/concerns/usage.js"; // Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES; @@ -98,8 +97,11 @@ function chatCompletionToResponses(responseBody, customToolNames = null) { background: false, error: null, output, - // Keep cached/reasoning details (input_tokens_details) — proxies bill cache hits from them - usage: toResponsesUsage(usage) || { input_tokens: 0, output_tokens: 0, total_tokens: 0 }, + usage: { + input_tokens: usage.prompt_tokens || usage.input_tokens || 0, + output_tokens: usage.completion_tokens || usage.output_tokens || 0, + total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0), + }, }; } diff --git a/open-sse/transformer/responsesTransformer.js b/open-sse/transformer/responsesTransformer.js index 6558a3f1..ac84db20 100644 --- a/open-sse/transformer/responsesTransformer.js +++ b/open-sse/transformer/responsesTransformer.js @@ -6,7 +6,6 @@ import fs from "fs"; import path from "path"; -import { toResponsesUsage } from "../translator/concerns/usage.js"; // Create log directory for responses (Node.js only) export function createResponsesLogger(model, logsDir = null) { @@ -74,7 +73,6 @@ export function createResponsesApiTransformStream(logger = null) { funcArgsDone: {}, funcItemDone: {}, buffer: "", - usage: null, completedSent: false }; @@ -227,17 +225,17 @@ export function createResponsesApiTransformStream(logger = null) { const sendCompleted = (controller) => { if (!state.completedSent) { state.completedSent = true; - const response = { - id: state.responseId, - object: "response", - created_at: state.created, - status: "completed", - background: false, - error: null - }; - const usage = toResponsesUsage(state.usage); - if (usage) response.usage = usage; - emit(controller, "response.completed", { type: "response.completed", response }); + emit(controller, "response.completed", { + type: "response.completed", + response: { + id: state.responseId, + object: "response", + created_at: state.created, + status: "completed", + background: false, + error: null + } + }); } }; @@ -266,9 +264,6 @@ export function createResponsesApiTransformStream(logger = null) { continue; } - // Remember usage (finish chunk or trailing include_usage frame) for response.completed - if (parsed.usage && typeof parsed.usage === "object") state.usage = parsed.usage; - if (!parsed.choices?.length) continue; const choice = parsed.choices[0]; diff --git a/open-sse/translator/concerns/usage.js b/open-sse/translator/concerns/usage.js index 1ee28489..3ace3062 100644 --- a/open-sse/translator/concerns/usage.js +++ b/open-sse/translator/concerns/usage.js @@ -67,34 +67,3 @@ export function toOpenAIUsage(raw, kind) { if (!extract || !raw || typeof raw !== "object") return null; return buildUsage(extract(raw)); } - -// Convert an OpenAI-shaped (or already-canonical / Claude-shaped) usage object into the -// Responses API shape emitted by `response.completed`. Details objects are always present -// (like the real API) so proxies that read `input_tokens_details.cached_tokens` never see undefined. -// Returns null when there is nothing countable. -export function toResponsesUsage(usage) { - if (!usage || typeof usage !== "object") return null; - const input = n(usage.prompt_tokens ?? usage.input_tokens); - const output = n(usage.completion_tokens ?? usage.output_tokens); - if (input === 0 && output === 0) return null; - const cached = n( - usage.input_tokens_details?.cached_tokens ?? - usage.prompt_tokens_details?.cached_tokens ?? - usage.cached_tokens ?? - usage.cache_read_input_tokens - ); - const reasoning = n( - usage.output_tokens_details?.reasoning_tokens ?? - usage.completion_tokens_details?.reasoning_tokens ?? - usage.reasoning_tokens - ); - const out = { - input_tokens: input, - output_tokens: output, - total_tokens: typeof usage.total_tokens === "number" ? usage.total_tokens : input + output, - input_tokens_details: { cached_tokens: cached }, - output_tokens_details: { reasoning_tokens: reasoning }, - }; - if (usage.estimated) out.estimated = true; - return out; -} diff --git a/open-sse/translator/response/openai-responses.js b/open-sse/translator/response/openai-responses.js index 1d9a7dec..bd435f9c 100644 --- a/open-sse/translator/response/openai-responses.js +++ b/open-sse/translator/response/openai-responses.js @@ -5,7 +5,7 @@ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; import { buildChunk } from "../concerns/chunk.js"; -import { buildUsage, toResponsesUsage } from "../concerns/usage.js"; +import { buildUsage } from "../concerns/usage.js"; import { fallbackToolCallId } from "../concerns/toolCall.js"; import { reasoningDelta, extractReasoningText } from "../concerns/reasoning.js"; import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } from "../schema/index.js"; @@ -18,13 +18,7 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { if (!chunk) { return flushEvents(state); } - - // Usage riding on the finish chunk (include_usage style, e.g. coalesced Qoder frames): - // remember it so response.completed can report tokens even outside stream.js. - if (chunk.usage && typeof chunk.usage === "object" && !state.usage) { - state.usage = chunk.usage; - } - + if (!chunk.choices?.length) return []; const events = []; @@ -374,19 +368,17 @@ function closeToolCall(state, emit, idx) { function sendCompleted(state, emit) { if (!state.completedSent) { state.completedSent = true; - const response = { - id: state.responseId, - object: "response", - created_at: state.created, - status: "completed", - background: false, - error: null - }; - // Carry provider usage (recorded by stream.js or from the finish chunk itself) in the - // Responses shape; proxies such as sub2api/Codex read tokens only from here. - const usage = toResponsesUsage(state.usage); - if (usage) response.usage = usage; - emit("response.completed", { type: "response.completed", response }); + emit("response.completed", { + type: "response.completed", + response: { + id: state.responseId, + object: "response", + created_at: state.created, + status: "completed", + background: false, + error: null + } + }); } } diff --git a/open-sse/translator/response/openai-to-claude.js b/open-sse/translator/response/openai-to-claude.js index f291e83a..3998cc84 100644 --- a/open-sse/translator/response/openai-to-claude.js +++ b/open-sse/translator/response/openai-to-claude.js @@ -67,50 +67,48 @@ function stopTextBlock(state, results) { state.textBlockStarted = false; } -function recordOpenAIUsage(chunk, state) { - if (!chunk?.usage || typeof chunk.usage !== "object") return; - - const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0; - const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0; - - // Extract cache tokens from prompt_tokens_details - const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens; - const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens; - const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0; - const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0; - - // input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens - // Because OpenAI's prompt_tokens includes all prompt-side tokens - const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens; - - state.usage = { - input_tokens: inputTokens, - output_tokens: outputTokens - }; - - if (cacheReadTokens > 0) { - state.usage.cache_read_input_tokens = cacheReadTokens; - } - if (cacheCreateTokens > 0) { - state.usage.cache_creation_input_tokens = cacheCreateTokens; - } -} - // Convert OpenAI stream chunk to Claude format export function openaiToClaudeResponse(chunk, state) { - if (!chunk) return null; - - // Track usage from OpenAI chunk if available - if (chunk.usage && typeof chunk.usage === "object") { - recordOpenAIUsage(chunk, state); - } - - if (!chunk.choices?.[0]) return null; + if (!chunk || !chunk.choices?.[0]) return null; const results = []; const choice = chunk.choices[0]; const delta = choice.delta; + // Track usage from OpenAI chunk if available + if (chunk.usage && typeof chunk.usage === "object") { + const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0; + const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0; + + // Extract cache tokens from prompt_tokens_details + const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens; + const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens; + const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0; + const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0; + + // input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens + // Because OpenAI's prompt_tokens includes all prompt-side tokens + const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens; + + state.usage = { + input_tokens: inputTokens, + output_tokens: outputTokens + }; + + // Add cache_read_input_tokens if present + if (cacheReadTokens > 0) { + state.usage.cache_read_input_tokens = cacheReadTokens; + } + + // Add cache_creation_input_tokens if present + if (cacheCreateTokens > 0) { + state.usage.cache_creation_input_tokens = cacheCreateTokens; + } + + // Note: completion_tokens_details.reasoning_tokens is already included in output_tokens + // No need to add separately as Claude expects total output_tokens + } + // First chunk - ALWAYS send message_start first if (!state.messageStartSent) { state.messageStartSent = true; @@ -223,9 +221,8 @@ export function openaiToClaudeResponse(chunk, state) { } } - // Finish (OpenAI puts this on the choice; Qoder often puts it on delta) - const finishReason = choice.finish_reason || delta?.finish_reason; - if (finishReason) { + // Finish + if (choice.finish_reason) { stopThinkingBlock(state, results); stopTextBlock(state, results); @@ -247,13 +244,13 @@ export function openaiToClaudeResponse(chunk, state) { } // Mark finish for later usage injection in stream.js - state.finishReason = finishReason; + state.finishReason = choice.finish_reason; // Use tracked usage (will be estimated in stream.js if not valid) const finalUsage = state.usage || { input_tokens: 0, output_tokens: 0 }; results.push({ type: "message_delta", - delta: { stop_reason: convertFinishReason(finishReason) }, + delta: { stop_reason: convertFinishReason(choice.finish_reason) }, usage: finalUsage }); results.push({ type: "message_stop" }); diff --git a/open-sse/utils/stream.js b/open-sse/utils/stream.js index f61fd193..15ee37d8 100644 --- a/open-sse/utils/stream.js +++ b/open-sse/utils/stream.js @@ -3,7 +3,6 @@ import { FORMATS } from "../translator/formats.js"; import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb.js"; import { extractUsage, mergeUsage, hasValidUsage, estimateUsage, logUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js"; import { parseSSELine, hasValuableContent, fixInvalidId, formatSSE } from "./streamHelpers.js"; -import { toResponsesUsage } from "../translator/concerns/usage.js"; import { getOpenAIResponsesEventName, isOpenAIResponsesTerminalEvent, formatIncompleteOpenAIResponsesStreamFailure } from "./responsesStreamHelpers.js"; import { dbg, isDebugEnabled } from "./debugLog.js"; @@ -202,8 +201,7 @@ export function createSSEStream(options = {}) { responsesTerminal = isOpenAIResponsesTerminalEvent(currentOpenAIResponsesEvent, parsed); - const isFinishChunk = parsed.choices?.[0]?.finish_reason - || parsed.choices?.[0]?.delta?.finish_reason; + const isFinishChunk = parsed.choices?.[0]?.finish_reason; if (isFinishChunk && !hasValidUsage(parsed.usage)) { const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI); parsed.usage = filterUsageForFormat(estimated, FORMATS.OPENAI); @@ -367,19 +365,6 @@ export function createSSEStream(options = {}) { item.usage = filterUsageForFormat(buffered, sourceFormat); } - // Responses API clients (Codex, sub2api /v1/responses): usage lives on - // response.completed → response.usage. Same buffer/estimate policy as above. - const completedResponse = item.event === "response.completed" ? item.data?.response : null; - if (completedResponse && typeof completedResponse === "object") { - if (state.usage) { - completedResponse.usage = toResponsesUsage(addBufferToUsage(state.usage)) ?? completedResponse.usage; - } else if (!completedResponse.usage && totalContentLength > 0) { - const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI); - completedResponse.usage = toResponsesUsage(estimated); - state.usage = estimated; - } - } - const output = formatSSE(item, sourceFormat); reqLogger?.appendConvertedChunk?.(output); controller.enqueue(sharedEncoder.encode(output)); diff --git a/tests/unit/openai-responses-usage.test.js b/tests/unit/openai-responses-usage.test.js deleted file mode 100644 index 38ec581d..00000000 --- a/tests/unit/openai-responses-usage.test.js +++ /dev/null @@ -1,182 +0,0 @@ -/** - * Responses API clients (Codex, sub2api /v1/responses) read token usage only from - * `response.completed → response.usage`. For chat-native upstreams (Qoder, most - * OpenAI-compatible providers) the translator used to emit that event without usage, - * so proxies logged 0 input / 0 output / 0 cached tokens. - */ -import { describe, expect, it, vi } from "vitest"; - -vi.mock("@/lib/usageDb.js", () => ({ - appendRequestLog: vi.fn(async () => {}), - saveRequestDetail: vi.fn(async () => {}), - saveRequestUsage: vi.fn(async () => {}), - trackPendingRequest: vi.fn(() => {}), -})); - -const { FORMATS } = await import("../../open-sse/translator/formats.js"); -const { initState } = await import("../../open-sse/translator/index.js"); -const { toResponsesUsage } = await import("../../open-sse/translator/concerns/usage.js"); -const { openaiToOpenAIResponsesResponse } = await import("../../open-sse/translator/response/openai-responses.js"); -const { createSSETransformStreamWithLogger } = await import("../../open-sse/utils/stream.js"); -const { createResponsesApiTransformStream } = await import("../../open-sse/transformer/responsesTransformer.js"); -const { addBufferToUsage } = await import("../../open-sse/utils/usageTracking.js"); -// stream.js adds the same context-safety buffer it applies to chat/claude clients -const BUFFER_TOKENS = addBufferToUsage({ prompt_tokens: 0 }).prompt_tokens; - -const QODER_FINISH_CHUNK = { - id: "chatcmpl-qoder-1", - object: "chat.completion.chunk", - created: 1_700_000_000, - model: "qmodel_38max", - choices: [{ index: 0, delta: {}, finish_reason: "stop" }], - usage: { - prompt_tokens: 27_339, - completion_tokens: 437, - total_tokens: 27_776, - prompt_tokens_details: { cached_tokens: 27_200 }, - }, -}; - -function sse(chunks) { - return chunks.map((c) => `data: ${typeof c === "string" ? c : JSON.stringify(c)}\n\n`).join(""); -} - -async function pipe(input, transform) { - const encoder = new TextEncoder(); - const stream = new ReadableStream({ - start(controller) { - controller.enqueue(encoder.encode(input)); - controller.close(); - }, - }); - const reader = stream.pipeThrough(transform).getReader(); - const decoder = new TextDecoder(); - let text = ""; - for (;;) { - const { value, done } = await reader.read(); - if (done) break; - text += decoder.decode(value, { stream: true }); - } - return text + decoder.decode(); -} - -function completedEvent(text) { - const m = text.match(/event: response\.completed\ndata: (.+)\n/); - return m ? JSON.parse(m[1]) : null; -} - -describe("toResponsesUsage", () => { - it("maps OpenAI usage (nested cached_tokens) to the Responses shape", () => { - expect(toResponsesUsage(QODER_FINISH_CHUNK.usage)).toEqual({ - input_tokens: 27_339, - output_tokens: 437, - total_tokens: 27_776, - input_tokens_details: { cached_tokens: 27_200 }, - output_tokens_details: { reasoning_tokens: 0 }, - }); - }); - - it("accepts canonical flat fields and Claude-style cache fields", () => { - expect(toResponsesUsage({ prompt_tokens: 10, completion_tokens: 2, cached_tokens: 4, reasoning_tokens: 1 })).toMatchObject({ - input_tokens: 10, - output_tokens: 2, - total_tokens: 12, - input_tokens_details: { cached_tokens: 4 }, - output_tokens_details: { reasoning_tokens: 1 }, - }); - expect(toResponsesUsage({ input_tokens: 5, output_tokens: 1, cache_read_input_tokens: 3 }).input_tokens_details.cached_tokens).toBe(3); - }); - - it("keeps the estimated marker and returns null for empty usage", () => { - expect(toResponsesUsage({ prompt_tokens: 1, completion_tokens: 1, estimated: true }).estimated).toBe(true); - expect(toResponsesUsage({})).toBeNull(); - expect(toResponsesUsage(null)).toBeNull(); - }); -}); - -describe("openai → openai-responses translator", () => { - it("puts usage from the finish chunk on response.completed", () => { - const state = initState(FORMATS.OPENAI_RESPONSES); - const events = openaiToOpenAIResponsesResponse(QODER_FINISH_CHUNK, state); - const completed = events.find((e) => e.event === "response.completed"); - expect(completed).toBeTruthy(); - expect(completed.data.response.usage).toEqual({ - input_tokens: 27_339, - output_tokens: 437, - total_tokens: 27_776, - input_tokens_details: { cached_tokens: 27_200 }, - output_tokens_details: { reasoning_tokens: 0 }, - }); - }); - - it("omits usage when the upstream never reported any", () => { - const state = initState(FORMATS.OPENAI_RESPONSES); - const events = openaiToOpenAIResponsesResponse({ ...QODER_FINISH_CHUNK, usage: undefined }, state); - const completed = events.find((e) => e.event === "response.completed"); - expect(completed.data.response.usage).toBeUndefined(); - }); -}); - -describe("stream.js translate mode: chat upstream → Responses client", () => { - const transform = () => createSSETransformStreamWithLogger( - FORMATS.OPENAI, // provider (Qoder executor emits OpenAI chunks) - FORMATS.OPENAI_RESPONSES, // client - "qoder", - null, - null, - "qmodel_38max", - null, - { model: "qd/qmodel_38max", messages: [{ role: "user", content: "hi" }] }, - ); - - it("emits provider usage (+buffer) with cached tokens on response.completed", async () => { - const out = await pipe(sse([ - { ...QODER_FINISH_CHUNK, choices: [{ index: 0, delta: { role: "assistant", content: "Hello" }, finish_reason: null }], usage: undefined }, - QODER_FINISH_CHUNK, - "[DONE]", - ]), transform()); - - const completed = completedEvent(out); - expect(completed).toBeTruthy(); - expect(completed.response.usage).toEqual({ - input_tokens: 27_339 + BUFFER_TOKENS, - output_tokens: 437, - total_tokens: 27_776 + BUFFER_TOKENS, - input_tokens_details: { cached_tokens: 27_200 }, - output_tokens_details: { reasoning_tokens: 0 }, - }); - // Responses clients terminate on response.completed (no [DONE] sentinel in translate mode) - expect(out.indexOf("event: response.completed")).toBeGreaterThan(out.indexOf("event: response.output_item.done")); - }); - - it("injects estimated usage when the upstream reports none", async () => { - const out = await pipe(sse([ - { ...QODER_FINISH_CHUNK, choices: [{ index: 0, delta: { role: "assistant", content: "Hello world" }, finish_reason: null }], usage: undefined }, - { ...QODER_FINISH_CHUNK, usage: undefined }, - "[DONE]", - ]), transform()); - - const completed = completedEvent(out); - expect(completed.response.usage).toBeTruthy(); - expect(completed.response.usage.estimated).toBe(true); - expect(completed.response.usage.input_tokens).toBeGreaterThan(0); - expect(completed.response.usage.output_tokens).toBeGreaterThan(0); - }); -}); - -describe("responsesTransformer (Chat SSE → Codex Responses SSE)", () => { - it("forwards finish-chunk usage on response.completed", async () => { - const out = await pipe(sse([ - { ...QODER_FINISH_CHUNK, choices: [{ index: 0, delta: { role: "assistant", content: "Hello" }, finish_reason: null }], usage: undefined }, - QODER_FINISH_CHUNK, - "[DONE]", - ]), createResponsesApiTransformStream()); - - const completed = completedEvent(out); - expect(completed.response.usage).toMatchObject({ - input_tokens: 27_339, - output_tokens: 437, - input_tokens_details: { cached_tokens: 27_200 }, - }); - }); -}); diff --git a/tests/unit/openai-to-claude.test.js b/tests/unit/openai-to-claude.test.js index 0c5768d7..45b67fb2 100644 --- a/tests/unit/openai-to-claude.test.js +++ b/tests/unit/openai-to-claude.test.js @@ -203,31 +203,4 @@ describe("openaiToClaudeResponse", () => { limit: 120 }); }); - - it("records usage from a choices:[] frame so the finish chunk can emit it", () => { - const state = { toolCalls: new Map() }; - expect(openaiToClaudeResponse({ - usage: { - prompt_tokens: 90, - completion_tokens: 7, - prompt_tokens_details: { cached_tokens: 30 }, - }, - choices: [], - }, state)).toBeNull(); - expect(state.usage).toEqual({ - input_tokens: 60, - output_tokens: 7, - cache_read_input_tokens: 30, - }); - - const events = openaiToClaudeResponse({ - id: "chatcmpl-qoder-finish", - model: "qoder/auto", - choices: [{ index: 0, delta: {}, finish_reason: "stop" }], - }, state); - const delta = events.find((e) => e.type === "message_delta"); - expect(delta.usage.input_tokens).toBe(60); - expect(delta.usage.output_tokens).toBe(7); - expect(delta.usage.cache_read_input_tokens).toBe(30); - }); }); From da6aa90128c0861beccdeea1a6f23b0d1816673a Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 23:18:45 +0700 Subject: [PATCH 42/78] fix(video/vertex): reject job ids and model ids that escape the URL path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A Vertex job id is a base64url-encoded operation name, and base64url decoding accepts arbitrary bytes without throwing, so the previous decode-and-split check let a crafted id splice a path traversal into the fetch URL while the Bearer token stayed attached — e.g. "..%2F..%2Fevil" resolved to /v1/evil:fetchPredictOperation on the Vertex host. body.model had the same shape on the create path, where it is interpolated into the URL unescaped. decodeJobId now requires a charset-only id, a byte-for-byte round-trip, and a decoded name matching ^projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}$ — no field may contain "/", so ".." can never reach the URL. model ids are restricted to [A-Za-z0-9._-]. Co-Authored-By: Claude Code --- open-sse/handlers/videoProviders/vertex.js | 34 +++++++++++++++------- tests/unit/video-providers.test.js | 33 +++++++++++++++++++++ 2 files changed, 57 insertions(+), 10 deletions(-) diff --git a/open-sse/handlers/videoProviders/vertex.js b/open-sse/handlers/videoProviders/vertex.js index 25394b94..a4ff66c6 100644 --- a/open-sse/handlers/videoProviders/vertex.js +++ b/open-sse/handlers/videoProviders/vertex.js @@ -13,7 +13,25 @@ import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefre const DEFAULT_LOCATION = "us-central1"; const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url"); -const decodeJobId = (id) => Buffer.from(id, "base64url").toString("utf8"); + +// Operation name shape: projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}. +// Anchored and single-segment-per-field so a decoded path can never carry `..` or a +// host-changing prefix into the request URL. +const OPERATION_NAME_RE = /^projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/models\/[^/]+\/operations\/[^/]+$/; + +function modelPathOf(operationName) { + return operationName.slice(0, operationName.indexOf("/operations/")); +} + +function decodeJobId(id) { + const raw = String(id ?? ""); + // Buffer.from(x, "base64url") silently drops invalid characters instead of + // throwing, so only ids that re-encode byte-for-byte are accepted. + if (!raw || raw.length > 1024 || !/^[A-Za-z0-9_-]+$/.test(raw)) return null; + const decoded = Buffer.from(raw, "base64url").toString("utf8"); + if (Buffer.from(decoded, "utf8").toString("base64url") !== raw) return null; + return OPERATION_NAME_RE.test(decoded) ? decoded : null; +} async function resolveAuth(credentials, log) { const saJson = parseVertexSaJson(credentials?.apiKey); @@ -103,17 +121,11 @@ export default { const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` }; if (requestId) { - let operationName; - try { - operationName = decodeJobId(requestId); - } catch { - return { error: "Invalid Vertex video job id" }; - } - const modelPath = operationName.split("/operations/")[0]; - if (!modelPath || modelPath === operationName) return { error: "Invalid Vertex video job id" }; + const operationName = decodeJobId(requestId); + if (!operationName) return { error: "Invalid Vertex video job id" }; return { method: "POST", - url: `${base}/v1/${modelPath}:fetchPredictOperation`, + url: `${base}/v1/${modelPathOf(operationName)}:fetchPredictOperation`, headers, body: JSON.stringify({ operationName }), }; @@ -131,6 +143,8 @@ export default { return { error: "Invalid JSON body" }; } if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" }; + // Plain model id only — a path segment carrying "/" or ".." would rewrite the URL. + if (!/^[A-Za-z0-9._-]+$/.test(body.model)) return { error: "Invalid Vertex video model id" }; if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" }; return { diff --git a/tests/unit/video-providers.test.js b/tests/unit/video-providers.test.js index 680db5a5..bfa26589 100644 --- a/tests/unit/video-providers.test.js +++ b/tests/unit/video-providers.test.js @@ -260,4 +260,37 @@ describe("vertex (veo) video adapter", () => { expect(result.status).toBe(400); expect(global.fetch).not.toHaveBeenCalled(); }); + + // A base64url id decodes to arbitrary bytes, so a crafted one used to splice a + // path traversal into the fetch URL while the Authorization header stayed on. + it("rejects job ids that decode outside the projects/…/operations/ shape", async () => { + refreshVertexToken.mockResolvedValue({ accessToken: "vertex-tok" }); + const jid = (s) => Buffer.from(s, "utf8").toString("base64url"); + + for (const id of [ + jid("../../evil"), + jid("projects/p/locations/l/publishers/google/models/m/operations/../../x"), + jid("../../evil/operations/op"), + "!!!not-base64!!!", + `${JOB_ID}=`, + `${JOB_ID}\n`, + ]) { + const result = await handleVideoProxyCore({ provider: "vertex", requestId: id, credentials: { apiKey: saJson } }); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + } + }); + + it("rejects a model id carrying path separators", async () => { + refreshVertexToken.mockResolvedValue({ accessToken: "vertex-tok" }); + const result = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ model: "../../evil", prompt: "x" }), + contentType: "application/json", + credentials: { apiKey: saJson }, + }); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + }); }); From c712641123d3d155ecce9d375f15415026d89547 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 23:18:58 +0700 Subject: [PATCH 43/78] feat(opencode-go): list deepseek-v4.1-flash first in the model catalog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Registry order is the display order for the provider page, /v1/models, and the CLI selector, so moving the entry to the head of `models` is the whole change. deepseek-flash keeps its id and supportedFormats — only its position moves. Co-Authored-By: Claude Code --- open-sse/providers/registry/opencode-go.js | 2 +- tests/unit/opencode-go-models.test.js | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index a715a98e..1d1d12c0 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -35,6 +35,7 @@ export default { ], // supportedFormats follow the endpoint table in https://opencode.ai/docs/go/ models: [ + { id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] }, { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] }, { id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] }, { id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] }, @@ -45,7 +46,6 @@ export default { { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] }, - { id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] }, { id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] }, { id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] }, { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] }, diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 0ed0804b..4ee2042a 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -24,8 +24,9 @@ describe("OpenCode Go model catalog", () => { it("matches the documented model IDs", () => { const ids = (PROVIDER_MODELS["opencode-go"] || []).map((m) => m.id); expect(ids).toEqual([ + "deepseek-flash", "glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", - "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "deepseek-flash", + "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", "minimax-m3", "minimax-m2.7", "minimax-m2.5", "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", From accf2c52969b0876625ffa8d66a5add803dc056f Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 23:32:55 +0700 Subject: [PATCH 44/78] fix(providers): remove the duplicate qwen provider MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit qwen.js points at the same transport as alims-intl — identical baseUrl (dashscope-intl compatible-mode), headers, and quirks — but carries only 8 Qwen models against alims-intl's broader catalog, so it adds nothing a user could not already reach. Node count is back to 81. The id was dropped on 2026-08-05 (dcdd4628b) when the Qwen OAuth flow died; this removes the standalone API-key entry that shadowed it. The `qw` alias returns to unassigned, its state before today. Co-Authored-By: Claude Code --- open-sse/providers/registry/index.js | 2 -- open-sse/providers/registry/qwen.js | 36 ---------------------------- 2 files changed, 38 deletions(-) delete mode 100644 open-sse/providers/registry/qwen.js diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 48a49ebe..fb355d6d 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -123,7 +123,6 @@ import p119 from "./selfhosted-embedding.js"; import p120 from "./fish-audio.js"; import p121 from "./alitp-intl.js"; import p122 from "./xquik.js"; -import p124 from "./qwen.js"; export default [ p0, p1, @@ -247,5 +246,4 @@ export default [ p120, p121, p122, - p124, ]; diff --git a/open-sse/providers/registry/qwen.js b/open-sse/providers/registry/qwen.js deleted file mode 100644 index d51e88a8..00000000 --- a/open-sse/providers/registry/qwen.js +++ /dev/null @@ -1,36 +0,0 @@ -export default { - id: "qwen", - priority: 12, - alias: "qwen", - aliases: ["qw"], - display: { - name: "Qwen", - icon: "sparkles", - color: "#6366F1", - textIcon: "Qw", - website: "https://www.alibabacloud.com/en/product/model-studio", - notice: { - apiKeyUrl: - "https://modelstudio.console.alibabacloud.com/?apiKey=1", - }, - }, - category: "apikey", - transport: { - baseUrl: - "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions", - headers: {}, - quirks: { preserveCacheControl: true }, - }, - models: [ - // Flagship & API models - { id: "qwen3.8-max", name: "Qwen3.8 Max" }, - { id: "qwen3.8-flash", name: "Qwen3.8 Flash" }, - { id: "qwen3.7-plus", name: "Qwen3.7 Plus" }, - { id: "qwen3.7-flash", name: "Qwen3.7 Flash" }, - // Open-source models - { id: "qwen3.8-2.4t-a95b", name: "Qwen3.8 2.4T MoE (Open)" }, - { id: "qwen3.8-27b", name: "Qwen3.8 27B (Open)" }, - { id: "qwen3.6-35b-a3b", name: "Qwen3.6 35B MoE (Open)" }, - { id: "qwen3.6-27b", name: "Qwen3.6 27B (Open)" }, - ], -}; From 83af3f1853b295eda428f3d3d8f5ff96ca25bfab Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 10 Sep 2026 23:35:10 +0700 Subject: [PATCH 45/78] # v0.5.75 (2026-09-10) Covers the 24 commits since the v0.5.69 tag. The package version already moved to 0.5.75 in 4a390685b (CLI model selector), so the release commit is the changelog alone, matching the convention in eb712ca82/4eda76e2a. Co-Authored-By: Claude Code --- CHANGELOG.md | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 39b3010c..a2b6139d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,29 @@ +# v0.5.75 (2026-09-10) + +## Features +- **Video**: add OpenRouter and Vertex AI (Veo) video generation on `/v1/videos/*` via a provider adapter layer; poll requests resolve their provider from `x-connection-id` or `?provider=` +- **Antigravity**: add weekly quota tracking (Gemini weekly / Claude & GPT weekly) and free-tier handling from `retrieveUserQuotaSummary` (#3892) +- **Codex**: add GPT Image 2.5, Flare and Sunburst image models with multi-image support; add the same ids to the OpenAI catalog +- **Qoder**: surface usage to all clients and stop inlining large attachments — images upload through `/api/v2/image/upload` like qodercli, oversized file blocks become stubs, context tier auto-escalates +- **OpenCode Go**: add newly published models (glm-5.3, kimi-k3, deepseek-flash, longcat-2.0, hy4-preview, hy3 on chat/completions; qwen3.8-max, qwen3.8-flash on `/messages`; grok-4.6, gpt-5.6-luna on Responses) and list `deepseek-v4.1-flash` first in the catalog +- **CLI tools**: group the model selector by provider with full-text search and manual custom model ID entry +- **CodeBuddy-CN**: replace `deepseek-v4-flash` with `deepseek-v4.1-flash` + +## Fixes +- **Tools**: scope Claude tool type defaulting to gateways declaring `requireClaudeToolType` — the global default broke Anthropic-compatible endpoints that only accept the legacy typeless tool shape (#3905) +- **Claude**: cap re-anchored `cache_control` at the 4-marker budget so a spent budget no longer 400s and triggers a full combo failover; wrap bare single-object content turns before the mid-conversation-system fold +- **Cline / Airforce**: unwrap the `{"success":true,"data":…}` envelope on non-stream chat completions (#3644); add the live Cline/ClinePass model catalog and refresh Airforce free models +- **Cline**: stop `workos:`-prefixing ClinePass API keys (401 on every request, #2333) and add clinepass token refresh +- **Kiro**: never send a top-level `systemPrompt` (`400 REQUEST_BODY_INVALID`); route requests through current runtime surfaces (#3776) +- **Codex**: strip Unicode-property tool schema patterns the validator rejects (#3922); restore the `Version` header and single-source the CLI version +- **DeepSeek**: keep Anthropic-only tool types when forwarding to `/anthropic/v1/messages` +- **Qoder**: drop the Responses usage plumbing from shared translator/handler code, which changed token accounting for every provider, not just Qoder +- **Antigravity**: normalize contents and handle intermediate tool responses; protect the OAuth token-refresh path from Google anti-abuse rate limits (#3813) +- **Providers**: clear stale connection health state (`modelLock_*`, `backoffLevel`, `rateLimitedUntil`, `errorCode`) when a connection is re-validated (#3810, #3830); remove the duplicate `qwen` provider that shadowed `alims-intl` +- **Video / Vertex**: reject job ids and model ids that would escape the request URL path (SSRF) +- **Usage**: parse the Fable weekly limit from `limits[]` instead of fabricating a row (#3847) +- **Auth**: set a 24h `maxAge` on the dashboard session cookie + # v0.5.69 (2026-09-05) ## Features From 73cb89143c29575e098e460b514b44c234c67739 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=8F=B6=E7=82=9C=E6=9C=8B?= Date: Thu, 10 Sep 2026 23:41:40 +0700 Subject: [PATCH 46/78] feat(xiaomi-mimo): merge MiMo Desktop support into xiaomi-mimo as dual auth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the Desktop-exclusive Preview models and the Xiaomi account-session route to the existing xiaomi-mimo provider instead of a separate xiaomi-desktop provider, so the dashboard shows one MiMo entry rather than three overlapping ones. Dual auth, same pattern as kimi — API key (sk-) covers the cloud API, Desktop/OAuth adds the account session used by the Preview models: - registry: category oauth, authModes [oauth, apikey], oauth block, the two mimo-x-*-preview models, and the invite signupUrl - executor: routes Preview models to the account-service route with a Cookie session, everything else keeps the sourceFormat-matched transport - oauth: custom ECDH encrypted-callback flow (X25519 -> SHA256 -> AES-256-GCM) with a loopback callback proxy, plus one-click import of the local Desktop auth.json - usage: weekly quota from the account session Fixes found while merging: - the OAuth browser flow was dead: poll-status cleared the session before the client could POST /exchange, so every exchange returned 400 - a Claude-format client was sent to /v1/chat/completions instead of the declared /anthropic/v1/messages transport, because buildUrl ignored runtimeTransport - stopXiaomiMimoProxy leaked every pending session (each holding an X25519 private key) for the process lifetime - the OAuth exchange did not persist the Desktop passToken, so the Preview models could never work after a browser sign-in Removes dead code: the local engine token minting (mimoEngine, never called on the request path), the model-catalog and usage routes, engineToken/ engineUrl plumbing, and an unread top-level usage block. Adds tests/unit/xiaomi-mimo-{executor,oauth-session,oauth-proxy}.test.js — the provider previously had none. --- .gitignore | 11 + open-sse/executors/index.js | 3 + open-sse/executors/xiaomi-mimo.js | 99 +++++++ open-sse/providers/registry/xiaomi-mimo.js | 32 +- open-sse/services/usage.js | 2 + open-sse/services/usage/xiaomi-mimo.js | 125 ++++++++ open-sse/shared/mimoAccount.js | 264 +++++++++++++++++ open-sse/translator/concerns/paramSupport.js | 3 + .../dashboard/providers/[id]/page.js | 15 +- .../api/oauth/[provider]/[action]/route.js | 115 +++++++- .../api/oauth/xiaomi-mimo/api-key/route.js | 136 +++++++++ .../oauth/xiaomi-mimo/auto-import/route.js | 140 +++++++++ src/lib/db/driver.js | 4 + src/lib/oauth/constants/oauth.js | 15 + src/lib/oauth/providers/xiaomi-mimo.js | 123 ++++++++ src/lib/oauth/utils/server.js | 182 ++++++++++++ src/shared/components/XiaomiMimoAuthModal.js | 276 ++++++++++++++++++ src/shared/components/index.js | 1 + tests/unit/xiaomi-mimo-executor.test.js | 80 +++++ tests/unit/xiaomi-mimo-oauth-proxy.test.js | 54 ++++ tests/unit/xiaomi-mimo-oauth-session.test.js | 152 ++++++++++ 21 files changed, 1828 insertions(+), 4 deletions(-) create mode 100644 open-sse/executors/xiaomi-mimo.js create mode 100644 open-sse/services/usage/xiaomi-mimo.js create mode 100644 open-sse/shared/mimoAccount.js create mode 100644 src/app/api/oauth/xiaomi-mimo/api-key/route.js create mode 100644 src/app/api/oauth/xiaomi-mimo/auto-import/route.js create mode 100644 src/lib/oauth/providers/xiaomi-mimo.js create mode 100644 src/shared/components/XiaomiMimoAuthModal.js create mode 100644 tests/unit/xiaomi-mimo-executor.test.js create mode 100644 tests/unit/xiaomi-mimo-oauth-proxy.test.js create mode 100644 tests/unit/xiaomi-mimo-oauth-session.test.js diff --git a/.gitignore b/.gitignore index 6773d550..e17d69b7 100644 --- a/.gitignore +++ b/.gitignore @@ -90,3 +90,14 @@ graphify-out/* # Kiro local workspace state .kiro/ 9router-* + +# Local sensitive / temp files +.engine-token.txt +.tmp-prov.json +.tmp-providers.json +.oauth-session.json +.dev-server.log +.dev-server-err.log +.start-dev.ps1 +start-dev-silent.cjs +debug.log diff --git a/open-sse/executors/index.js b/open-sse/executors/index.js index 8dd03421..f48a8ecb 100644 --- a/open-sse/executors/index.js +++ b/open-sse/executors/index.js @@ -17,6 +17,7 @@ import { PerplexityWebExecutor } from "./perplexity-web.js"; import { OllamaLocalExecutor } from "./ollama-local.js"; import { CommandCodeExecutor } from "./commandcode.js"; import { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js"; +import { XiaomiMimoExecutor } from "./xiaomi-mimo.js"; import { MimoFreeExecutor } from "./mimo-free.js"; import { CodeBuddyExecutor } from "./codebuddy-cn.js"; import { CodeBuddyIntlExecutor } from "./codebuddy-intl.js"; @@ -50,6 +51,7 @@ const executors = { "ollama-local": new OllamaLocalExecutor(), commandcode: new CommandCodeExecutor(), "xiaomi-tokenplan": new XiaomiTokenplanExecutor(), + "xiaomi-mimo": new XiaomiMimoExecutor(), "mimo-free": new MimoFreeExecutor(), mmf: new MimoFreeExecutor(), // Alias for mimo-free "codebuddy-cn": new CodeBuddyExecutor(), @@ -93,6 +95,7 @@ export { PerplexityWebExecutor } from "./perplexity-web.js"; export { OllamaLocalExecutor } from "./ollama-local.js"; export { CommandCodeExecutor } from "./commandcode.js"; export { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js"; +export { XiaomiMimoExecutor } from "./xiaomi-mimo.js"; export { MimoFreeExecutor } from "./mimo-free.js"; export { CodeBuddyExecutor } from "./codebuddy-cn.js"; export { CodeBuddyIntlExecutor } from "./codebuddy-intl.js"; diff --git a/open-sse/executors/xiaomi-mimo.js b/open-sse/executors/xiaomi-mimo.js new file mode 100644 index 00000000..8b69412a --- /dev/null +++ b/open-sse/executors/xiaomi-mimo.js @@ -0,0 +1,99 @@ +import { DefaultExecutor } from "./default.js"; +import { getMimoAccountCookie, invalidateMimoAccountCookieCache, MIMO_API_BASE, MIMO_API_UA } from "../shared/mimoAccount.js"; + +// Desktop-exclusive Preview models. These are served by the account service's +// /api/route proxy, authorized by the Xiaomi account session (NOT the sk- key). +// See shared/mimoAccount.js for the session handshake. +const PREVIEW_MODELS = new Set(["mimo-x-pro-preview", "mimo-x-flash-preview"]); + +// Session cookie resolved in execute() (async) and read back by buildHeaders() +// (sync — BaseExecutor.execute does not await it). Carried on the per-request +// credentials object, same as runtimeTransport. +const COOKIE_KEY = "__mimoAccountCookie"; + +// Upstream calls may hand us either the bare id or a `provider/model` ref. +function bareModel(model) { + const s = String(model || ""); + const i = s.indexOf("/"); + return i >= 0 ? s.slice(i + 1) : s; +} + +export class XiaomiMimoExecutor extends DefaultExecutor { + constructor() { + super("xiaomi-mimo"); + } + + static isPreviewModel(model) { + return PREVIEW_MODELS.has(bareModel(model)); + } + + buildUrl(model, stream, urlIndex = 0, credentials = null) { + // Preview models live on the account-service route, which is not one of the + // declared transports — resolve it before the default runtimeTransport path. + if (XiaomiMimoExecutor.isPreviewModel(model)) { + return `${MIMO_API_BASE}/api/route/chat/completions`; + } + // Cloud API models keep default handling, so a Claude-format client reaches + // the /anthropic/v1/messages transport. + return super.buildUrl(model, stream, urlIndex, credentials); + } + + buildHeaders(credentials, stream = true, url, model) { + if (XiaomiMimoExecutor.isPreviewModel(model) && credentials?.[COOKIE_KEY]) { + // Preview models authenticate with the account-session cookie, not the key. + return { + "Content-Type": "application/json", + Accept: stream ? "text/event-stream" : "application/json", + "User-Agent": MIMO_API_UA, + Cookie: credentials[COOKIE_KEY], + }; + } + return super.buildHeaders(credentials, stream, url, model); + } + + transformRequest(model, body, stream, credentials) { + // super runs stripUnsupportedParams, which flattens Preview content-part + // arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js). + const out = super.transformRequest(model, body, stream, credentials); + + // Preview models: thinking/params get defaults only — never override what the + // caller set explicitly. (body.model is already `xiaomi/` via upstreamModelId.) + if (XiaomiMimoExecutor.isPreviewModel(model)) { + if (out.thinking == null) out.thinking = { type: "enabled" }; + if (out.temperature == null) out.temperature = 1.0; + if (out.top_p == null) out.top_p = 0.95; + if (!out.max_tokens) out.max_tokens = 4096; + } + + return out; + } + + async execute(args) { + const { model, credentials, proxyOptions = null } = args; + if (!XiaomiMimoExecutor.isPreviewModel(model)) return super.execute(args); + + const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions); + if (!cookie) { + throw new Error( + "Xiaomi MiMo account session unavailable. Sign in to MiMo Desktop once so its passToken is present, then retry.", + ); + } + credentials[COOKIE_KEY] = cookie; + const result = await super.execute(args); + + // A cached session can expire early — drop it and retry once with a fresh one. + if (result.response.status === 401) { + invalidateMimoAccountCookieCache(); + const fresh = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions).catch(() => null); + if (fresh) { + credentials[COOKIE_KEY] = fresh; + return super.execute(args); + } + } + return result; + } +} + +export const __test__ = { PREVIEW_MODELS, bareModel, COOKIE_KEY }; + +export default XiaomiMimoExecutor; diff --git a/open-sse/providers/registry/xiaomi-mimo.js b/open-sse/providers/registry/xiaomi-mimo.js index 49465f43..cb0139b3 100644 --- a/open-sse/providers/registry/xiaomi-mimo.js +++ b/open-sse/providers/registry/xiaomi-mimo.js @@ -1,11 +1,19 @@ import { CLAUDE_API_HEADERS } from "../shared.js"; +// Dual auth (same pattern as kimi): +// - API key (sk-...) → cloud API on api.xiaomimimo.com +// - Desktop account/OAuth → same cloud host, plus the Desktop-exclusive Preview +// models served by the account-service route on mimo-server-cn.xiaomimimo.com +// (authorized by a Xiaomi account session cookie, not the key). +// Endpoint is picked per model in the executor, same as opencode-go's /responses split. export default { id: "xiaomi-mimo", priority: 290, alias: "xiaomi-mimo", aliases: [ "mimo", + "mimo-desktop", + "xmd", ], uiAlias: "mimo", display: { @@ -16,9 +24,12 @@ export default { website: "https://xiaomimimo.com", notice: { apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys", + signupUrl: "https://mimo.xiaomimimo.com/desktop/invite/", }, }, - category: "apikey", + category: "oauth", + authModes: ["oauth", "apikey"], + hasOAuth: true, serviceKinds: ["llm", "tts"], transport: { baseUrl: "https://api.xiaomimimo.com/v1/chat/completions", @@ -39,6 +50,11 @@ export default { }, ], models: [ + // Desktop-exclusive — served by the account-service route, which only accepts + // OpenAI format, so supportedFormats pins them to the openai transport. + { id: "mimo-x-pro-preview", name: "MiMo-X-Pro-Preview", upstreamModelId: "xiaomi/mimo-x-pro-preview", supportedFormats: ["openai"] }, + { id: "mimo-x-flash-preview", name: "MiMo-X-Flash-Preview", upstreamModelId: "xiaomi/mimo-x-flash-preview", supportedFormats: ["openai"] }, + // Cloud API models (api.xiaomimimo.com/v1) { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" }, { id: "mimo-v2.5", name: "MiMo V2.5" }, { id: "mimo-v2-omni", name: "MiMo V2 Omni" }, @@ -51,4 +67,18 @@ export default { authHeader: "bearer", format: "xiaomi-mimo-tts", }, + features: { + usage: true, + usageApikey: true, + }, + // Custom OAuth — non-standard ECDH encrypted-callback flow. + // Handled by the Xiaomi MiMo OAuth service, not the generic PKCE pipeline. + oauth: { + custom: true, + authorizeUrl: "https://platform.xiaomimimo.com/authorize", + // The callback carries ?u= instead of ?code=. + // Decryption yields { uid, sk, url }. + callbackParam: "u", + kn: "mimocode", + }, }; diff --git a/open-sse/services/usage.js b/open-sse/services/usage.js index eeb46c3c..d7868760 100644 --- a/open-sse/services/usage.js +++ b/open-sse/services/usage.js @@ -17,6 +17,7 @@ import { getDeepseekUsage } from "./usage/deepseek.js"; import { getOpenCodeGoUsage } from "./usage/opencode-go.js"; import { getGroqUsage } from "./usage/groq.js"; import { getZedUsage } from "./usage/zed.js"; +import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js"; import { resolveQoderCredentials } from "./qoderModels.js"; import { getGlmUsage } from "./usage/glm.js"; import { @@ -60,6 +61,7 @@ const USAGE_HANDLERS = { deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions), groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions), zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), + "xiaomi-mimo": (c) => getXiaomiMimoUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), }; export async function getUsageForProvider(connection, proxyOptions = null, options = {}) { diff --git a/open-sse/services/usage/xiaomi-mimo.js b/open-sse/services/usage/xiaomi-mimo.js new file mode 100644 index 00000000..9d42fe37 --- /dev/null +++ b/open-sse/services/usage/xiaomi-mimo.js @@ -0,0 +1,125 @@ +/** + * Xiaomi MiMo usage — weekly quota from the Xiaomi account session. + * + * Primary path: GET {mimo-server}/api/user/usage authorized by the account-session + * cookie (see shared/mimoAccount.js). Response: { code: 0, data: { percent (remaining + * %), resetDate, resetAt } }. + * + * Fallback: the sk- API key cannot read the quota, so when no account session is + * available we surface a graceful message instead of failing. + */ + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { getMimoAccountUsage } from "../../shared/mimoAccount.js"; + +const USAGE_URL = "https://aistudio.xiaomimimo.com/open-apis/v1/user/usage"; + +/** + * @param {string|null|undefined} accessToken - sk- API key + * @param {object|null} providerSpecificData - may contain mimoPassToken, uid, etc. + * @param {object|null} proxyOptions + */ +export async function getXiaomiMimoUsage(accessToken = null, providerSpecificData = null, proxyOptions = null) { + // Preferred path: the weekly quota comes from the account service session + // (mimo-server /api/user/usage), which the sk- key cannot reach. The session is + // derived from MiMo Desktop's persisted passToken via the SSO/sts handshake. + const account = await getMimoAccountUsage(providerSpecificData, proxyOptions); + if (typeof account.percent === "number" && Number.isFinite(account.percent)) { + const remaining = Math.max(0, Math.min(100, Math.round(account.percent))); + const used = 100 - remaining; + let resetAt = null; + if (typeof account.resetAt === "number" && account.resetAt > 0) { + resetAt = new Date(account.resetAt * 1000).toISOString(); + } else if (typeof account.resetDate === "string") { + const parsed = new Date(`${account.resetDate}T00:00:00Z`); + if (!Number.isNaN(parsed.getTime())) resetAt = parsed.toISOString(); + } + return { + plan: "Xiaomi MiMo Desktop", + quotas: { + Weekly: { used, total: 100, remainingPercentage: remaining, resetAt, unlimited: false }, + }, + }; + } + + // Fallback: no account session available (Desktop never logged in, or its cookie + // store is locked). The sk- key cannot read the quota, so surface a clear message. + const key = accessToken || providerSpecificData?.apiKey; + if (!key || typeof key !== "string" || !key.trim()) { + return { message: "Xiaomi MiMo Desktop not connected. Add credentials to view usage." }; + } + + try { + const response = await proxyAwareFetch( + USAGE_URL, + { + method: "GET", + headers: { + Authorization: `Bearer ${key.trim()}`, + "X-Mimo-Source": "mimocode-cli", + Accept: "application/json", + }, + signal: AbortSignal.timeout(10000), + }, + proxyOptions, + ); + + if (response.status === 401) { + return { + plan: "Xiaomi MiMo Desktop", + message: "Weekly quota requires Xiaomi account session. API key alone is insufficient.", + }; + } + + if (!response.ok) { + return { + plan: "Xiaomi MiMo Desktop", + message: `Usage API error (${response.status})`, + }; + } + + const data = await response.json().catch(() => null); + if (!data || data.code !== 0 || !data.data) { + return { + plan: "Xiaomi MiMo Desktop", + message: "Usage endpoint returned unexpected response.", + }; + } + + const { percent, resetDate } = data.data; + if (typeof percent !== "number" || !Number.isFinite(percent)) { + return { + plan: "Xiaomi MiMo Desktop", + message: "Usage data missing percent field.", + }; + } + + // percent = remaining percentage (e.g. 94 means 94% remaining) + const remaining = Math.max(0, Math.min(100, Math.round(percent))); + const used = 100 - remaining; + + // Parse resetDate — expected format "2026-09-16" + let resetAt = null; + if (resetDate && typeof resetDate === "string") { + const parsed = new Date(`${resetDate}T00:00:00Z`); + if (!Number.isNaN(parsed.getTime())) { + resetAt = parsed.toISOString(); + } + } + + return { + plan: "Xiaomi MiMo Desktop", + quotas: { + Weekly: { + used, + total: 100, + remainingPercentage: remaining, + resetAt, + unlimited: false, + }, + }, + }; + } catch (error) { + return { message: `Xiaomi MiMo Desktop usage error: ${error.message}` }; + } +} diff --git a/open-sse/shared/mimoAccount.js b/open-sse/shared/mimoAccount.js new file mode 100644 index 00000000..996f38df --- /dev/null +++ b/open-sse/shared/mimoAccount.js @@ -0,0 +1,264 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import crypto from "node:crypto"; +import { proxyAwareFetch } from "../utils/proxyFetch.js"; + +/** + * Xiaomi MiMo account-session helpers (used for weekly quota). + * + * The weekly quota endpoint lives on the account service domain and is authorized + * by an account session cookie, NOT the sk- API key. Acquiring that cookie mirrors + * MiMo Desktop: a passToken (persisted in Desktop's cookie store) is exchanged via + * the passportapi SSO, then authorized for the `mimopc` service, and finally stamped + * by the mimo-server /api/sts callback into a `serviceToken` cookie. + * + * Flow (verified against MiMo Desktop traffic): + * 1. GET {api}/api/user/xiaomi/me -> 302 to account SSO (sid=mimopc) + * 2. GET account /pass/serviceLogin?sid=passportapi&_json=true -> nonce/ssecurity + * 3. GET {location}&clientSign=... -> account-level serviceToken + * 4. GET account /pass/serviceLogin?sid=mimopc&callback=&_json=true + * 5. GET {api}/api/sts?...&ticket... -> Set-Cookie: serviceToken (mimopc scope) + */ + +const API_BASE = "https://mimo-server-cn.xiaomimimo.com"; +const ACCOUNT_HOST = "account.xiaomi.com"; +const API_UA = + "miNative PC/Normal Windows_NT/10.0.19045 SDKV/1.0.0 DEVT/PC DEVS/Windows APP/miaccount_desktop APPV/0.1.0"; +const SSO_UA = "MiClaw/1.0"; +const COOKIE_TTL_MS = 30 * 60 * 1000; + +// Per-account session caches (keyed by passToken hash) so multiple Xiaomi +// accounts / connections can rotate without clobbering each other. +const _cache = new Map(); // key -> { cookie, at } +const _inflight = new Map(); // key -> Promise + +function desktopCookiePath() { + const home = os.homedir(); + if (process.platform === "win32") { + return path.join(home, "AppData", "Roaming", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies"); + } + if (process.platform === "darwin") { + return path.join(home, "Library", "Application Support", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies"); + } + return path.join(home, ".config", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies"); +} + +/** + * Read the persisted Xiaomi account cookies from MiMo Desktop's Electron profile. + * The Chromium cookie DB is held with an exclusive lock while Desktop runs, so we + * copy it first and bail (return null) if that fails. + * @returns {Promise|null>} + */ +async function readDesktopAccountCookies() { + const src = desktopCookiePath(); + if (!fs.existsSync(src)) return null; + const tmp = path.join(os.tmpdir(), `9r-mimo-cookies-${process.pid}-${crypto.randomBytes(4).toString("hex")}.db`); + try { + fs.copyFileSync(src, tmp); + } catch { + return null; // locked by a running Desktop + } + try { + const { DatabaseSync } = await import("node:sqlite"); + const db = new DatabaseSync(tmp, { readOnly: true }); + const rows = db.prepare("SELECT name, value FROM cookies WHERE host_key = ?").all("." + ACCOUNT_HOST); + db.close(); + const jar = Object.fromEntries(rows.map((r) => [r.name, r.value])); + return jar.passToken ? jar : null; + } catch { + return null; + } finally { + try { + fs.unlinkSync(tmp); + } catch { + /* ignore */ + } + } +} + +/** + * Read just the passToken + identity cookies from Desktop's profile. + * Exported so the connect flow can persist a per-account passToken into the + * connection's providerSpecificData — this is what enables multi-account rotation. + * @returns {Promise<{passToken:string, userId:string|null, cUserId:string|null}|null>} + */ +export async function readDesktopPassToken() { + try { + const jar = await readDesktopAccountCookies(); + if (!jar?.passToken) return null; + return { passToken: jar.passToken, userId: jar.userId || null, cUserId: jar.cUserId || null }; + } catch { + return null; + } +} + +function signatureClientSign(nonce, ssecurity) { + const input = `nonce=${nonce}` + (ssecurity && ssecurity.trim() ? `&${ssecurity}` : ""); + return encodeURIComponent(crypto.createHash("sha1").update(input).digest("base64")); +} + +function absorbSetCookie(jar, res) { + for (const c of res.headers.getSetCookie?.() || []) { + const m = /^([^=]+)=([^;]*)/.exec(c.trim()); + if (m && m[2]) jar[m[1]] = m[2]; + } +} + +function cookieHeader(jar) { + return Object.entries(jar) + .filter(([, v]) => v) + .map(([k, v]) => `${k}=${v}`) + .join("; "); +} + +/** + * Exchange a passToken for a mimo-server service session cookie. + * @returns {Promise} Cookie header value, or null on failure. + */ +async function acquireServiceCookie(passJar, proxyOptions) { + const jar = { ...passJar }; + const ck = () => cookieHeader(jar); + + // 1. Unauthenticated API call -> 302 carrying the sts callback (sid=mimopc) + const r1 = await proxyAwareFetch( + `${API_BASE}/api/user/xiaomi/me`, + { redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } }, + proxyOptions, + ); + const redirect = r1.headers.get("location"); + if (!redirect) return null; + const stsCallback = new URL(redirect).searchParams.get("callback"); + if (!stsCallback) return null; + + // 2. passportapi SSO phase 1 -> nonce + ssecurity + const sso1 = await proxyAwareFetch( + `https://${ACCOUNT_HOST}/pass/serviceLogin?sid=passportapi&_json=true`, + { headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } }, + proxyOptions, + ); + const j1 = JSON.parse((await sso1.text()).replace(/^&&&START&&&/, "")); + const nonce = j1.nonce || (j1.location ? new URL(j1.location).searchParams.get("nonce") : null); + if (!nonce || !j1.location) return null; + + // 3. passportapi SSO phase 2 -> account-level serviceToken + const sso2 = await proxyAwareFetch( + `${j1.location}&clientSign=${signatureClientSign(nonce, j1.ssecurity)}`, + { redirect: "manual", headers: { Cookie: ck(), "User-Agent": SSO_UA } }, + proxyOptions, + ); + absorbSetCookie(jar, sso2); + + // 4. mimopc SSO -> sts callback carrying a ticket + const sso3 = await proxyAwareFetch( + `https://${ACCOUNT_HOST}/pass/serviceLogin?sid=mimopc&callback=${encodeURIComponent(stsCallback)}&_json=true`, + { headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } }, + proxyOptions, + ); + const j3 = JSON.parse((await sso3.text()).replace(/^&&&START&&&/, "")); + absorbSetCookie(jar, sso3); + if (!j3?.location || !/\/api\/sts/.test(j3.location)) return null; + + // 5. sts callback -> Set-Cookie: serviceToken (mimopc scope) + const sts = await proxyAwareFetch( + j3.location, + { redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } }, + proxyOptions, + ); + absorbSetCookie(jar, sts); + + const needed = ["serviceToken", "mimopc_ph", "mimopc_slh", "userId"]; + if (!jar.serviceToken) return null; + const out = {}; + for (const k of needed) if (jar[k]) out[k] = jar[k]; + return cookieHeader(out); +} + +/** + * Get (and cache) the mimo-server account cookie. + * @param {object|null} providerSpecificData - may carry `mimoPassToken` override + */ +async function getServiceCookie(providerSpecificData, proxyOptions) { + const passJar = providerSpecificData?.mimoPassToken + ? { passToken: providerSpecificData.mimoPassToken, userId: providerSpecificData.mimoUserId, cUserId: providerSpecificData.mimoCUserId } + : await readDesktopAccountCookies(); + if (!passJar) return { cookie: null, reason: "no-pass-token" }; + + // One cached session per passToken — accounts/connections rotate independently. + const key = crypto.createHash("sha256").update(passJar.passToken).digest("hex"); + + const cached = _cache.get(key); + if (cached && Date.now() - cached.at < COOKIE_TTL_MS) { + return { cookie: cached.cookie }; + } + + // De-dupe concurrent handshakes for the same account: a burst of requests must + // not each run the full 5-step SSO chain. + const inflight = _inflight.get(key); + if (inflight) { + const cookie = await inflight; + return cookie ? { cookie } : { cookie: null, reason: "sso-failed" }; + } + + const promise = (async () => { + try { + return await acquireServiceCookie(passJar, proxyOptions); + } catch { + return null; // network/parse failure — callers degrade, never throw + } finally { + _inflight.delete(key); + } + })(); + _inflight.set(key, promise); + + const cookie = await promise; + if (!cookie) return { cookie: null, reason: "sso-failed" }; + _cache.set(key, { cookie, at: Date.now() }); + return { cookie }; +} + +/** Drop cached sessions so the next call re-runs the handshake (e.g. after a 401). */ +export function invalidateMimoAccountCookieCache() { + _cache.clear(); +} + +/** mimo-server account API base + the User-Agent its backend expects. */ +export const MIMO_API_BASE = API_BASE; +export const MIMO_API_UA = API_UA; + +/** + * Resolve the mimo-server account-session cookie, for upstream /api/route/* calls. + * @returns {Promise} Cookie header value, or null when unavailable. + */ +export async function getMimoAccountCookie(providerSpecificData = null, proxyOptions = null) { + try { + const { cookie } = await getServiceCookie(providerSpecificData, proxyOptions); + return cookie; + } catch { + return null; + } +} + +/** + * Fetch the weekly quota from the account service. + * @returns {Promise<{percent?:number, resetDate?:string, resetAt?:number, error?:string}>} + */ +export async function getMimoAccountUsage(providerSpecificData = null, proxyOptions = null) { + const { cookie, reason } = await getServiceCookie(providerSpecificData, proxyOptions); + if (!cookie) { + return { error: reason === "no-pass-token" ? "no-session" : "session-failed" }; + } + try { + const res = await proxyAwareFetch( + `${API_BASE}/api/user/usage`, + { headers: { "User-Agent": API_UA, Cookie: cookie, Accept: "application/json" }, signal: AbortSignal.timeout(10000) }, + proxyOptions, + ); + if (!res.ok) return { error: `http-${res.status}` }; + const data = await res.json().catch(() => null); + if (!data || data.code !== 0 || !data.data) return { error: "bad-response" }; + return { percent: data.data.percent, resetDate: data.data.resetDate, resetAt: data.data.resetAt }; + } catch (e) { + return { error: e.message }; + } +} diff --git a/open-sse/translator/concerns/paramSupport.js b/open-sse/translator/concerns/paramSupport.js index e222b23f..863b627e 100644 --- a/open-sse/translator/concerns/paramSupport.js +++ b/open-sse/translator/concerns/paramSupport.js @@ -14,6 +14,9 @@ const STRIP_RULES = [ { provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] }, // Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926) { provider: "cloudflare-ai", flattenContent: true }, + // MiMo Desktop Preview models (account-service route): content must be plain string, + // rejects OpenAI content-part array. Cloud models keep their parts (mimo-v2-omni is multi-modal). + { provider: "xiaomi-mimo", match: /preview/i, flattenContent: true }, { provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true }, // VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's // advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144), diff --git a/src/app/(dashboard)/dashboard/providers/[id]/page.js b/src/app/(dashboard)/dashboard/providers/[id]/page.js index 9657b7ed..5ae217fb 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/page.js +++ b/src/app/(dashboard)/dashboard/providers/[id]/page.js @@ -5,7 +5,7 @@ import { useParams, useRouter } from "next/navigation"; import Link from "next/link"; import Image from "next/image"; import { getProviderIconSrc, markProviderIconMissing } from "@/shared/utils/providerIcon"; -import { Card, Button, Badge, Input, Modal, CardSkeleton, OAuthModal, KiroOAuthWrapper, CursorAuthModal, IFlowCookieModal, GitLabAuthModal, Toggle, Select, EditConnectionModal, NoAuthProxyCard, ConfirmModal } from "@/shared/components"; +import { Card, Button, Badge, Input, Modal, CardSkeleton, OAuthModal, KiroOAuthWrapper, CursorAuthModal, XiaomiMimoAuthModal, IFlowCookieModal, GitLabAuthModal, Toggle, Select, EditConnectionModal, NoAuthProxyCard, ConfirmModal } from "@/shared/components"; import { OAUTH_PROVIDERS, APIKEY_PROVIDERS, FREE_PROVIDERS, FREE_TIER_PROVIDERS, WEB_COOKIE_PROVIDERS, getProviderAlias, isOpenAICompatibleProvider, isAnthropicCompatibleProvider, AI_PROVIDERS } from "@/shared/constants/providers"; import { getModelsByProviderId, getModelKind } from "@/shared/constants/models"; import { getThinkingLevels } from "open-sse/providers/thinkingLevels.js"; @@ -45,6 +45,7 @@ export default function ProviderDetailPage() { const [providerNode, setProviderNode] = useState(null); const [proxyPools, setProxyPools] = useState([]); const [showOAuthModal, setShowOAuthModal] = useState(false); + const [showXiaomiMimoModal, setShowXiaomiMimoModal] = useState(false); const [showIFlowCookieModal, setShowIFlowCookieModal] = useState(false); const [showAddApiKeyModal, setShowAddApiKeyModal] = useState(false); const [addConnectionError, setAddConnectionError] = useState(""); @@ -98,6 +99,11 @@ export default function ProviderDetailPage() { return; } } + // Xiaomi Desktop: auto-import local credentials first, OAuth as fallback + if (providerId === "xiaomi-mimo") { + setShowXiaomiMimoModal(true); + return; + } if (isOAuth) { openOAuthConnection(); return; @@ -1796,6 +1802,13 @@ export default function ProviderDetailPage() { onClose={() => setShowOAuthModal(false)} /> )} + + {/* Xiaomi Desktop: auto-import local credentials modal */} + setShowXiaomiMimoModal(false)} + /> {providerId === "iflow" && ( c.provider === "xiaomi-mimo" && ( + (uid && c.email === `${uid}@xiaomi`) || + c.accessToken === key + ), + ); + if (existing) { + const updated = await updateProviderConnection(existing.id, { + accessToken: key, + providerSpecificData: { + ...existing.providerSpecificData, + uid: uid || existing.providerSpecificData?.uid || null, + baseUrl: effectiveBaseUrl, + // Per-account session credential — enables multi-account rotation. + mimoPassToken: mimoPassToken || existing.providerSpecificData?.mimoPassToken || null, + mimoUserId: mimoUserId || existing.providerSpecificData?.mimoUserId || null, + mimoCUserId: mimoCUserId || existing.providerSpecificData?.mimoCUserId || null, + modelCount, + }, + testStatus: validated ? "active" : existing.testStatus, + }); + return NextResponse.json({ + success: true, + validated, + modelCount, + updated: true, + connection: { + id: existing.id, + provider: existing.provider, + email: existing.email, + displayName: existing.displayName, + }, + }); + } + + const connection = await createProviderConnection({ + provider: "xiaomi-mimo", + authType: "api_key", + accessToken: key, + refreshToken: null, + // API keys don't expire on a fixed schedule; use a long horizon + expiresAt: new Date(Date.now() + 365 * 24 * 60 * 60 * 1000).toISOString(), + email: uid ? `${uid}@xiaomi` : null, + displayName: uid ? `Xiaomi ${uid}` : "Xiaomi MiMo", + providerSpecificData: { + uid: uid || null, + baseUrl: effectiveBaseUrl, + authMethod: "api_key", + provider: "API Key", + modelCount, + // Per-account session credential — enables multi-account rotation. + mimoPassToken: mimoPassToken || null, + mimoUserId: mimoUserId || null, + mimoCUserId: mimoCUserId || null, + }, + testStatus: validated ? "active" : "untested", + }); + + return NextResponse.json({ + success: true, + validated, + modelCount, + connection: { + id: connection.id, + provider: connection.provider, + email: connection.email, + displayName: connection.displayName, + }, + }); + } catch (error) { + console.log("Xiaomi MiMo API key import error:", error); + return NextResponse.json( + { error: "API key import failed" }, + { status: 500 }, + ); + } +} diff --git a/src/app/api/oauth/xiaomi-mimo/auto-import/route.js b/src/app/api/oauth/xiaomi-mimo/auto-import/route.js new file mode 100644 index 00000000..0508c222 --- /dev/null +++ b/src/app/api/oauth/xiaomi-mimo/auto-import/route.js @@ -0,0 +1,140 @@ +import { NextResponse } from "next/server"; +import { readFile, access, constants } from "fs/promises"; +import { homedir } from "os"; +import { join } from "path"; +import { readDesktopPassToken } from "open-sse/shared/mimoAccount.js"; + +/** + * GET /api/oauth/xiaomi-mimo/auto-import + * Auto-detect Xiaomi MiMo credentials from local auth.json. + * + * Sources (in priority order): + * 1. ~/.local/share/mimocode/auth.json → xiaomi field + * 2. %APPDATA%/Xiaomi MiMo/... → (future: Desktop keychain) + * + * auth.json shape: + * { + * "xiaomi": { + * "type": "api", + * "key": "sk-xxxx", + * "metadata": { "uid": "...", "base_url": "https://api.xiaomimimo.com/v1" } + * } + * } + */ + +function getCandidatePaths() { + const home = homedir(); + const paths = []; + + // MiMoCode / MiMo Desktop shared data dir (cross-platform XDG) + paths.push(join(home, ".local", "share", "mimocode", "auth.json")); + + // Windows: also check USERPROFILE-based XDG + if (process.platform === "win32") { + const appData = process.env.APPDATA || join(home, "AppData", "Roaming"); + // Desktop's own storage (may have separate credentials in the future) + paths.push(join(appData, "Xiaomi MiMo", "auth.json")); + } + + // macOS + if (process.platform === "darwin") { + paths.push( + join(home, "Library", "Application Support", "mimocode", "auth.json"), + ); + } + + return paths; +} + +/** + * GET /api/oauth/xiaomi-mimo/auto-import + */ +export async function GET() { + try { + const candidates = getCandidatePaths(); + + let authPath = null; + for (const candidate of candidates) { + try { + await access(candidate, constants.R_OK); + authPath = candidate; + break; + } catch { + // Try next candidate + } + } + + if (!authPath) { + return NextResponse.json({ + found: false, + error: `Xiaomi MiMo Desktop auth file not found. Checked:\n${candidates.join("\n")}\n\nMake sure Xiaomi MiMo Desktop is installed and you are signed in.`, + }); + } + + const raw = await readFile(authPath, "utf-8"); + let auth; + try { + auth = JSON.parse(raw); + } catch { + return NextResponse.json({ + found: false, + error: "auth.json is not valid JSON. Please sign in to Xiaomi MiMo Desktop again.", + }); + } + + const xiaomi = auth?.xiaomi; + if (!xiaomi || !xiaomi.key) { + return NextResponse.json({ + found: false, + error: "No Xiaomi credentials found in auth.json. Please sign in to Xiaomi MiMo Desktop.", + }); + } + + // Validate key format + const key = String(xiaomi.key).trim(); + if (!key.startsWith("sk-")) { + return NextResponse.json({ + found: false, + error: "Xiaomi key does not appear to be a valid API key (expected sk- prefix).", + }); + } + + const metadata = xiaomi.metadata || {}; + const uid = metadata.uid || null; + const baseUrl = metadata.base_url || "https://api.xiaomimimo.com/v1"; + + // Account-session passToken from Desktop's cookie store. Persisting it per + // connection is what lets multiple Xiaomi accounts rotate independently. + // (null while Desktop is running — its cookie DB is exclusively locked.) + let mimoPassToken = null; + let mimoUserId = null; + let mimoCUserId = null; + try { + const pt = await readDesktopPassToken(); + if (pt) { + mimoPassToken = pt.passToken; + mimoUserId = pt.userId; + mimoCUserId = pt.cUserId; + } + } catch (e) { + console.log("[xiaomi-mimo] passToken read failed (non-fatal):", e.message); + } + + return NextResponse.json({ + found: true, + apiKey: key, + uid, + baseUrl, + source: authPath, + mimoPassToken, + mimoUserId, + mimoCUserId, + }); + } catch (error) { + console.log("Xiaomi MiMo auto-import error:", error); + return NextResponse.json( + { found: false, error: error.message }, + { status: 500 }, + ); + } +} diff --git a/src/lib/db/driver.js b/src/lib/db/driver.js index 050514d9..17b6dd49 100644 --- a/src/lib/db/driver.js +++ b/src/lib/db/driver.js @@ -19,6 +19,10 @@ async function tryBunSqlite() { async function tryBetterSqlite() { // Skip on Bun — better-sqlite3 native bindings unsupported if (process.versions.bun) return null; + // Skip on Node >= 24: the native addon SIGSEGVs on load there, which is a + // process-level crash the try/catch below cannot recover from. node:sqlite covers it. + const [nodeMajor] = process.versions.node.split(".").map(Number); + if (nodeMajor >= 24) return null; try { const { createBetterSqliteAdapter } = await import("./adapters/betterSqliteAdapter.js"); return createBetterSqliteAdapter(DATA_FILE); diff --git a/src/lib/oauth/constants/oauth.js b/src/lib/oauth/constants/oauth.js index c9e3cae6..77cd3c13 100644 --- a/src/lib/oauth/constants/oauth.js +++ b/src/lib/oauth/constants/oauth.js @@ -130,6 +130,21 @@ export const GROK_CLI_CONFIG = { ...PROVIDER_OAUTH["grok-cli"] }; // 3) Redirect → ${cb}?refreshToken=...&loginHost=...&isRedirect=true // 4) POST ExchangeToken {ClientID, RefreshToken, ClientSecret:"-"} → {Result.AccessToken, ExpiresAt} // 5) POST GetUserInfo (x-cloudide-token) → email/name +// Xiaomi MiMo Desktop OAuth — custom ECDH encrypted-callback flow (NOT standard OAuth2). +// 1) Client generates X25519 keypair +// 2) Browser opens ${platformUrl}/authorize?pk=&redirect_uri=http://localhost:/&kn=mimocode&key_name=... +// 3) Redirect → http://localhost:/?u= +// 4) Decrypt: ECDH(shared) → SHA256 → AES-256-GCM +// Layout: [12-byte nonce][32-byte ephemeral pubkey][ciphertext][16-byte GCM tag] +// 5) Result JSON: { uid, sk, url } +export const XIAOMI_MIMO_CONFIG = { + platformUrl: process.env.MIMO_PLATFORM_URL || "https://platform.xiaomimimo.com", + defaultBaseUrl: "https://api.xiaomimimo.com/v1", + kn: "mimocode", + callbackPath: "/", + timeoutMs: 300000, // 5 minutes +}; + export const TRAE_CONFIG = { clientId: "ono9krqynydwx5", clientSecret: "-", diff --git a/src/lib/oauth/providers/xiaomi-mimo.js b/src/lib/oauth/providers/xiaomi-mimo.js new file mode 100644 index 00000000..ff85cf7b --- /dev/null +++ b/src/lib/oauth/providers/xiaomi-mimo.js @@ -0,0 +1,123 @@ +import crypto from "crypto"; +import { XIAOMI_MIMO_CONFIG } from "../constants/oauth.js"; + +// ─────────────────────────────────────────────────────────────────────────── +// Xiaomi MiMo OAuth helpers +// Custom ECDH + AES-256-GCM encrypted-callback flow (NOT standard OAuth2). +// ─────────────────────────────────────────────────────────────────────────── + +/** + * Generate an X25519 keypair for the OAuth handshake. + * @returns {{ publicKey: string, privateKeyDer: Buffer }} + * publicKey — base64 SPKI (for the `pk` URL param) + * privateKeyDer — PKCS8 DER Buffer (for ECDH later) + */ +export function generateKeyPair() { + const { publicKey, privateKey } = crypto.generateKeyPairSync("x25519"); + + const publicKeyDer = publicKey.export({ format: "der", type: "spki" }); + // SPKI for X25519 is 44 bytes; the raw 32-byte key is the last 32 bytes. + // But the platform expects the full base64 SPKI — pass as-is. + const publicKeyB64 = publicKeyDer.toString("base64"); + + const privateKeyDer = privateKey.export({ format: "der", type: "pkcs8" }); + + return { publicKey: publicKeyB64, privateKeyDer }; +} + +/** + * Decrypt the `u` query parameter from the Xiaomi OAuth callback. + * + * Wire format (base64-decoded): + * bytes 0..11 — 12-byte AES-GCM nonce + * bytes 12..43 — 32-byte ephemeral public key (raw X25519) + * bytes 44..n-16 — ciphertext + * last 16 bytes — GCM auth tag + * + * Key derivation: SHA256(ECDH(clientPrivateKey, ephemeralPublicKey)) + * + * @param {Buffer} privateKeyDer — PKCS8 DER private key from generateKeyPair() + * @param {string} encryptedB64 — the `u` query param value (base64) + * @returns {{ uid: string, sk: string, url?: string }} + */ +export function decryptCallback(privateKeyDer, encryptedB64) { + const raw = Buffer.from(encryptedB64, "base64"); + + if (raw.length < 12 + 32 + 16 + 1) { + throw new Error(`Encrypted payload too short: ${raw.length} bytes`); + } + + const nonce = raw.subarray(0, 12); + const ephemeralPubRaw = raw.subarray(12, 44); + const ciphertextAndTag = raw.subarray(44); + const tag = ciphertextAndTag.subarray(ciphertextAndTag.length - 16); + const ciphertext = ciphertextAndTag.subarray(0, ciphertextAndTag.length - 16); + + // Reconstruct the ephemeral public key as SPKI DER for Node crypto. + // X25519 SPKI prefix: 302a300506032b656e032100 + const ephemeralPub = crypto.createPublicKey({ + key: Buffer.concat([ + Buffer.from("302a300506032b656e032100", "hex"), + ephemeralPubRaw, + ]), + format: "der", + type: "spki", + }); + + const privateKey = crypto.createPrivateKey({ + key: privateKeyDer, + format: "der", + type: "pkcs8", + }); + + const sharedSecret = crypto.diffieHellman({ privateKey, publicKey: ephemeralPub }); + const derivedKey = crypto.createHash("sha256").update(sharedSecret).digest(); + + const decipher = crypto.createDecipheriv("aes-256-gcm", derivedKey, nonce); + decipher.setAuthTag(tag); + const decrypted = Buffer.concat([decipher.update(ciphertext), decipher.final()]); + + const parsed = JSON.parse(decrypted.toString("utf-8")); + + if (!parsed || typeof parsed !== "object") { + throw new Error("Decrypted payload is not a valid object"); + } + + return { + uid: parsed.uid || null, + sk: parsed.sk || null, + url: parsed.url || XIAOMI_MIMO_CONFIG.defaultBaseUrl, + }; +} + +/** + * Build the browser authorization URL. + * @param {string} publicKey — base64 SPKI from generateKeyPair() + * @param {string} redirectUri — e.g. http://localhost:12345/ + * @param {string} [keyName] — optional stable key name + * @returns {string} + */ +export function buildAuthorizeUrl(publicKey, redirectUri, keyName) { + const params = new URLSearchParams({ + pk: publicKey, + redirect_uri: redirectUri, + kn: XIAOMI_MIMO_CONFIG.kn, + }); + if (keyName) params.set("key_name", keyName); + return `${XIAOMI_MIMO_CONFIG.platformUrl}/authorize?${params.toString()}`; +} + +/** + * Get or create a stable key name for this installation. + * Stored in the 9Router data dir so re-auth reuses the same name. + */ +export function getKeyName() { + // Use a deterministic name based on machine — avoids needing filesystem writes + // in the OAuth provider layer. The platform treats key_name as a label only. + const machineId = crypto + .createHash("sha256") + .update(`${process.platform}-${process.env.COMPUTERNAME || process.env.HOSTNAME || "unknown"}`) + .digest("hex") + .slice(0, 8); + return `9router-xmd-${machineId}`; +} diff --git a/src/lib/oauth/utils/server.js b/src/lib/oauth/utils/server.js index 56eb67b1..80377752 100644 --- a/src/lib/oauth/utils/server.js +++ b/src/lib/oauth/utils/server.js @@ -755,3 +755,185 @@ export function stopZedProxy() { zedProxyPort = null; } +// ─────────────────────────────────────────────────────────────────────────── +// Xiaomi MiMo Desktop OAuth callback proxy +// Receives the ECDH-encrypted `u` param, decrypts it, stores the session. +// ─────────────────────────────────────────────────────────────────────────── + +let xiaomiMimoProxyServer = null; +let xiaomiMimoProxyPort = null; +let xiaomiMimoProxyTimeout = null; + +const xiaomiMimoSessions = new Map(); + +export function registerXiaomiMimoSession({ state, privateKeyDer }) { + if (!state || !privateKeyDer) return false; + xiaomiMimoSessions.set(state, { + privateKeyDer, + status: "pending", + createdAt: Date.now(), + }); + return true; +} + +export function getXiaomiMimoSessionStatus(state) { + const s = xiaomiMimoSessions.get(state); + if (!s) return null; + // Don't leak the private key to the client + return { status: s.status, result: s.result || null, error: s.error || null }; +} + +export function clearXiaomiMimoSession(state) { + xiaomiMimoSessions.delete(state); +} + +function renderXiaomiMimoResultPage(success, message) { + const color = success ? "#22c55e" : "#ef4444"; + const icon = success ? "✓" : "✗"; + const title = success ? "Authentication Successful" : "Authentication Failed"; + return ` + +${title} + + + +
+
${icon}
+

${title}

+

${message || (success ? "You can close this tab and return to 9Router." : "Please try again.")}

+ ${success ? "" : ""} +
+ +`; +} + +/** + * Start the Xiaomi Desktop OAuth callback proxy. + * @returns {Promise<{success: boolean, port?: number, callbackUrl?: string, reason?: string}>} + */ +export function startXiaomiMimoProxy() { + return new Promise((resolve) => { + if (xiaomiMimoProxyServer) { + resolve({ + success: true, + port: xiaomiMimoProxyPort, + callbackUrl: `http://127.0.0.1:${xiaomiMimoProxyPort}/`, + }); + return; + } + + const server = http.createServer(async (req, res) => { + // Origin guard + if (!isLoopbackOrigin(req.headers.origin)) { + res.writeHead(403); + res.end("Forbidden"); + return; + } + + const url = new URL(req.url, "http://127.0.0.1"); + const u = url.searchParams.get("u"); + + if (!u) { + res.writeHead(400, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(false, "Missing encrypted payload (u parameter).")); + return; + } + + // Try each pending session's private key — the callback URL carries no + // state param, so we attempt decryption with every pending key. + const pendingSessions = [...xiaomiMimoSessions.entries()] + .filter(([, s]) => s.status === "pending"); + + if (pendingSessions.length === 0) { + res.writeHead(500, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(false, "No active OAuth session. Please restart the login flow.")); + return; + } + + try { + const { decryptCallback } = await import("../providers/xiaomi-mimo.js"); + let result = null; + let matchedState = null; + + for (const [state, session] of pendingSessions) { + try { + result = decryptCallback(session.privateKeyDer, u); + matchedState = state; + break; + } catch { + // Wrong key for this session — try next + } + } + + if (!result || !matchedState) { + throw new Error("Could not decrypt with any pending session key"); + } + + if (!result.sk) { + throw new Error("Decrypted payload missing sk (API key)"); + } + + // Store result only in the matched session + const session = xiaomiMimoSessions.get(matchedState); + if (session) { + session.status = "done"; + session.result = { + uid: result.uid, + accessToken: result.sk, + baseUrl: result.url || "https://api.xiaomimimo.com/v1", + }; + } + + res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(true, "Xiaomi account linked. You can close this tab.")); + console.log("[xiaomi-mimo oauth] callback decrypted, uid:", result.uid); + } catch (err) { + console.error("[xiaomi-mimo oauth] decrypt failed:", err.message); + for (const [, session] of pendingSessions) { + session.status = "error"; + session.error = err.message; + } + res.writeHead(400, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(false, `Decryption failed: ${err.message}`)); + } + }); + + server.on("error", (err) => { + console.log("[xiaomi-mimo oauth] listen error:", err.message); + resolve({ success: false, reason: err.message }); + }); + + server.listen(0, "127.0.0.1", () => { + xiaomiMimoProxyServer = server; + xiaomiMimoProxyPort = server.address().port; + xiaomiMimoProxyTimeout = setTimeout(() => { + console.log("[xiaomi-mimo oauth] timeout, stopping"); + stopXiaomiMimoProxy(); + }, 300000); + console.log(`[xiaomi-mimo oauth] listening on port ${xiaomiMimoProxyPort}`); + resolve({ + success: true, + port: xiaomiMimoProxyPort, + callbackUrl: `http://127.0.0.1:${xiaomiMimoProxyPort}/`, + }); + }); + }); +} + +export function stopXiaomiMimoProxy() { + console.log(`[xiaomi-mimo oauth] stopping (port ${xiaomiMimoProxyPort || "-"})`); + if (xiaomiMimoProxyTimeout) { clearTimeout(xiaomiMimoProxyTimeout); xiaomiMimoProxyTimeout = null; } + if (xiaomiMimoProxyServer) { xiaomiMimoProxyServer.close(); xiaomiMimoProxyServer = null; } + xiaomiMimoProxyPort = null; + // No callback can arrive once the listener is down, so drop every pending + // session — each holds an X25519 private key and they would otherwise + // accumulate for the process lifetime (one per /authorize call). + xiaomiMimoSessions.clear(); +} + diff --git a/src/shared/components/XiaomiMimoAuthModal.js b/src/shared/components/XiaomiMimoAuthModal.js new file mode 100644 index 00000000..e501698f --- /dev/null +++ b/src/shared/components/XiaomiMimoAuthModal.js @@ -0,0 +1,276 @@ +"use client"; + +import { useState, useEffect } from "react"; +import PropTypes from "prop-types"; +import { Modal, Button } from "@/shared/components"; + +/** + * Xiaomi MiMo Auth Modal + * + * Auto-imports credentials from the local Xiaomi MiMo Desktop auth.json (~/.local/share/mimocode/auth.json). + * If auto-import fails, offers a one-click browser OAuth fallback. + * Reached only via the "Connect with OAuth" button — the API-key path uses the + * standard Add API Key modal, since Xiaomi MiMo supports both auth modes. + */ +export default function XiaomiMimoAuthModal({ isOpen, onSuccess, onClose }) { + const [phase, setPhase] = useState("detecting"); // detecting | found | not-found | importing | error + const [detectResult, setDetectResult] = useState(null); + const [error, setError] = useState(null); + const [oauthUrl, setOauthUrl] = useState(null); + const [oauthState, setOauthState] = useState(null); + + // Auto-detect local credentials when modal opens + useEffect(() => { + if (!isOpen) return; + let cancelled = false; + + (async () => { + setPhase("detecting"); + setError(null); + setDetectResult(null); + setOauthUrl(null); + + try { + const res = await fetch("/api/oauth/xiaomi-mimo/auto-import"); + const data = await res.json(); + if (cancelled) return; + + if (data.found && data.apiKey) { + setDetectResult(data); + setPhase("found"); + } else { + setPhase("not-found"); + setError(data.error || "Xiaomi MiMo Desktop credentials not found on this machine."); + } + } catch { + if (!cancelled) { + setPhase("not-found"); + setError("Failed to read local Xiaomi MiMo Desktop credentials."); + } + } + })(); + + return () => { cancelled = true; }; + }, [isOpen]); + + // Import the auto-detected key + const handleImport = async () => { + if (!detectResult?.apiKey) return; + setPhase("importing"); + setError(null); + + try { + const res = await fetch("/api/oauth/xiaomi-mimo/api-key", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + apiKey: detectResult.apiKey, + uid: detectResult.uid, + baseUrl: detectResult.baseUrl, + mimoPassToken: detectResult.mimoPassToken || null, + mimoUserId: detectResult.mimoUserId || null, + mimoCUserId: detectResult.mimoCUserId || null, + }), + }); + const data = await res.json(); + + if (!res.ok || !data.success) { + throw new Error(data.error || "Import failed"); + } + + onSuccess?.(data.connection); + onClose(); + } catch (err) { + setPhase("found"); + setError(err.message); + } + }; + + // Start browser OAuth fallback + const handleStartOAuth = async () => { + setError(null); + try { + const state = crypto.randomUUID(); + const res = await fetch(`/api/oauth/xiaomi-mimo/authorize?state=${state}`); + const data = await res.json(); + if (data.authorizeUrl) { + setOauthUrl(data.authorizeUrl); + setOauthState(data.state); + window.open(data.authorizeUrl, "_blank", "width=600,height=700"); + } else { + throw new Error(data.error || "Failed to start OAuth"); + } + } catch (err) { + setError(err.message); + } + }; + + // Poll OAuth result + const handlePollOAuth = async () => { + if (!oauthState) return; + setError(null); + try { + const res = await fetch(`/api/oauth/xiaomi-mimo/poll-status?state=${oauthState}`); + const data = await res.json(); + + if (data.status === "done" && data.result) { + // Exchange to create the connection + const exRes = await fetch("/api/oauth/xiaomi-mimo/exchange", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ state: oauthState }), + }); + const exData = await exRes.json(); + if (exData.success) { + onSuccess?.(exData.connection); + onClose(); + } else { + throw new Error(exData.error || "Exchange failed"); + } + } else if (data.status === "error") { + throw new Error(data.error || "OAuth failed"); + } else { + setError("Authorization not completed yet. Finish in the browser, then click Check Again."); + } + } catch (err) { + setError(err.message); + } + }; + + return ( + +
+ {/* Detecting */} + {phase === "detecting" && ( +
+
+ + progress_activity + +
+

Reading local credentials...

+

+ Checking ~/.local/share/mimocode/auth.json +

+
+ )} + + {/* Found — one-click import */} + {phase === "found" && detectResult && ( + <> +
+
+ check_circle +
+

Xiaomi MiMo Desktop credentials found!

+

+ UID: {detectResult.uid || "—"} · Source: {detectResult.source?.split(/[\\/]/).pop()} +

+
+
+
+ + {error && ( +
+

{error}

+
+ )} + +
+ + +
+ + )} + + {/* Importing */} + {phase === "importing" && ( +
+
+ + progress_activity + +
+

Connecting...

+
+ )} + + {/* Not found — offer OAuth fallback */} + {phase === "not-found" && ( + <> +
+
+ info +
+

Local credentials not found

+

{error}

+

+ Make sure Xiaomi MiMo Desktop is installed and you are signed in, then retry. + Or sign in via browser below. +

+
+
+
+ + {!oauthUrl ? ( +
+ + +
+ ) : ( +
+
+

+ Browser opened. Complete the Xiaomi sign-in, then click{" "} + Check Again. +

+
+
+ + +
+
+ )} + + )} +
+
+ ); +} + +XiaomiMimoAuthModal.propTypes = { + isOpen: PropTypes.bool.isRequired, + onSuccess: PropTypes.func, + onClose: PropTypes.func.isRequired, +}; diff --git a/src/shared/components/index.js b/src/shared/components/index.js index c04453a9..3d508e91 100644 --- a/src/shared/components/index.js +++ b/src/shared/components/index.js @@ -28,6 +28,7 @@ export { default as KiroAuthModal } from "./KiroAuthModal"; export { default as KiroOAuthWrapper } from "./KiroOAuthWrapper"; export { default as KiroSocialOAuthModal } from "./KiroSocialOAuthModal"; export { default as CursorAuthModal } from "./CursorAuthModal"; +export { default as XiaomiMimoAuthModal } from "./XiaomiMimoAuthModal"; export { default as IFlowCookieModal } from "./IFlowCookieModal"; export { default as GitLabAuthModal } from "./GitLabAuthModal"; export { default as EditConnectionModal } from "./EditConnectionModal"; diff --git a/tests/unit/xiaomi-mimo-executor.test.js b/tests/unit/xiaomi-mimo-executor.test.js new file mode 100644 index 00000000..10c6a520 --- /dev/null +++ b/tests/unit/xiaomi-mimo-executor.test.js @@ -0,0 +1,80 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; +import { XiaomiMimoExecutor, __test__ } from "../../open-sse/executors/xiaomi-mimo.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; + +const { bareModel, COOKIE_KEY } = __test__; + +const OPENAI_T = { runtimeTransport: { format: "openai", baseUrl: "https://api.xiaomimimo.com/v1/chat/completions" } }; +const CLAUDE_T = { runtimeTransport: { format: "claude", baseUrl: "https://api.xiaomimimo.com/anthropic/v1/messages" } }; + +describe("xiaomi-mimo executor", () => { + let ex; + beforeEach(() => { + ex = new XiaomiMimoExecutor(); + }); + + it("is registered for xiaomi-mimo", () => { + expect(getExecutor("xiaomi-mimo")).toBeInstanceOf(XiaomiMimoExecutor); + }); + + it("routes Preview models to the account-service route regardless of transport", () => { + const expected = "https://mimo-server-cn.xiaomimimo.com/api/route/chat/completions"; + expect(ex.buildUrl("mimo-x-pro-preview", true, 0, OPENAI_T)).toBe(expected); + expect(ex.buildUrl("mimo-x-pro-preview", true, 0, CLAUDE_T)).toBe(expected); + // body.model arrives as `xiaomi/` via upstreamModelId + expect(ex.buildUrl("xiaomi/mimo-x-flash-preview", true, 0, OPENAI_T)).toBe(expected); + }); + + it("keeps the sourceFormat-matched endpoint for cloud models", () => { + // Regression: a Claude client must reach /anthropic/v1/messages, not /v1/chat/completions. + expect(ex.buildUrl("mimo-v2.5-pro", true, 0, CLAUDE_T)).toBe(CLAUDE_T.runtimeTransport.baseUrl); + expect(ex.buildUrl("mimo-v2.5-pro", true, 0, OPENAI_T)).toBe(OPENAI_T.runtimeTransport.baseUrl); + }); + + it("authenticates Preview calls with the account cookie", () => { + const headers = ex.buildHeaders({ [COOKIE_KEY]: "serviceToken=abc", accessToken: "sk-x" }, true, "u", "mimo-x-pro-preview"); + expect(headers.Cookie).toBe("serviceToken=abc"); + expect(headers.Authorization).toBeUndefined(); + }); + + it("authenticates cloud calls with the bearer key", () => { + const headers = ex.buildHeaders({ accessToken: "sk-x" }, true, "u", "mimo-v2.5-pro"); + expect(headers.Authorization).toBe("Bearer sk-x"); + expect(headers.Cookie).toBeUndefined(); + }); + + it("fails fast when a Preview call has no account session", async () => { + await expect( + ex.execute({ model: "mimo-x-pro-preview", body: {}, stream: true, credentials: {}, log: null }), + ).rejects.toThrow(/account session unavailable/); + }); + + it("flattens content-part arrays to plain strings", () => { + const out = ex.transformRequest( + "mimo-x-pro-preview", + { messages: [{ role: "user", content: [{ type: "text", text: "a" }, { type: "text", text: "b" }] }] }, + true, + {}, + ); + expect(out.messages[0].content).toBe("ab"); + }); + + it("applies Preview defaults without overriding explicit values", () => { + const body = { messages: [{ role: "user", content: "hi" }], temperature: 0.2 }; + const out = ex.transformRequest("mimo-x-pro-preview", body, true, {}); + expect(out.temperature).toBe(0.2); // caller's value kept + expect(out.top_p).toBe(0.95); // default filled in + expect(out.max_tokens).toBe(4096); + }); + + it("leaves cloud bodies free of Preview defaults", () => { + const out = ex.transformRequest("mimo-v2.5-pro", { messages: [{ role: "user", content: "hi" }] }, true, {}); + expect(out.thinking).toBeUndefined(); + expect(out.max_tokens).toBeUndefined(); + }); + + it("strips a provider/model prefix when testing preview ids", () => { + expect(bareModel("xiaomi/mimo-x-pro-preview")).toBe("mimo-x-pro-preview"); + expect(bareModel("mimo-x-pro-preview")).toBe("mimo-x-pro-preview"); + }); +}); diff --git a/tests/unit/xiaomi-mimo-oauth-proxy.test.js b/tests/unit/xiaomi-mimo-oauth-proxy.test.js new file mode 100644 index 00000000..c74573ff --- /dev/null +++ b/tests/unit/xiaomi-mimo-oauth-proxy.test.js @@ -0,0 +1,54 @@ +/** + * Regression: the xiaomi-mimo OAuth session store must not retain sessions + * once the callback listener is down. + * + * Each /authorize registers a session holding an X25519 private key, keyed by a + * fresh state. Unlike trae/windsurf/zed (singleton session) this is a Map, so + * without an explicit clear every login attempt would leak a private key for + * the whole process lifetime. + */ +import { describe, it, expect } from "vitest"; +import { + registerXiaomiMimoSession, + getXiaomiMimoSessionStatus, + clearXiaomiMimoSession, + stopXiaomiMimoProxy, +} from "../../src/lib/oauth/utils/server.js"; + +const KEY = Buffer.from("x25519-private-key-material"); + +describe("xiaomi-mimo OAuth session store", () => { + it("drops pending sessions when the proxy stops", () => { + registerXiaomiMimoSession({ state: "s1", privateKeyDer: KEY }); + expect(getXiaomiMimoSessionStatus("s1")).not.toBeNull(); + + stopXiaomiMimoProxy(); + + expect(getXiaomiMimoSessionStatus("s1")).toBeNull(); + }); + + it("drops every session, not just the last one", () => { + registerXiaomiMimoSession({ state: "a", privateKeyDer: KEY }); + registerXiaomiMimoSession({ state: "b", privateKeyDer: KEY }); + registerXiaomiMimoSession({ state: "c", privateKeyDer: KEY }); + + stopXiaomiMimoProxy(); + + for (const s of ["a", "b", "c"]) { + expect(getXiaomiMimoSessionStatus(s)).toBeNull(); + } + }); + + it("ignores registrations with a missing state or key", () => { + expect(registerXiaomiMimoSession({ state: "", privateKeyDer: KEY })).toBe(false); + expect(registerXiaomiMimoSession({ state: "s", privateKeyDer: null })).toBe(false); + }); + + it("never exposes the private key to callers", () => { + registerXiaomiMimoSession({ state: "s1", privateKeyDer: KEY }); + const view = getXiaomiMimoSessionStatus("s1"); + expect(view).toEqual({ status: "pending", result: null, error: null }); + expect(JSON.stringify(view)).not.toContain("privateKeyDer"); + clearXiaomiMimoSession("s1"); + }); +}); diff --git a/tests/unit/xiaomi-mimo-oauth-session.test.js b/tests/unit/xiaomi-mimo-oauth-session.test.js new file mode 100644 index 00000000..3aee1cc2 --- /dev/null +++ b/tests/unit/xiaomi-mimo-oauth-session.test.js @@ -0,0 +1,152 @@ +/** + * Regression: the poll-status/exchange session lifecycle for xiaomi-mimo. + * + * The original PR cleared the session inside poll-status, so the client's + * following POST /exchange always saw a missing session and returned 400 — + * the whole browser-OAuth fallback was dead. These tests pin the contract: + * - a finished session survives /poll-status until /exchange consumes it + * - a failed session is cleaned up by /poll-status itself + */ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("next/server", () => ({ + NextResponse: { + json: (body, init) => ({ + status: init?.status || 200, + body, + json: async () => body, + }), + }, +})); + +vi.mock("@/lib/oauth/providers", () => ({ + getProvider: vi.fn(), + generateAuthData: vi.fn(), + exchangeTokens: vi.fn(), + requestDeviceCode: vi.fn(), + pollForToken: vi.fn(), +})); + +vi.mock("@/models", () => ({ + createProviderConnection: vi.fn(async (d) => ({ id: "conn-1", ...d })), +})); + +vi.mock("open-sse/shared/mimoAccount.js", () => ({ + readDesktopPassToken: vi.fn(async () => ({ passToken: "pt-abc", userId: "u1", cUserId: "c1" })), +})); + +vi.mock("@/lib/oauth/utils/ideDetect", () => ({ detectIdeInstalled: vi.fn() })); + +// Session store backing the mocked OAuth server helpers, so the test can assert +// on real lifecycle transitions rather than on call counts alone. +const sessions = new Map(); +const stopped = { count: 0 }; + +vi.mock("@/lib/oauth/utils/server", () => { + const notUsed = () => { throw new Error("unexpected helper"); }; + const noop = () => {}; + return { + startCodexProxy: notUsed, stopCodexProxy: noop, registerCodexSession: noop, + getCodexSessionStatus: () => null, clearCodexSession: noop, + startXaiProxy: notUsed, stopXaiProxy: noop, registerXaiSession: noop, + getXaiSessionStatus: () => null, clearXaiSession: noop, + startTraeProxy: notUsed, stopTraeProxy: noop, registerTraeSession: noop, + getTraeSessionStatus: () => null, clearTraeSession: noop, + startWindsurfProxy: notUsed, stopWindsurfProxy: noop, registerWindsurfSession: noop, + getWindsurfSessionStatus: () => null, clearWindsurfSession: noop, + startZedProxy: notUsed, stopZedProxy: noop, registerZedSession: noop, + getZedSessionStatus: () => null, clearZedSession: noop, + startXiaomiMimoProxy: notUsed, + stopXiaomiMimoProxy: () => { stopped.count += 1; }, + registerXiaomiMimoSession: () => {}, + getXiaomiMimoSessionStatus: (state) => { + const s = sessions.get(state); + return s ? { status: s.status, result: s.result || null, error: s.error || null } : null; + }, + clearXiaomiMimoSession: (state) => { sessions.delete(state); }, + }; +}); + +const { GET, POST } = await import("../../src/app/api/oauth/[provider]/[action]/route.js"); + +const get = (action, state) => + GET(new Request(`http://localhost/api/oauth/xiaomi-mimo/${action}?state=${state}`), { + params: Promise.resolve({ provider: "xiaomi-mimo", action }), + }); + +const exchange = (state) => + POST( + new Request("http://localhost/api/oauth/xiaomi-mimo/exchange", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ state }), + }), + { params: Promise.resolve({ provider: "xiaomi-mimo", action: "exchange" }) }, + ); + +describe("xiaomi-mimo OAuth session lifecycle", () => { + beforeEach(() => { + sessions.clear(); + stopped.count = 0; + }); + + it("keeps a finished session alive so /exchange can consume it", async () => { + sessions.set("st1", { status: "done", result: { uid: "u1", accessToken: "sk-x", baseUrl: "https://api.xiaomimimo.com/v1" } }); + + const poll = await get("poll-status", "st1"); + expect(poll.status).toBe(200); + expect(await poll.json()).toMatchObject({ status: "done" }); + + // The bug: this used to be gone, making /exchange always 400. + expect(sessions.has("st1")).toBe(true); + + const res = await exchange("st1"); + expect(res.status).toBe(200); + expect((await res.json()).success).toBe(true); + }); + + it("clears the session once /exchange consumed it", async () => { + sessions.set("st1", { status: "done", result: { uid: "u1", accessToken: "sk-x" } }); + await exchange("st1"); + expect(sessions.has("st1")).toBe(false); + }); + + it("cleans up a failed session in poll-status and stops the proxy", async () => { + sessions.set("st2", { status: "error", error: "Could not decrypt with any pending session key" }); + + const poll = await get("poll-status", "st2"); + expect(await poll.json()).toMatchObject({ status: "error" }); + + expect(sessions.has("st2")).toBe(false); + expect(stopped.count).toBe(1); + }); + + it("persists the Desktop passToken onto the connection (Preview models need it)", async () => { + const { createProviderConnection } = await import("@/models"); + sessions.set("st3", { status: "done", result: { uid: "u1", accessToken: "sk-x" } }); + + await exchange("st3"); + + const arg = createProviderConnection.mock.calls.at(-1)[0]; + expect(arg.provider).toBe("xiaomi-mimo"); + expect(arg.providerSpecificData.mimoPassToken).toBe("pt-abc"); + expect(arg.providerSpecificData.mimoUserId).toBe("u1"); + }); + + it("still reports unknown for an unregistered state", async () => { + const poll = await get("poll-status", "nope"); + expect(await poll.json()).toEqual({ status: "unknown" }); + }); + + it("rejects /exchange without a state", async () => { + const res = await POST( + new Request("http://localhost/api/oauth/xiaomi-mimo/exchange", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({}), + }), + { params: Promise.resolve({ provider: "xiaomi-mimo", action: "exchange" }) }, + ); + expect(res.status).toBe(400); + }); +}); From 17c4cc76877bd1755030a8414f8d0083f48dcccf Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 11 Sep 2026 00:09:15 +0700 Subject: [PATCH 47/78] feat(claude-code): drive auto-compact window, add a 1M-context toggle MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The "Context window" dropdown wrote CLAUDE_CODE_MAX_CONTEXT_TOKENS, which Claude Code ignores for any model it recognizes: its window resolver returns the env value only when the id is unknown to the model table, so every claude-* mapping kept the built-in 200K and the dropdown did nothing. It was never the compaction threshold either. - Replace it with CLAUDE_CODE_AUTO_COMPACT_WINDOW — the documented trigger (100K–1M, clamped to the model window, env beats the autoCompactWindow setting) — and relabel the field Auto-compact. The 1M preset becomes 700K, which no longer collides with the marker it depends on. - Add a "1M context" checkbox that appends the `[1m]` marker to the ANTHROPIC_DEFAULT_*_MODEL envs. Claude Code assumes 200K unless the name carries the marker — the resolver is a plain /\[1m\]/i test on the string, so it applies to any id and no model lookup is involved; the user decides which models are worth declaring as 1M. - Toggling rewrites the model inputs immediately, and Apply writes them verbatim, so a marker typed by hand is not stripped. Rename maxContextTokens -> autoCompactWindow through the POST body and RESET_ENV_KEYS so a reset clears the key actually written. Co-Authored-By: Claude Code --- .../cli-tools/components/ClaudeToolCard.js | 81 +++++++++++++++---- .../api/cli-tools/claude-settings/route.js | 15 ++-- 2 files changed, 73 insertions(+), 23 deletions(-) diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js b/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js index fad05f9e..f49d8a38 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js @@ -7,19 +7,25 @@ import BaseUrlSelect from "./BaseUrlSelect"; import { rememberEndpoint } from "./cliEndpointPresets"; import ApiKeySelect from "./ApiKeySelect"; import { matchKnownEndpoint } from "./cliEndpointMatch"; +import { stripModelContextMarker } from "open-sse/utils/modelMarkers.js"; const CLOUD_URL = process.env.NEXT_PUBLIC_CLOUD_URL; -// Context window presets. UI shows the round number; the value written is nudged -// down 2K to stay safely under the upstream hard cap. +// Auto-compact window presets (CLAUDE_CODE_AUTO_COMPACT_WINDOW, valid 100K–1M). +// UI shows the round number; the value written is nudged down 2K to stay safely +// under the upstream hard cap. const CONTEXT_OPTIONS = [ { label: "Default", value: "" }, { label: "200K", value: "198000" }, { label: "300K", value: "298000" }, { label: "500K", value: "498000" }, - { label: "1M", value: "998000" }, + { label: "700K", value: "698000" }, ]; +// Claude Code assumes a model's window is 200K unless the name carries the `[1m]` +// marker, which is why the 1M auto-compact preset only takes effect once the +// marker is applied. + export default function ClaudeToolCard({ tool, isExpanded, @@ -51,9 +57,28 @@ export default function ClaudeToolCard({ const [customBaseUrl, setCustomBaseUrl] = useState(""); const [ccFilterNaming, setCcFilterNaming] = useState(false); const [exaMcpEnabled, setExaMcpEnabled] = useState(false); - const [maxContextTokens, setMaxContextTokens] = useState(""); + const [autoCompactWindow, setAutoCompactWindow] = useState(""); + const [oneMContext, setOneMContext] = useState(false); const hasInitializedModels = useRef(false); + // Claude Code only string-matches the marker against the model name, so it + // applies to any id — the user decides which models are worth declaring as 1M. + // Stripping first keeps repeated toggles from stacking `[1m][1m]`. + const withContextMarker = (value, enabled) => { + const { model } = stripModelContextMarker(value); + return enabled ? `${model}[1m]` : model; + }; + + // Rewrite the mappings in place on toggle, so the inputs show what will be + // written without waiting for Apply. + const handleOneMContextToggle = (enabled) => { + setOneMContext(enabled); + tool.defaultModels.forEach((model) => { + const current = modelMappings[model.alias]; + if (current) onModelMappingChange(model.alias, withContextMarker(current, enabled)); + }); + }; + const currentBaseUrl = claudeStatus?.settings?.env?.ANTHROPIC_BASE_URL || ""; const getConfigStatus = () => { @@ -80,9 +105,15 @@ export default function ClaudeToolCard({ }, [initialStatus]); useEffect(() => { - const v = claudeStatus?.settings?.env?.CLAUDE_CODE_MAX_CONTEXT_TOKENS; - setMaxContextTokens(v || ""); - }, [claudeStatus?.settings?.env?.CLAUDE_CODE_MAX_CONTEXT_TOKENS]); + const v = claudeStatus?.settings?.env?.CLAUDE_CODE_AUTO_COMPACT_WINDOW; + setAutoCompactWindow(v || ""); + }, [claudeStatus?.settings?.env?.CLAUDE_CODE_AUTO_COMPACT_WINDOW]); + + useEffect(() => { + const env = claudeStatus?.settings?.env; + if (!env) return; + setOneMContext(tool.defaultModels.some((model) => env[model.envKey]?.endsWith("[1m]"))); + }, [claudeStatus?.settings?.env, tool.defaultModels]); useEffect(() => { if (isExpanded) { @@ -124,6 +155,8 @@ export default function ClaudeToolCard({ tool.defaultModels.forEach((model) => { if (model.envKey) { + // Kept verbatim (marker included) so the input matches what is on disk; + // withContextMarker strips before appending, so re-applying cannot double it. const value = env[model.envKey] || model.defaultValue || ""; // Only sync initial values from file once if (value) { @@ -180,15 +213,17 @@ export default function ClaudeToolCard({ tool.defaultModels.forEach((model) => { const targetModel = modelMappings[model.alias]; + // Written verbatim — the input may hold a marker typed by hand, and the + // toggle already decided the marker when it was flipped. if (targetModel && model.envKey) env[model.envKey] = targetModel; }); - if (maxContextTokens) { - env.CLAUDE_CODE_MAX_CONTEXT_TOKENS = maxContextTokens; + if (autoCompactWindow) { + env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = autoCompactWindow; } const res = await fetch("/api/cli-tools/claude-settings", { method: "POST", headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ env, exaMcpEnabled, maxContextTokens }), + body: JSON.stringify({ env, exaMcpEnabled, autoCompactWindow }), }); const data = await res.json(); if (res.ok) { @@ -217,7 +252,8 @@ export default function ClaudeToolCard({ tool.defaultModels.forEach((model) => onModelMappingChange(model.alias, model.defaultValue || "")); setSelectedApiKey(""); setExaMcpEnabled(false); - setMaxContextTokens(""); + setAutoCompactWindow(""); + setOneMContext(false); } else { setMessage({ type: "error", text: data.error || "Failed to reset settings" }); } @@ -247,8 +283,8 @@ export default function ClaudeToolCard({ const targetModel = modelMappings[model.alias]; if (targetModel && model.envKey) env[model.envKey] = targetModel; }); - if (maxContextTokens) { - env.CLAUDE_CODE_MAX_CONTEXT_TOKENS = maxContextTokens; + if (autoCompactWindow) { + env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = autoCompactWindow; } return [ @@ -374,17 +410,30 @@ export default function ClaudeToolCard({ ))} - {/* Context Window */} + {/* Auto-compact window */}
- Context window + Auto-compact arrow_forward - setAutoCompactWindow(e.target.value)} className="w-full min-w-0 px-2 py-2 bg-surface rounded border border-border text-xs focus:outline-none focus:ring-1 focus:ring-primary/50 sm:py-1.5"> {CONTEXT_OPTIONS.map((opt) => ( ))}
+ {/* 1M context */} +
+ 1M context + arrow_forward + +
+ {/* CC Filter Naming */}
Filter naming diff --git a/src/app/api/cli-tools/claude-settings/route.js b/src/app/api/cli-tools/claude-settings/route.js index 76ba9232..c7087577 100644 --- a/src/app/api/cli-tools/claude-settings/route.js +++ b/src/app/api/cli-tools/claude-settings/route.js @@ -123,7 +123,7 @@ export async function GET() { // POST - Backup old fields and write new settings export async function POST(request) { try { - const { env, exaMcpEnabled, maxContextTokens } = await request.json(); + const { env, exaMcpEnabled, autoCompactWindow } = await request.json(); if (!env || typeof env !== "object") { return NextResponse.json( @@ -166,12 +166,13 @@ export async function POST(request) { }, }; - // CLAUDE_CODE_MAX_CONTEXT_TOKENS — only set when a concrete value is chosen; - // "Default" removes the key so Claude Code falls back to the model's window. - if (maxContextTokens) { - newSettings.env.CLAUDE_CODE_MAX_CONTEXT_TOKENS = String(maxContextTokens); + // CLAUDE_CODE_AUTO_COMPACT_WINDOW — the token threshold that triggers + // auto-compact. Only set when a concrete value is chosen; "Default" removes + // the key so Claude Code derives the window from the model. + if (autoCompactWindow) { + newSettings.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = String(autoCompactWindow); } else { - delete newSettings.env.CLAUDE_CODE_MAX_CONTEXT_TOKENS; + delete newSettings.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW; } // Write new settings @@ -203,7 +204,7 @@ const RESET_ENV_KEYS = [ "ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL", "API_TIMEOUT_MS", - "CLAUDE_CODE_MAX_CONTEXT_TOKENS", + "CLAUDE_CODE_AUTO_COMPACT_WINDOW", ]; // DELETE - Reset settings (remove env fields) From 5c399b64062f09f359668bfc036c2aeedf1e4bb7 Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 11 Sep 2026 22:03:48 +0700 Subject: [PATCH 48/78] feat(codebuddy-intl,ollama): add DeepSeek-V4.1-Flash codebuddy-intl: deepseek-v4-flash replaced by deepseek-v4.1-flash (same gateway catalog as CN) and a capability override so the model keeps the openai-style reasoning_effort path instead of the vendor-native "deepseek" thinking shape the gateway rejects. Thinking levels low/high/xhigh. ollama: add deepseek-v4.1-flash:cloud (verified on ollama.com/api/tags) with vision + 1M context caps. Co-Authored-By: Claude Code --- open-sse/providers/capabilities.js | 17 +++++++++++++++++ open-sse/providers/registry/codebuddy-intl.js | 4 +++- open-sse/providers/registry/ollama.js | 1 + open-sse/providers/thinkingLevels.js | 2 ++ 4 files changed, 23 insertions(+), 1 deletion(-) diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index e011b788..8ad55314 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -214,6 +214,13 @@ export const PROVIDER_CAPABILITIES = { // contract). maxOutput 128000 per the server's product-config payload. "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 }, }, + // CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors + // the codebuddy-cn entry (the openai-style reasoning_effort format matters: + // the generic *deepseek-v4* pattern would otherwise pick the vendor-native + // "deepseek" thinking shape, which the CodeBuddy gateway does not accept). + "codebuddy-intl": { + "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 }, + }, // Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the // registry `name` is display-only and capability lookup matches on the raw // id, so every qoder model would fall through to DEFAULT_CAPABILITIES @@ -251,6 +258,16 @@ export const PROVIDER_CAPABILITIES = { "laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 }, "laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 }, }, + // Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge + // the library page publishes for this model (text+image in, 1M context). + // ponytail: thinkingFormat stays "deepseek" to preserve today's body shape; + // Ollama's native toggle is the top-level `think` field (bool or + // low/medium/high/max), which no format in thinkingUnified.js emits yet — + // openai-to-ollama.js drops it. Wire a "think" format when thinking on + // Ollama Cloud is actually needed. + "ollama": { + "deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + }, }; /** diff --git a/open-sse/providers/registry/codebuddy-intl.js b/open-sse/providers/registry/codebuddy-intl.js index eab1ce93..65e787d1 100644 --- a/open-sse/providers/registry/codebuddy-intl.js +++ b/open-sse/providers/registry/codebuddy-intl.js @@ -58,7 +58,9 @@ export default { { id: "kimi-k2.5", name: "Kimi-K2.5" }, { id: "hy3-preview", name: "Hy3 Preview" }, { id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" }, - { id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" }, + // deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same + // catalog as CN; the old endpoint still answers 200 but the list is the contract). + { id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" }, { id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" }, ], oauth: { diff --git a/open-sse/providers/registry/ollama.js b/open-sse/providers/registry/ollama.js index 6965484a..605c4b89 100644 --- a/open-sse/providers/registry/ollama.js +++ b/open-sse/providers/registry/ollama.js @@ -30,6 +30,7 @@ export default { { id: "glm-4.7-flash", name: "GLM 4.7 Flash" }, { id: "qwen3.5", name: "Qwen3.5" }, { id: "minimax-m3", name: "MiniMax M3" }, + { id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" }, ], serviceKinds: ["llm", "webFetch"], fetchConfig: { diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index 7cc1b28f..404fa75e 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -52,6 +52,8 @@ const PATTERN_THINKING = [ { provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] }, { provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] }, { provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] }, + // codebuddy-intl rides the same gateway catalog, so its deepseek levels match. + { provider: "codebuddy-intl", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] }, ]; // Returns valid thinking levels for a model, or null when the model has no reasoning. From 9300121366c8baf41df57ff588d81703400a234b Mon Sep 17 00:00:00 2001 From: decolua Date: Wed, 16 Sep 2026 20:04:10 +0700 Subject: [PATCH 49/78] fix(stream): report aborts after HTTP 200 in-band instead of closing silently MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A stream that stalled or lost its upstream was closed with no terminal frame at all, so clients saw "200 OK, a few chunks, then nothing" and could not tell a truncated reply from a finished one. The Responses passthrough path already synthesized response.failed; every other client format got nothing. The watchdog now hands its reason ("stream stall timeout" or "upstream connection lost") to onAbortTerminal, and buildStreamErrorBytes frames it per client format: OpenAI-compatible clients get data: {"error":{...}} followed by data: [DONE], Anthropic clients get `event: error`. The error frame always precedes [DONE] (openai-python raises APIError on any data payload carrying an error key), and no synthetic finish_reason is ever emitted — a truncated stream must not look like a clean stop. Co-Authored-By: Claude Code --- .../handlers/chatCore/streamingHandler.js | 12 +++- open-sse/utils/streamHandler.js | 7 +- open-sse/utils/streamHelpers.js | 25 +++++++ tests/unit/responses-abort-terminal.test.js | 72 ++++++++++++++++++- 4 files changed, 111 insertions(+), 5 deletions(-) diff --git a/open-sse/handlers/chatCore/streamingHandler.js b/open-sse/handlers/chatCore/streamingHandler.js index 7b7634e2..6751256b 100644 --- a/open-sse/handlers/chatCore/streamingHandler.js +++ b/open-sse/handlers/chatCore/streamingHandler.js @@ -3,8 +3,9 @@ import { needsTranslation } from "../../translator/index.js"; import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js"; import { pipeWithDisconnect } from "../../utils/streamHandler.js"; import { PROVIDERS } from "../../config/providers.js"; -import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js"; +import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js"; import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js"; +import { buildStreamErrorBytes } from "../../utils/streamHelpers.js"; import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { saveRequestDetail } from "@/lib/usageDb.js"; import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js"; @@ -81,9 +82,14 @@ export async function handleStreamingResponse({ providerResponse, provider, mode const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }); - // Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event + // Terminal bytes when the stream aborts after HTTP 200 was already sent, so the + // client sees a real error instead of a silently truncated stream. + // Responses passthrough keeps its own response.failed shape; every other client + // format gets the OpenAI error frame + [DONE], or `event: error` for Claude. const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES; - const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null; + const onAbortTerminal = isResponsesPassthrough + ? buildAbortedResponsesTerminalBytes + : (message) => buildStreamErrorBytes(HTTP_STATUS.GATEWAY_TIMEOUT, message, sourceFormat); const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS; const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs); diff --git a/open-sse/utils/streamHandler.js b/open-sse/utils/streamHandler.js index 27a623fc..fce01706 100644 --- a/open-sse/utils/streamHandler.js +++ b/open-sse/utils/streamHandler.js @@ -95,6 +95,9 @@ export function createStreamController({ onDisconnect, onError, log, provider, m * activity), not here — output of the transform stream may be silent * for long periods while raw bytes still flow (e.g. Kiro EventStream * binary frames buffering, Claude reasoning streams). + * + * @param {function} [onAbortTerminal] - Receives a human-readable abort + * message and returns terminal SSE bytes to emit downstream. */ export function createDisconnectAwareStream(transformStream, streamController, onAbortTerminal = null) { const reader = transformStream.readable.getReader(); @@ -194,6 +197,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont let chunkCount = 0; let totalBytes = 0; let lastChunkAt = Date.now(); + let abortMessage = "upstream connection lost"; const t0 = Date.now(); const tag = "STREAM"; const clearStall = () => { @@ -203,6 +207,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont clearStall(); stallTimer = setTimeout(() => { stallTimer = null; + abortMessage = "stream stall timeout"; dbg(tag, `STALL TIMEOUT ${stallTimeoutMs}ms | chunks=${chunkCount} | bytes=${totalBytes} | sinceLast=${Date.now() - lastChunkAt}ms`); streamController.handleError?.(new Error("stream stall timeout")); streamController.abort?.(); @@ -249,7 +254,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont return createDisconnectAwareStream( { readable: transformedBody, writable: { getWriter: () => ({ abort: () => Promise.resolve() }) } }, wrappedController, - onAbortTerminal + onAbortTerminal ? () => onAbortTerminal(abortMessage) : null ); } diff --git a/open-sse/utils/streamHelpers.js b/open-sse/utils/streamHelpers.js index a7a19180..ee4e720a 100644 --- a/open-sse/utils/streamHelpers.js +++ b/open-sse/utils/streamHelpers.js @@ -1,4 +1,8 @@ import { FORMATS } from "../translator/formats.js"; +import { buildErrorBody } from "./error.js"; +import { SSE_DONE } from "./sseConstants.js"; + +const sharedEncoder = new TextEncoder(); // Parse SSE data line export function parseSSELine(line, format = null) { @@ -120,3 +124,24 @@ export function formatSSE(data, sourceFormat) { return `data: ${JSON.stringify(data)}\n\n`; } + +// Terminal frames for a stream that aborted after HTTP 200 was already sent, so +// the status code can no longer change. OpenAI-compatible clients (openai-python +// raises APIError on any `data:` payload carrying an `error` key, checked before +// [DONE]) need the error frame first, then [DONE]; Anthropic clients need +// `event: error`. Never fabricate a successful finish_reason instead. +// +// Returns encoded bytes: onAbortTerminal callbacks are enqueued verbatim, same +// as buildAbortedResponsesTerminalBytes. +// +// NOTE: non-SSE client formats (Ollama NDJSON) get an SSE frame here — dead in +// practice because detectFormatByEndpoint never resolves to OLLAMA. +export function buildStreamErrorBytes(statusCode, message, clientFormat) { + const { error } = buildErrorBody(statusCode, message); + + const sse = clientFormat === FORMATS.CLAUDE + ? formatSSE({ type: "error", error }, FORMATS.CLAUDE) + : formatSSE({ error }, clientFormat) + SSE_DONE; + + return sharedEncoder.encode(sse); +} diff --git a/tests/unit/responses-abort-terminal.test.js b/tests/unit/responses-abort-terminal.test.js index 10af5848..e05c9d14 100644 --- a/tests/unit/responses-abort-terminal.test.js +++ b/tests/unit/responses-abort-terminal.test.js @@ -1,7 +1,9 @@ import { describe, expect, it } from "vitest"; -import { createDisconnectAwareStream } from "../../open-sse/utils/streamHandler.js"; +import { createDisconnectAwareStream, pipeWithDisconnect, createStreamController } from "../../open-sse/utils/streamHandler.js"; import { buildAbortedResponsesTerminalBytes } from "../../open-sse/utils/responsesStreamHelpers.js"; +import { buildStreamErrorBytes } from "../../open-sse/utils/streamHelpers.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; // Minimal stream controller stub function makeController() { @@ -70,3 +72,71 @@ describe("Responses abort terminal synthesis", () => { expect(text).not.toContain("[DONE]"); }); }); + +// A stream that aborts after HTTP 200 cannot change status, so the failure must +// travel in-band: structured error frame first, then [DONE]. openai-python raises +// APIError on any `data:` payload carrying an `error` key (checked before [DONE]); +// Anthropic clients need `event: error`. Never a fabricated finish_reason. +describe("buildStreamErrorBytes", () => { + const jsonOf = (sse) => JSON.parse(sse.match(/\{.*\}/s)[0]); + const textOf = (bytes) => new TextDecoder().decode(bytes); + + // onAbortTerminal callbacks are enqueued verbatim, so a string here is a + // silent no-op at runtime (createDisconnectAwareStream swallows the throw). + it("returns bytes, not a string", () => { + expect(buildStreamErrorBytes(504, "x", FORMATS.OPENAI)).toBeInstanceOf(Uint8Array); + }); + + it("emits error frame then [DONE] for OpenAI clients", () => { + const out = textOf(buildStreamErrorBytes(504, "stream stall timeout", FORMATS.OPENAI)); + + expect(out).toContain('data: {"error"'); + expect(out.indexOf("data: [DONE]")).toBeGreaterThan(out.indexOf('data: {"error"')); + + expect(jsonOf(out).error).toEqual({ + message: "stream stall timeout", + type: "server_error", + code: "gateway_timeout", + }); + }); + + it("emits event: error (no [DONE]) for Claude clients", () => { + const out = textOf(buildStreamErrorBytes(504, "stream stall timeout", FORMATS.CLAUDE)); + + expect(out).toContain("event: error\n"); + expect(out).not.toContain("[DONE]"); + expect(jsonOf(out)).toMatchObject({ type: "error", error: { message: "stream stall timeout" } }); + }); +}); + +// The wiring, not just the frame builder: the watchdog must hand its reason to +// onAbortTerminal and the bytes must reach a real consumer. +describe("stall abort through pipeWithDisconnect", () => { + it("delivers the error frame and closes the stream", async () => { + // Real controller: the stub above never fires its signal, and the abort + // must reach the upstream body for the pipe to end. + const ctrl = createStreamController({ provider: "ollama", model: "test" }); + + // Emits one chunk then goes silent; errors on abort like a real fetch body. + const upstream = new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode("data: hi\n\n")); + ctrl.signal.addEventListener("abort", () => controller.error(new Error("aborted")), { once: true }); + }, + }); + + let seen = null; + const out = pipeWithDisconnect( + { body: upstream }, + new TransformStream(), + ctrl, + (message) => { seen = message; return buildStreamErrorBytes(504, message, FORMATS.OPENAI); }, + 50 + ); + + const text = await readAll(out); + expect(seen).toBe("stream stall timeout"); + expect(text).toContain('"stream stall timeout"'); + expect(text).toContain("data: [DONE]"); + }); +}); From 13b468b88996155d32cc3308f077352227896fb2 Mon Sep 17 00:00:00 2001 From: KhuatHieu Date: Wed, 16 Sep 2026 20:16:32 +0700 Subject: [PATCH 50/78] fix(commandcode): preserve images and reasoning_effort on /alpha/generate Command Code dropped vision and ignored client effort through the router: image blocks became "[image omitted]", HTTP image URLs were never inlined, and reasoning_effort landed on the envelope wrapper instead of params (so the DeepSeek family mapping remapped low -> high). The catalog also treated deepseek/deepseek-v4.1-flash as text-only, so the vision adapter stole those requests to another provider. - Map OpenAI image_url / Claude image blocks (base64 or data-URI) to the native {type:"image", image:"data:...;base64,...", mimeType} generate block. - Add FORMATS.COMMANDCODE to TARGETS_NEED_BASE64 so remote http(s) images are inlined by the existing SSRF-safe fetcher before translation. - Write reasoning_effort inside params for targetFormat commandcode and pass low|medium|high|xhigh|max through unmapped; allow it in thinkingLevels. - Provider-scoped capabilities for commandcode/cmc: vision except the CLI text-only denylist, thinkingFormat commandcode, so family patterns (deepseek-v4 -> thinkingFormat deepseek, vision false) no longer win. - Quota Tracker: whoami + billing credits/subscriptions (credits vs plan cap, 5h and weekly windows), labels from AI_PROVIDERS[].name. --- open-sse/providers/capabilities.js | 54 +++++++ open-sse/providers/registry/commandcode.js | 4 + open-sse/providers/thinkingLevels.js | 1 + open-sse/services/usage.js | 2 + open-sse/services/usage/commandcode.js | 134 +++++++++++++++++ open-sse/translator/concerns/prefetch.js | 1 + .../translator/concerns/thinkingUnified.js | 17 +++ .../request/openai-to-commandcode.js | 51 ++++++- .../usage/components/ProviderLimits/index.js | 18 ++- .../bugs-gemini-cursor-commandcode.test.js | 11 +- tests/translator/thinking-unified.test.js | 23 +++ tests/unit/capabilities.test.js | 22 +++ tests/unit/commandcode-usage.test.js | 135 ++++++++++++++++++ tests/unit/openai-to-commandcode.test.js | 54 +++++++ tests/unit/prefetch-images.test.js | 17 +++ tests/unit/usage-dispatch.test.js | 2 +- 16 files changed, 529 insertions(+), 17 deletions(-) create mode 100644 open-sse/services/usage/commandcode.js create mode 100644 tests/unit/commandcode-usage.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 8ad55314..177db4a7 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -467,12 +467,66 @@ function refine(base, provider, model) { return result; } +// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models +// default to vision; only this denylist stays text-only. +const COMMANDCODE_TEXT_ONLY = new Set([ + "deepseek/deepseek-v4-pro", + "deepseek/deepseek-v4-flash", + "deepseek/deepseek-v4-flash-fast", + "zai-org/glm-5.3", + "zai-org/glm-5.2", + "zai-org/glm-5.2-fast", + "zai-org/glm-5.1", + "zai-org/glm-5", + "minimaxai/minimax-m2.7", + "minimax/minimax-m2.7-free", + "minimaxai/minimax-m2.5", + "xiaomi/mimo-v2.5-pro", + "qwen/qwen3.6-max-preview", + "qwen/qwen3.7-max", + "meituan/longcat-2.0:free", + "stepfun/step-3.5-flash", + "tencent/hy4-preview", + "tencent/hy3", + "tencent/hy3-paid", + "nvidia/nemotron-3-ultra-550b-a55b", + "poolside/laguna-s-2.1-free", + "inclusionai/ling-3.0-flash-free", + "inclusionai/ling-3.0-flash-sante:free", +]); + +function isCommandCodeTextOnly(model) { + const key = String(model || "").toLowerCase(); + if (COMMANDCODE_TEXT_ONLY.has(key)) return true; + for (const id of COMMANDCODE_TEXT_ONLY) { + const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id; + if (key === base || key.endsWith("/" + base)) return true; + } + return false; +} export function getCapabilitiesForModel(provider, model) { if (!model) return { ...DEFAULT_CAPABILITIES }; // Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7". const baseModel = model.includes("/") ? model.split("/").pop() : model; + // CommandCode wire is /alpha/generate for every model. Family patterns + // (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here. + if (provider === "commandcode" || provider === "cmc") { + const providerCaps = PROVIDER_CAPABILITIES.commandcode; + if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] }; + if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] }; + return { + ...DEFAULT_CAPABILITIES, + reasoning: true, + thinkingFormat: "commandcode", + thinkingEffortSupported: true, + vision: !isCommandCodeTextOnly(model), + contextWindow: 1000000, + maxOutput: 384000, + }; + } + // 1. Provider-specific override if (provider) { const providerCaps = PROVIDER_CAPABILITIES[provider]; diff --git a/open-sse/providers/registry/commandcode.js b/open-sse/providers/registry/commandcode.js index 3b21fbbc..f59aac04 100644 --- a/open-sse/providers/registry/commandcode.js +++ b/open-sse/providers/registry/commandcode.js @@ -40,4 +40,8 @@ export default { { id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" }, { id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" }, ], + features: { + usage: true, + usageApikey: true, + }, }; diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index 404fa75e..0b77e4b4 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -26,6 +26,7 @@ const FORMAT_LEVELS = { qwen: L.base, kimi: L.levelMax, deepseek: L.hiMax, + commandcode: ["none", "low", "medium", "high", "xhigh", "max"], minimax: L.onOff, hunyuan: L.base, step: L.base, diff --git a/open-sse/services/usage.js b/open-sse/services/usage.js index d7868760..cca5d861 100644 --- a/open-sse/services/usage.js +++ b/open-sse/services/usage.js @@ -20,6 +20,7 @@ import { getZedUsage } from "./usage/zed.js"; import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js"; import { resolveQoderCredentials } from "./qoderModels.js"; import { getGlmUsage } from "./usage/glm.js"; +import { getCommandCodeUsage } from "./usage/commandcode.js"; import { getIflowUsage, getOllamaUsage, @@ -62,6 +63,7 @@ const USAGE_HANDLERS = { groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions), zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), "xiaomi-mimo": (c) => getXiaomiMimoUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), + commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions), }; export async function getUsageForProvider(connection, proxyOptions = null, options = {}) { diff --git a/open-sse/services/usage/commandcode.js b/open-sse/services/usage/commandcode.js new file mode 100644 index 00000000..7c62480c --- /dev/null +++ b/open-sse/services/usage/commandcode.js @@ -0,0 +1,134 @@ +/** + * Command Code usage — billing credits + 5h/weekly rate windows. + * Mirrors ~/cc-usage.mjs: whoami → credits + subscriptions. + */ + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { parseResetTime, toFiniteNumber } from "./shared.js"; + +const BASE = (process.env.COMMAND_CODE_API_BASE_URL || "https://api.commandcode.ai").replace(/\/$/, ""); + +const PLAN_NAMES = { + "individual-go": "Go", + "individual-goat": "GOAT", + "individual-pro": "Pro", + "individual-pro-v1": "Pro", + "individual-provider": "Provider", + "individual-max": "Max", + "individual-ultra": "Ultra", + "teams-pro": "Teams Pro", +}; + +const PLAN_CAPS = { + "individual-go": 10, + "individual-goat": 70, + "individual-pro": 30, + "individual-pro-v1": 80, + "individual-provider": 15, + "individual-max": 150, + "individual-ultra": 300, + "teams-pro": 40, +}; + +function qs(route, params) { + const s = new URLSearchParams( + Object.entries(params || {}).filter(([, v]) => v != null), + ).toString(); + return s ? `${route}?${s}` : route; +} + +function windowQuota(win) { + if (!win || typeof win !== "object") return null; + const used = toFiniteNumber(win.used, 0); + const total = toFiniteNumber(win.cap, 0); + if (total <= 0 && used <= 0) return null; + return { + used, + total, + remaining: Math.max(0, total - used), + unlimited: false, + resetAt: parseResetTime(win.resetAt), + }; +} + +/** + * @param {string|null|undefined} apiKey + * @param {object|null} proxyOptions + */ +export async function getCommandCodeUsage(apiKey, proxyOptions = null) { + if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) { + return { message: "Command Code API key not available. Add a key to view usage." }; + } + + const headers = { + Authorization: `Bearer ${apiKey.trim()}`, + Accept: "application/json", + }; + + const get = async (route) => { + const response = await proxyAwareFetch( + BASE + route, + { method: "GET", headers }, + proxyOptions, + ); + return response; + }; + + try { + const whoamiRes = await get(qs("/alpha/whoami", { limits: "1" })); + if (whoamiRes.status === 401 || whoamiRes.status === 403) { + return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." }; + } + if (!whoamiRes.ok) { + return { plan: "Command Code", message: `Command Code usage API error (${whoamiRes.status})` }; + } + const whoami = await whoamiRes.json().catch(() => ({})); + const orgId = whoami?.org?.id ?? null; + + const [creditsRes, subsRes] = await Promise.all([ + get(qs("/alpha/billing/credits", { orgId })), + get(qs("/alpha/billing/subscriptions", { orgId })), + ]); + + if (creditsRes.status === 401 || creditsRes.status === 403 || subsRes.status === 401 || subsRes.status === 403) { + return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." }; + } + if (!creditsRes.ok) { + return { plan: "Command Code", message: `Command Code credits API error (${creditsRes.status})` }; + } + if (!subsRes.ok) { + return { plan: "Command Code", message: `Command Code subscriptions API error (${subsRes.status})` }; + } + + const creditsBody = await creditsRes.json().catch(() => ({})); + const subsBody = await subsRes.json().catch(() => ({})); + const planId = subsBody?.data?.planId ?? null; + const plan = (planId && PLAN_NAMES[planId]) || planId || "Command Code"; + const cap = planId ? (PLAN_CAPS[planId] || 0) : 0; + const c = creditsBody?.credits || {}; + const remaining = + toFiniteNumber(c.monthlyCredits, 0) + + toFiniteNumber(c.purchasedCredits, 0) + + toFiniteNumber(c.freeCredits, 0); + const used = cap > 0 ? Math.max(0, cap - remaining) : 0; + const total = cap > 0 ? cap : remaining; + + const quotas = {}; + quotas.Credits = { + used, + total, + remaining, + unlimited: cap <= 0, + resetAt: parseResetTime(subsBody?.data?.currentPeriodEnd), + }; + + const fiveHour = windowQuota(creditsBody?.windowLimits?.fiveHour); + if (fiveHour) quotas["Session (5h)"] = fiveHour; + const weekly = windowQuota(creditsBody?.windowLimits?.weekly); + if (weekly) quotas.Weekly = weekly; + + return { plan, quotas }; + } catch (error) { + return { message: `Command Code error: ${error.message}` }; + } +} diff --git a/open-sse/translator/concerns/prefetch.js b/open-sse/translator/concerns/prefetch.js index 5055be3c..4d606ebd 100644 --- a/open-sse/translator/concerns/prefetch.js +++ b/open-sse/translator/concerns/prefetch.js @@ -8,6 +8,7 @@ import { fetchImageAsBase64, parseDataUri } from "./image.js"; const TARGETS_NEED_BASE64 = new Set([ FORMATS.GEMINI, FORMATS.GEMINI_CLI, FORMATS.VERTEX, FORMATS.ANTIGRAVITY, FORMATS.OLLAMA, FORMATS.KIRO, + FORMATS.COMMANDCODE, ]); function isRemoteUrl(url) { diff --git a/open-sse/translator/concerns/thinkingUnified.js b/open-sse/translator/concerns/thinkingUnified.js index e86dbef7..001259e3 100644 --- a/open-sse/translator/concerns/thinkingUnified.js +++ b/open-sse/translator/concerns/thinkingUnified.js @@ -19,6 +19,7 @@ const FORMAT_TO_NATIVE = { vertex: "gemini-budget", antigravity: "gemini-budget", kiro: "kiro", + commandcode: "commandcode", }; // Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent). @@ -108,6 +109,7 @@ export const captureThinking = extractThinking; const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]); function resolveFormat(targetFormat, model, provider) { + if (targetFormat === "commandcode") return "commandcode"; const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null; if (providerFmt) return providerFmt; const caps = getCapabilitiesForModel(provider, model); @@ -223,6 +225,10 @@ function stripAll(body) { delete body.output_config; if (body.generationConfig) delete body.generationConfig.thinkingConfig; if (body.request?.generationConfig) delete body.request.generationConfig.thinkingConfig; + if (body.params && typeof body.params === "object") { + delete body.params.reasoning_effort; + delete body.params.thinking; + } } // Apply unified thinking config to body in the resolved provider-native format. @@ -336,6 +342,17 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) { case "kiro": // Kiro thinking handled via system-tag injection in openai-to-kiro.js; no body field here. break; + case "commandcode": { + // Native CLI sends reasoning_effort inside params of the /alpha/generate envelope. + if (!body.params || typeof body.params !== "object") body.params = {}; + if (none && canDisable) { + delete body.params.reasoning_effort; + break; + } + const level = toLevel(eff); + if (level) body.params.reasoning_effort = level; + break; + } default: break; } diff --git a/open-sse/translator/request/openai-to-commandcode.js b/open-sse/translator/request/openai-to-commandcode.js index 9825048b..194078ae 100644 --- a/open-sse/translator/request/openai-to-commandcode.js +++ b/open-sse/translator/request/openai-to-commandcode.js @@ -5,6 +5,7 @@ * - params.system: STRING at top level (Anthropic-style; system messages NOT allowed in messages[]) * - params.messages[*].role ∈ {"user","assistant","tool"} * - params.messages[*].content: Array of content blocks (NEVER a string) + * - image_url / image source → {type:"image", image:"data:...;base64,...", mimeType} * - tool_use blocks (assistant): {type:"tool-call", toolCallId, toolName, input} * - tool_result blocks (role=user): {type:"tool-result", toolCallId, toolName, output} * - tools[*]: Anthropic plain {name, description, input_schema} @@ -12,8 +13,9 @@ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; import { randomUUID } from "crypto"; -import { ROLE, OPENAI_BLOCK } from "../schema/index.js"; +import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js"; import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js"; +import { parseDataUri, encodeDataUri } from "../concerns/image.js"; function flattenText(content) { if (content == null) return ""; @@ -29,6 +31,43 @@ function flattenText(content) { return String(content); } +function toNativeImageBlock(part) { + if (!part || typeof part !== "object") return null; + + if (part.type === OPENAI_BLOCK.IMAGE_URL) { + const url = typeof part.image_url === "string" ? part.image_url : part.image_url?.url; + const parsed = parseDataUri(url); + if (!parsed) return null; + return { + type: OPENAI_BLOCK.IMAGE, + image: encodeDataUri(parsed.mimeType, parsed.base64), + mimeType: parsed.mimeType, + }; + } + + if (part.type === OPENAI_BLOCK.IMAGE || part.type === CLAUDE_BLOCK.IMAGE) { + if (typeof part.image === "string" && part.image.startsWith("data:")) { + const parsed = parseDataUri(part.image); + return { + type: OPENAI_BLOCK.IMAGE, + image: part.image, + mimeType: part.mimeType || parsed?.mimeType || "image/png", + }; + } + const source = part.source; + if (source?.type === "base64" && typeof source.data === "string") { + const mime = source.media_type || "image/png"; + return { + type: OPENAI_BLOCK.IMAGE, + image: encodeDataUri(mime, source.data), + mimeType: mime, + }; + } + } + + return null; +} + function toContentBlocks(content) { if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }]; if (typeof content === "string") return [{ type: OPENAI_BLOCK.TEXT, text: content }]; @@ -40,10 +79,12 @@ function toContentBlocks(content) { } else if (part && typeof part === "object") { if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") { blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); - } else if (part.type === OPENAI_BLOCK.IMAGE_URL || part.type === OPENAI_BLOCK.IMAGE) { - blocks.push({ type: OPENAI_BLOCK.TEXT, text: "[image omitted]" }); - } else if (typeof part.text === "string") { - blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); + } else { + const image = toNativeImageBlock(part); + if (image) blocks.push(image); + else if (typeof part.text === "string") { + blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); + } } } } diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js index 568c4808..0a508265 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js @@ -40,7 +40,7 @@ import { } from "./utils"; import Card from "@/shared/components/Card"; import { ConfirmModal, EditConnectionModal } from "@/shared/components"; -import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; +import { USAGE_SUPPORTED_PROVIDERS, AI_PROVIDERS } from "@/shared/constants/providers"; import { useCopyToClipboard } from "@/shared/hooks/useCopyToClipboard"; // Maps the stored providerSpecificData.authMethod to a human label for Kiro. @@ -100,6 +100,10 @@ function getCodexResetCreditCount(quota) { return Number.isFinite(count) ? Math.max(0, count) : 0; } +function providerLabel(providerId) { + return AI_PROVIDERS[providerId]?.name || providerId; +} + function formatCreditDate(value) { if (!value) return "N/A"; const date = new Date(value); @@ -767,7 +771,7 @@ export default function ProviderLimits() { }; const selectedProviderLabel = - providerFilter === "all" ? "All providers" : providerFilter; + providerFilter === "all" ? "All providers" : providerLabel(providerFilter); const hasEligibleConnections = totals.eligibleConnections > 0; const hasVisibleConnections = sortedConnections.length > 0; const emptyState = getConnectionsEmptyMessage( @@ -844,7 +848,7 @@ export default function ProviderLimits() { fallbackText={providerFilter.slice(0, 2).toUpperCase()} /> )} - + {selectedProviderLabel} @@ -905,8 +909,8 @@ export default function ProviderLimits() { className="size-6 rounded-md object-contain" fallbackText={provider.slice(0, 2).toUpperCase()} /> - - {provider} + + {providerLabel(provider)} {providerFilter === provider && ( @@ -1079,8 +1083,8 @@ export default function ProviderLimits() { />
-

- {conn.provider} +

+ {providerLabel(conn.provider)}

{getConnectionLabel(conn) ? (

diff --git a/tests/translator/bugs-gemini-cursor-commandcode.test.js b/tests/translator/bugs-gemini-cursor-commandcode.test.js index 4b74c60e..8483bfc1 100644 --- a/tests/translator/bugs-gemini-cursor-commandcode.test.js +++ b/tests/translator/bugs-gemini-cursor-commandcode.test.js @@ -62,15 +62,18 @@ describe("OpenAI → CommandCode", () => { expect(Object.keys(call.input).length, "arguments silently dropped to {}").toBeGreaterThan(0); }); - // openai-to-commandcode.js:41-42 — image becomes "[image omitted]" - // KNOWN BUG - it.fails("image content is preserved", () => { + it("image content is preserved as native CommandCode image blocks", () => { const out = O2CC({ messages: [{ role: "user", content: [ { type: "text", text: "look" }, { type: "image_url", image_url: { url: "data:image/png;base64,BBBB" } }, ] }], }); - expect(JSON.stringify(out), "image omitted").toContain("BBBB"); + expect(JSON.stringify(out)).toContain("BBBB"); + expect(JSON.stringify(out)).not.toContain("[image omitted]"); + expect(out.params.messages[0].content).toEqual([ + { type: "text", text: "look" }, + { type: "image", image: "data:image/png;base64,BBBB", mimeType: "image/png" }, + ]); }); }); diff --git a/tests/translator/thinking-unified.test.js b/tests/translator/thinking-unified.test.js index fd8f9d28..f04b18bd 100644 --- a/tests/translator/thinking-unified.test.js +++ b/tests/translator/thinking-unified.test.js @@ -250,6 +250,29 @@ describe("applyThinking per provider format", () => { const out = apply("gemini-cli", "gemini-3.5-flash-lite", { reasoning_effort: "medium" }, "gemini-cli"); expect(out.generationConfig.thinkingConfig.thinkingLevel).toBe("medium"); }); + it("commandcode envelope writes params.reasoning_effort, not wrapper fields", () => { + const out = apply("commandcode", "deepseek/deepseek-v4.1-flash", { + params: { model: "deepseek/deepseek-v4.1-flash", messages: [] }, + reasoning_effort: "high", + }, "commandcode"); + expect(out.params.reasoning_effort).toBe("high"); + expect(out.reasoning_effort).toBeUndefined(); + expect(out.thinking).toBeUndefined(); + }); + it("commandcode preserves low effort instead of remapping to high", () => { + const out = apply("commandcode", "deepseek/deepseek-v4.1-flash", { + params: { messages: [] }, + reasoning_effort: "low", + }, "commandcode"); + expect(out.params.reasoning_effort).toBe("low"); + }); + it("commandcode preserves max effort", () => { + const out = apply("commandcode", "deepseek/deepseek-v4.1-flash", { + params: { messages: [] }, + reasoning_effort: "max", + }, "commandcode"); + expect(out.params.reasoning_effort).toBe("max"); + }); }); describe("extractReasoningText (response shapes)", () => { diff --git a/tests/unit/capabilities.test.js b/tests/unit/capabilities.test.js index 3482a2c8..cde8e9b4 100644 --- a/tests/unit/capabilities.test.js +++ b/tests/unit/capabilities.test.js @@ -73,4 +73,26 @@ describe("getCapabilitiesForModel", () => { maxOutput: 128000, }); }); + + it("CommandCode v4.1-flash is vision + effort capable", () => { + expect(getCapabilitiesForModel("commandcode", "deepseek/deepseek-v4.1-flash")).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "commandcode", + thinkingEffortSupported: true, + }); + }); + + it("CommandCode MiniMax-M3 is vision capable", () => { + expect(getCapabilitiesForModel("commandcode", "MiniMaxAI/MiniMax-M3").vision).toBe(true); + }); + + it("CommandCode text-only DeepSeek V4 Flash stays non-vision", () => { + expect(getCapabilitiesForModel("commandcode", "deepseek/deepseek-v4-flash").vision).toBe(false); + expect(getCapabilitiesForModel("commandcode", "deepseek/deepseek-v4-flash")).toMatchObject({ + reasoning: true, + thinkingFormat: "commandcode", + thinkingEffortSupported: true, + }); + }); }); diff --git a/tests/unit/commandcode-usage.test.js b/tests/unit/commandcode-usage.test.js new file mode 100644 index 00000000..5f38dc47 --- /dev/null +++ b/tests/unit/commandcode-usage.test.js @@ -0,0 +1,135 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import { getUsageForProvider } from "../../open-sse/services/usage.js"; +import { + USAGE_SUPPORTED_PROVIDERS, + USAGE_APIKEY_PROVIDERS, +} from "../../src/shared/constants/providers.js"; +import { parseQuotaData } from "../../src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; + +const BASE = "https://api.commandcode.ai"; + +function jsonResponse(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +const WHOAMI = { + user: { name: "Hieu", email: "hieu@example.com" }, + org: { id: "org_1", name: "personal" }, +}; +const CREDITS = { + credits: { monthlyCredits: 12.5, purchasedCredits: 1, freeCredits: 0.5 }, + windowLimits: { + fiveHour: { used: 2, cap: 10, resetAt: Date.now() + 3_600_000, exceeded: false }, + weekly: { used: 20, cap: 70, resetAt: Date.now() + 86_400_000, exceeded: false }, + }, +}; +const SUBS = { + data: { + planId: "individual-goat", + currentPeriodStart: "2026-09-01T00:00:00.000Z", + currentPeriodEnd: "2026-10-01T00:00:00.000Z", + }, +}; + +function mockHappyPath() { + proxyAwareFetch.mockImplementation(async (url) => { + const u = String(url); + if (u.includes("/alpha/whoami")) return jsonResponse(WHOAMI); + if (u.includes("/alpha/billing/credits")) return jsonResponse(CREDITS); + if (u.includes("/alpha/billing/subscriptions")) return jsonResponse(SUBS); + return jsonResponse({ error: "unexpected " + u }, 404); + }); +} + +describe("commandcode registry usage flags", () => { + it("is listed for apikey quota dashboard", () => { + expect(USAGE_SUPPORTED_PROVIDERS).toContain("commandcode"); + expect(USAGE_APIKEY_PROVIDERS).toContain("commandcode"); + }); +}); + +describe("getUsageForProvider(commandcode)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it("returns a message when apiKey is missing", async () => { + const usage = await getUsageForProvider({ provider: "commandcode" }); + expect(usage.message).toMatch(/api key/i); + expect(proxyAwareFetch).not.toHaveBeenCalled(); + }); + + it("GETs whoami, credits, and subscriptions with Bearer apiKey", async () => { + mockHappyPath(); + const usage = await getUsageForProvider({ + provider: "commandcode", + apiKey: "user_test", + }); + + expect(usage.message).toBeUndefined(); + expect(usage.plan).toBe("GOAT"); + const urls = proxyAwareFetch.mock.calls.map(([url]) => String(url)); + expect(urls.some((u) => u.startsWith(`${BASE}/alpha/whoami`))).toBe(true); + expect(urls.some((u) => u.includes("/alpha/billing/credits") && u.includes("orgId=org_1"))).toBe(true); + expect(urls.some((u) => u.includes("/alpha/billing/subscriptions") && u.includes("orgId=org_1"))).toBe(true); + expect(proxyAwareFetch.mock.calls[0][1].headers.Authorization).toBe("Bearer user_test"); + }); + + it("maps remaining credits vs plan cap and rate windows", async () => { + mockHappyPath(); + const usage = await getUsageForProvider({ + provider: "commandcode", + apiKey: "user_test", + }); + + // remaining = 12.5 + 1 + 0.5 = 14; cap GOAT = 70; used = 56 + expect(usage.quotas.Credits).toMatchObject({ + used: 56, + total: 70, + unlimited: false, + }); + expect(usage.quotas["Session (5h)"]).toMatchObject({ + used: 2, + total: 10, + unlimited: false, + }); + expect(usage.quotas.Weekly).toMatchObject({ + used: 20, + total: 70, + }); + expect(new Date(usage.quotas.Credits.resetAt).toISOString()).toBe("2026-10-01T00:00:00.000Z"); + }); + + it("returns an auth message on 401", async () => { + proxyAwareFetch.mockResolvedValueOnce(jsonResponse({ error: "unauthorized" }, 401)); + const usage = await getUsageForProvider({ + provider: "commandcode", + apiKey: "bad", + }); + expect(usage.message).toMatch(/auth|key|login/i); + }); +}); + +describe("parseQuotaData(commandcode)", () => { + it("forwards used/total/resetAt for the dashboard table", () => { + const rows = parseQuotaData("commandcode", { + plan: "GOAT", + quotas: { + Credits: { used: 56, total: 70, resetAt: "2026-10-01T00:00:00.000Z" }, + "Session (5h)": { used: 2, total: 10, resetAt: "2026-09-16T10:00:00.000Z" }, + }, + }); + expect(rows).toHaveLength(2); + expect(rows[0]).toMatchObject({ name: "Credits", used: 56, total: 70 }); + expect(rows[1]).toMatchObject({ name: "Session (5h)", used: 2, total: 10 }); + }); +}); diff --git a/tests/unit/openai-to-commandcode.test.js b/tests/unit/openai-to-commandcode.test.js index 7f12dc85..0a441dd7 100644 --- a/tests/unit/openai-to-commandcode.test.js +++ b/tests/unit/openai-to-commandcode.test.js @@ -179,3 +179,57 @@ describe("openaiToCommandCodeRequest — tools schema conversion", () => { expect(out.params.tools).toBeUndefined(); }); }); + +describe("openaiToCommandCodeRequest — native image blocks", () => { + const PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="; + const DATA_URI = `data:image/png;base64,${PNG_B64}`; + + it("maps OpenAI image_url data URI to CommandCode {type:image,image,mimeType}", () => { + const out = openaiToCommandCodeRequest(MODEL, { + messages: [{ + role: "user", + content: [ + { type: "text", text: "what color?" }, + { type: "image_url", image_url: { url: DATA_URI } }, + ], + }], + }, true); + + expect(out.params.messages[0].content).toEqual([ + { type: "text", text: "what color?" }, + { type: "image", image: DATA_URI, mimeType: "image/png" }, + ]); + }); + + it("maps Claude/OpenAI base64 image source to a data-URI image block", () => { + const out = openaiToCommandCodeRequest(MODEL, { + messages: [{ + role: "user", + content: [ + { type: "image", source: { type: "base64", media_type: "image/png", data: PNG_B64 } }, + ], + }], + }, true); + + expect(out.params.messages[0].content).toEqual([ + { type: "image", image: DATA_URI, mimeType: "image/png" }, + ]); + }); + + it("does not stub dropped images as [image omitted]", () => { + const out = openaiToCommandCodeRequest(MODEL, { + messages: [{ + role: "user", + content: [ + { type: "text", text: "see this" }, + { type: "image_url", image_url: { url: DATA_URI } }, + ], + }], + }, true); + + const texts = out.params.messages[0].content + .filter((b) => b.type === "text") + .map((b) => b.text); + expect(texts).not.toContain("[image omitted]"); + }); +}); diff --git a/tests/unit/prefetch-images.test.js b/tests/unit/prefetch-images.test.js index 2289697b..0d4c0db3 100644 --- a/tests/unit/prefetch-images.test.js +++ b/tests/unit/prefetch-images.test.js @@ -55,4 +55,21 @@ describe("prefetchRemoteImages", () => { expect(n).toBe(1); expect(body.messages[0].content[0].source.type).toBe("base64"); }); + + it("openai source -> commandcode target: converts remote URL to base64", async () => { + const body = { messages: [{ role: "user", content: [{ type: "image_url", image_url: { url: "https://x/a.png" } }] }] }; + const n = await prefetchRemoteImages(body, FORMATS.OPENAI, FORMATS.COMMANDCODE); + expect(n).toBe(1); + expect(body.messages[0].content[0].image_url.url.startsWith("data:image/png;base64,")).toBe(true); + expect(fetchImageAsBase64).toHaveBeenCalled(); + }); + + it("claude source -> commandcode target: source.url -> base64", async () => { + const body = { messages: [{ role: "user", content: [ + { type: "image", source: { type: "url", url: "https://x/a.png" } }, + ] }] }; + const n = await prefetchRemoteImages(body, FORMATS.CLAUDE, FORMATS.COMMANDCODE); + expect(n).toBe(1); + expect(body.messages[0].content[0].source.type).toBe("base64"); + }); }); diff --git a/tests/unit/usage-dispatch.test.js b/tests/unit/usage-dispatch.test.js index 84a5f469..ead63892 100644 --- a/tests/unit/usage-dispatch.test.js +++ b/tests/unit/usage-dispatch.test.js @@ -16,7 +16,7 @@ const SUPPORTED = [ "github", "gemini-cli", "antigravity", "claude", "codex", "kiro", "qoder", "iflow", "ollama", "glm", "glm-cn", "minimax", "minimax-cn", "vercel-ai-gateway", "grok-cli", "kimi", - "deepseek", "opencode-go", "zed", + "deepseek", "opencode-go", "zed", "commandcode", ]; describe("usage dispatch", () => { From 912ed295db3153f01f44f3e56f93fa4616b69607 Mon Sep 17 00:00:00 2001 From: izzzzzi Date: Thu, 17 Sep 2026 17:52:47 +0700 Subject: [PATCH 51/78] fix(deepseek,model-catalog): vision for V4.1-Flash ids, scope synced catalog to gateways - Declare deepseek-v4.1-flash and deepseek-flash as vision-capable in MODEL_CAPABILITIES - Share installed catalogSource across route chunks via globalThis.__9rCatalogSource - Scope catalog modality keys by provider:model to prevent cross-gateway collisions - Upgrade catalog format to v2 with automatic rebuild of older schemas --- open-sse/providers/capabilities.js | 32 ++++- open-sse/providers/catalogOverride.js | 22 +++- src/lib/modelCatalog/sync.js | 105 +++++++++------ tests/unit/capabilities.test.js | 18 +++ tests/unit/model-catalog-scope.test.js | 174 +++++++++++++++++++++++++ 5 files changed, 299 insertions(+), 52 deletions(-) create mode 100644 tests/unit/model-catalog-scope.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 177db4a7..eac9089e 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = { // DeepSeek's first V4 model with image input; text limits match V4-Flash. "deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + // DeepSeek V4.1-Flash is natively multimodal — models.dev lists + // opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream + // the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the + // same image capability as the exp id above. "deepseek-flash" is the GA id on the + // DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose + // 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry + // short-circuits the pattern table, so a vision-only delta would drop them. + "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + "deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 }, + // Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases "vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 }, "coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 }, @@ -431,14 +441,27 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"]; // Catalog lookups, installed by the server at startup. Left as no-ops in the // browser bundle, where there is no file to read. +// +// The server bundles this module into every route chunk that needs it, and each +// copy carries its own module state, so an install landing in the copy the +// startup hook imported stays invisible to the copy resolving requests. The slot +// lives on globalThis instead; the local binding is the fast path. let catalogSource = null; /** * Install the synced catalog reader (server only). - * @param {{ getModalities: Function, getLimits: Function } | null} source + * @param {{ getModalities: (provider: string, model: string) => object|null, + * getLimits: (provider: string, model: string) => object|null } | null} source */ export function setCatalogSource(source) { catalogSource = source; + if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source; +} + +function getCatalogSource() { + if (catalogSource) return catalogSource; + if (typeof globalThis === "undefined") return null; + return (catalogSource = globalThis.__9rCatalogSource || null); } // Apply the synced catalog + name heuristic on top of a table-resolved result. @@ -447,15 +470,16 @@ export function setCatalogSource(source) { function refine(base, provider, model) { const result = { ...DEFAULT_CAPABILITIES, ...base }; - if (catalogSource) { - const modalities = catalogSource.getModalities(model); + const source = getCatalogSource(); + if (source) { + const modalities = source.getModalities(provider, model); if (modalities) { for (const key of MODALITY_KEYS) { if (modalities[key] === true) result[key] = true; } } - const limits = catalogSource.getLimits(provider, model); + const limits = source.getLimits(provider, model); if (limits) { if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow; if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput; diff --git a/open-sse/providers/catalogOverride.js b/open-sse/providers/catalogOverride.js index 12914b7d..c34576d9 100644 --- a/open-sse/providers/catalogOverride.js +++ b/open-sse/providers/catalogOverride.js @@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json"); // Trimmed upstream catalog, read by the add-models skill (not by the router). export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json"); +// Schema of the file this module reads. The writer stamps it; a file carrying an +// older value predates provider-scoped modality keys, and its flat keys are not +// looked up here, so the sync rebuilds it instead of asking upstream for a 304. +export const CATALOG_VERSION = 2; + const EMPTY = { models: {}, providers: {} }; let cache = EMPTY; let cachedMtime = -1; @@ -45,14 +50,19 @@ function load() { return cache; } -// Modality is a property of the model itself — any gateway serving it inherits -// the same image/video/pdf support, so this is keyed by model id alone. -export function getCatalogModalities(model) { - return load().models[baseId(model)] || null; +// Modalities are recorded per gateway upstream, and gateways disagree about the +// same weights — some do not proxy images at all — so the key is provider + +// model, in the local provider id space, exactly like the limits below. Keying +// by model id alone made short ids collide across vendors: "auto", "free" and +// "efficient" are router modes in one catalog and model names in another, and a +// request to the router mode inherited a stranger's vision. +export function getCatalogModalities(provider, model) { + if (!provider) return null; + return load().models[`${provider}:${baseId(model)}`] || null; } -// Context and output limits are a property of the gateway, not the model: each -// one truncates differently, so these stay keyed by provider + model. +// Context and output limits are a property of the gateway too: each one +// truncates differently, so these stay keyed by provider + model. export function getCatalogLimits(provider, model) { const byProvider = provider && load().providers[provider]; if (!byProvider) return null; diff --git a/src/lib/modelCatalog/sync.js b/src/lib/modelCatalog/sync.js index 0d49c48e..973c18b3 100644 --- a/src/lib/modelCatalog/sync.js +++ b/src/lib/modelCatalog/sync.js @@ -6,7 +6,7 @@ import fs from "node:fs"; import path from "node:path"; -import { CATALOG_FILE, CATALOG_RAW_FILE, invalidateCatalog, installCatalogSource } from "open-sse/providers/catalogOverride.js"; +import { CATALOG_FILE, CATALOG_RAW_FILE, CATALOG_VERSION, invalidateCatalog, installCatalogSource } from "open-sse/providers/catalogOverride.js"; const CATALOG_URL = "https://models.dev/api.json"; const FETCH_TIMEOUT_MS = 60000; @@ -16,16 +16,14 @@ const STARTUP_DELAY_MS = 60 * 1000; // let the server boot and serve first req const RETRY_DELAY_MS = 30 * 60 * 1000; const MODALITY_BY_INPUT = { image: "vision", pdf: "pdf", audio: "audioInput", video: "videoInput" }; -// Gateways disagree about the same model, so a modality needs a majority of -// them to declare it — one reseller mislabelling a text model must not win. -const MIN_SHARE = 0.5; // Ignore limit differences below this: gateways round 200000 vs 202752. const LIMIT_TOLERANCE = 0.1; -// 9router provider id -> models.dev provider id, for context/maxOutput only. -// Providers absent here keep whatever the local pattern table resolves; names -// that already match are resolved automatically. -const PROVIDER_ALIASES = { +// 9router provider id -> models.dev provider id: the same gateway under another +// name. Both halves of the catalog are stored against the local id, so this runs +// while building rather than on every lookup. Providers absent here keep whatever +// the local pattern table resolves; names that already match need no entry. +export const PROVIDER_ALIASES = { "glm": "zai", "glm-cn": "zhipuai", "claude": "anthropic", @@ -40,7 +38,7 @@ const PROVIDER_ALIASES = { "cloudflare-ai": "cloudflare-workers-ai", }; -let state = { running: false, lastSync: null, lastError: null, lastResult: null, etag: null }; +let state = { running: false, lastSync: null, lastError: null, lastResult: null, etag: null, fileVersion: null }; let timer = null; export function getSyncState() { @@ -78,40 +76,57 @@ function slim(catalog) { return out; } -function build(catalog, entries) { - // Index once: per provider for limits, and tallied across all of them for - // modalities. - const byProvider = {}; - const tally = {}; - for (const [providerId, provider] of Object.entries(catalog)) { - const models = {}; - const counted = new Set(); - for (const [modelId, model] of Object.entries(provider?.models || {})) { - const id = baseId(modelId); - models[id] = model; - // One vote per provider: several ids can normalize to the same model - // (claude-opus-4-thinking:1024, :8192, :32768 …) and must not stack. - if (counted.has(id)) continue; - counted.add(id); - const counts = tally[id] || (tally[id] = { total: 0 }); - counts.total++; - for (const input of model?.modalities?.input || []) { - const key = MODALITY_BY_INPUT[input]; - if (key) counts[key] = (counts[key] || 0) + 1; - } - } - byProvider[providerId] = models; +export function build(catalog, entries) { + // Upstream provider id -> the local ids it belongs to, taken from the registry + // snapshot so a gateway listed upstream under another name is still filed + // under the name requests arrive with. One upstream name can back more than one + // local id (glm-cn and zhipu are both zhipuai) and each has to resolve; the + // snapshot only covers the built-in registry, so an upstream provider it does + // not mention keeps its own name. + const localIds = new Map(); + for (const { provider } of entries) { + const upstreamId = PROVIDER_ALIASES[provider] || provider; + let locals = localIds.get(upstreamId); + if (!locals) localIds.set(upstreamId, (locals = [])); + if (!locals.includes(provider)) locals.push(provider); } - // Modalities belong to the model — every gateway serving it has the same - // weights — so they are keyed by model id and shared across providers. + // Index once: the raw upstream record per provider+model for limits, and the + // modalities each gateway declares for it. + const byProvider = {}; + // Modalities are recorded per gateway upstream and gateways disagree about the + // same weights — some do not proxy images at all — so the key is provider + + // model. Keying by model id alone let short ids collide across vendors: "auto", + // "free" and "efficient" are router modes in one catalog and model names in + // another, so a router mode inherited a stranger's vision. const models = {}; - for (const [id, counts] of Object.entries(tally)) { - const declared = {}; - for (const key of Object.values(MODALITY_BY_INPUT)) { - if ((counts[key] || 0) / counts.total >= MIN_SHARE) declared[key] = true; + for (const [providerId, provider] of Object.entries(catalog)) { + const locals = localIds.get(providerId) || [providerId]; + const modelsById = {}; + const seen = new Set(); + for (const [modelId, model] of Object.entries(provider?.models || {})) { + const id = baseId(modelId); + modelsById[id] = model; + // One entry per provider+model: several upstream ids can normalize to the + // same model (claude-opus-4-thinking:1024, :8192, :32768 …) and must not + // stack their modalities. + if (seen.has(id)) continue; + seen.add(id); + const declared = {}; + for (const input of model?.modalities?.input || []) { + const key = MODALITY_BY_INPUT[input]; + if (key) declared[key] = true; + } + if (Object.keys(declared).length) { + // Filed under every local id requests arrive with, and under the upstream + // id too: a custom provider node can carry the upstream name without + // appearing in the registry snapshot, and nothing else would resolve for + // it. The reader takes whichever key it is handed. + for (const local of locals) models[`${local}:${id}`] = declared; + if (!locals.includes(providerId)) models[`${providerId}:${id}`] = declared; + } } - if (Object.keys(declared).length) models[id] = declared; + byProvider[providerId] = modelsById; } // Limits belong to the gateway — each truncates differently — so only the @@ -172,7 +187,9 @@ export async function syncModelCatalog() { state.running = true; try { const headers = { accept: "application/json" }; - if (state.etag) headers["if-none-match"] = state.etag; + // A file written by an older schema has to be rebuilt even when upstream is + // unchanged, so only ask upstream for a 304 when the file is current. + if (state.etag && state.fileVersion === CATALOG_VERSION) headers["if-none-match"] = state.etag; const response = await fetch(CATALOG_URL, { headers, signal: AbortSignal.timeout(FETCH_TIMEOUT_MS) }); let result; @@ -187,12 +204,13 @@ export async function syncModelCatalog() { const etag = response.headers.get("etag") || null; const entries = await collectEntries(); const { models, providers } = build(catalog, entries); - const serialized = JSON.stringify({ v: 1, etag, syncedAt: Date.now(), models, providers }); + const serialized = JSON.stringify({ v: CATALOG_VERSION, etag, syncedAt: Date.now(), models, providers }); writeAtomic(CATALOG_FILE, serialized); writeAtomic(CATALOG_RAW_FILE, JSON.stringify(slim(catalog))); state.etag = etag; + state.fileVersion = CATALOG_VERSION; invalidateCatalog(); result = { status: "updated", @@ -223,10 +241,13 @@ export async function syncModelCatalog() { // of re-downloading 4.3MB to be told nothing changed. function restoreEtag() { try { - state.etag = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8")).etag || null; + const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8")); + state.etag = parsed.etag || null; + state.fileVersion = parsed.v || 1; state.lastSync = fs.statSync(CATALOG_FILE).mtimeMs; } catch { state.etag = null; + state.fileVersion = null; } } diff --git a/tests/unit/capabilities.test.js b/tests/unit/capabilities.test.js index cde8e9b4..84a8643c 100644 --- a/tests/unit/capabilities.test.js +++ b/tests/unit/capabilities.test.js @@ -2,6 +2,24 @@ import { describe, expect, it } from "vitest"; import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; describe("getCapabilitiesForModel", () => { + + it("reports DeepSeek V4.1-Flash ids as vision-capable without dropping their thinking/context", () => { + const v41 = { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }; + expect(getCapabilitiesForModel(undefined, "deepseek-v4.1-flash")).toMatchObject(v41); + expect(getCapabilitiesForModel("opencode-go", "deepseek-v4.1-flash")).toMatchObject(v41); + expect(getCapabilitiesForModel("openrouter", "deepseek/deepseek-v4.1-flash")).toMatchObject(v41); + // "deepseek-flash" is the GA id for V4.1-Flash on the DeepSeek API; the pattern it + // used to fall through to gives it 128K/64K, which the exact entry keeps. + expect(getCapabilitiesForModel("opencode-go", "deepseek-flash")).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "deepseek", + contextWindow: 128000, + maxOutput: 64000, + }); + // the superseded text-only Flash id stays text-only + expect(getCapabilitiesForModel("opencode-go", "deepseek-v4-flash").vision).toBe(false); + }); const claudeSonnet5Expected = { contextWindow: 1000000, maxOutput: 128000, diff --git a/tests/unit/model-catalog-scope.test.js b/tests/unit/model-catalog-scope.test.js new file mode 100644 index 00000000..c582d28f --- /dev/null +++ b/tests/unit/model-catalog-scope.test.js @@ -0,0 +1,174 @@ +import { afterAll, beforeAll, describe, expect, it } from "vitest"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// Both modules read their file path from DATA_DIR at import time, so the temp +// data dir has to be in place before the first import. +const dataDir = fs.mkdtempSync(path.join(os.tmpdir(), "9r-catalog-")); +process.env.DATA_DIR = dataDir; +const catalogFile = path.join(dataDir, "model-catalog.json"); + +// One upstream record per gateway: the same short id means different things to +// different vendors, which is what used to leak capabilities across providers. +const upstream = { + zai: { models: { "glm-4.6v": { modalities: { input: ["text", "image"] } } } }, + // two local ids alias this one upstream provider + zhipuai: { models: { "glm-5-canary": { modalities: { input: ["text", "image", "pdf"] } } } }, + moonshotai: { models: { "kimi-k3": { modalities: { input: ["text"] } } } }, + kilo: { models: { "kilo-auto/efficient": { modalities: { input: ["text", "image"] } } } }, +}; +// The registry snapshot the sync feeds build(): local ids, with the capabilities +// the tables resolve on their own. +const entries = [ + { provider: "glm", model: "glm-4.6v", current: { contextWindow: 200000, maxOutput: 128000 } }, + { provider: "glm-cn", model: "glm-5-canary", current: { contextWindow: 200000, maxOutput: 128000 } }, + { provider: "zhipu", model: "glm-5-canary", current: { contextWindow: 200000, maxOutput: 128000 } }, + { provider: "kimi", model: "kimi-k3", current: { contextWindow: 128000, maxOutput: 32000 } }, +]; + +let build, getCatalogModalities, invalidateCatalog, syncModelCatalog, startModelCatalogSync, capabilities; + +beforeAll(async () => { + ({ build, syncModelCatalog, startModelCatalogSync } = await import("../../src/lib/modelCatalog/sync.js")); + // the builder is exercised directly; a missing export must fail loudly here + // rather than skip every case below + expect(typeof build).toBe("function"); + const { models, providers } = build(upstream, entries); + fs.writeFileSync(catalogFile, JSON.stringify({ v: 2, models, providers })); + ({ getCatalogModalities, invalidateCatalog } = await import("../../open-sse/providers/catalogOverride.js")); + capabilities = await import("../../open-sse/providers/capabilities.js"); +}); + +afterAll(() => { + fs.rmSync(dataDir, { recursive: true, force: true }); +}); + +describe("model catalog", () => { + it("keys modalities by gateway and writes no model-only key", () => { + const { models } = build(upstream, entries); + // upstream "zai" is filed under the local id requests arrive with... + expect(models["glm:glm-4.6v"]).toEqual({ vision: true }); + // ...and under its upstream name, because a custom provider node can carry + // that name without being in the registry snapshot + expect(models["zai:glm-4.6v"]).toEqual({ vision: true }); + expect(models["kilo:efficient"]).toEqual({ vision: true }); + // the vendor-stripped key is what used to be shared with every other gateway + expect(models["glm-4.6v"]).toBeUndefined(); + expect(models["efficient"]).toBeUndefined(); + // a gateway that only declares text has nothing to contribute + expect(models["kimi:kimi-k3"]).toBeUndefined(); + }); + + it("files an upstream provider under every local id that aliases it", () => { + const { models } = build(upstream, entries); + // glm-cn and zhipu are both zhipuai upstream; neither may be dropped + expect(models["glm-cn:glm-5-canary"]).toEqual({ vision: true, pdf: true }); + expect(models["zhipu:glm-5-canary"]).toEqual({ vision: true, pdf: true }); + expect(models["zhipuai:glm-5-canary"]).toEqual({ vision: true, pdf: true }); + expect(getCatalogModalities("glm-cn", "glm-5-canary")).toEqual({ vision: true, pdf: true }); + expect(getCatalogModalities("zhipu", "glm-5-canary")).toEqual({ vision: true, pdf: true }); + }); + + it("does not hand a router mode another vendor's modalities", () => { + // kilo's "efficient" is a real model; another gateway's "efficient" is a mode + expect(getCatalogModalities("kilo", "kilo-auto/efficient")).toEqual({ vision: true }); + expect(getCatalogModalities("kilo-gateway", "kilo-auto/efficient")).toBeNull(); + expect(getCatalogModalities("qoder", "efficient")).toBeNull(); + }); + + it("resolves a gateway the file was written for, and nobody else", () => { + expect(getCatalogModalities("glm", "glm-4.6v")).toEqual({ vision: true }); + expect(getCatalogModalities("zai", "glm-4.6v")).toEqual({ vision: true }); + expect(getCatalogModalities("unrelated", "glm-4.6v")).toBeNull(); + expect(getCatalogModalities(undefined, "glm-4.6v")).toBeNull(); + }); + + it("passes the gateway to the catalog reader when refining", () => { + const seen = []; + capabilities.setCatalogSource({ + getModalities: (provider) => { + seen.push(provider); + return provider === "gateway-a" ? { vision: true } : null; + }, + getLimits: () => null, + }); + try { + // "*laguna*" resolves from the pattern table, so refine() runs + expect(capabilities.getCapabilitiesForModel("gateway-a", "laguna-9-preview").vision).toBe(true); + expect(capabilities.getCapabilitiesForModel("gateway-b", "laguna-9-preview").vision).toBe(false); + expect(seen).toContain("gateway-a"); + } finally { + capabilities.setCatalogSource(null); + } + }); + + it("shares the installed source with every copy of the module", async () => { + const source = { + getModalities: (provider) => (provider === "gateway-a" ? { vision: true } : null), + getLimits: () => null, + }; + capabilities.setCatalogSource(source); + try { + // The server bundles this module into more than one chunk and the startup + // hook only runs in one of them, so the slot has to be process-wide. + expect(globalThis.__9rCatalogSource).toBe(source); + const other = await import("../../open-sse/providers/capabilities.js?copy=2"); + expect(other.getCapabilitiesForModel).not.toBe(capabilities.getCapabilitiesForModel); + // ...and that second copy resolves through the source it never installed + expect(other.getCapabilitiesForModel("gateway-a", "laguna-9-preview").vision).toBe(true); + } finally { + capabilities.setCatalogSource(null); + } + expect(globalThis.__9rCatalogSource).toBeNull(); + }); +}); + +describe("catalog schema", () => { + it("ignores a file written before the keys were scoped", () => { + const scoped = fs.readFileSync(catalogFile); + // v1: flat model keys, which is exactly the shape that collided + fs.writeFileSync(catalogFile, JSON.stringify({ v: 1, models: { "kimi-k3": { vision: true } }, providers: {} })); + invalidateCatalog(); + expect(getCatalogModalities("kimi", "kimi-k3")).toBeNull(); + fs.writeFileSync(catalogFile, scoped); + invalidateCatalog(); + }); + + it("rebuilds an older-schema file instead of trusting its etag", async () => { + fs.writeFileSync(catalogFile, JSON.stringify({ v: 1, etag: 'W/"old"', models: {}, providers: {} })); + invalidateCatalog(); + startModelCatalogSync(); // picks the file's etag + schema version back up + + const sent = []; + const realFetch = globalThis.fetch; + globalThis.fetch = async (_url, options) => { + sent.push(options?.headers || {}); + return { ok: true, status: 200, headers: new Map([["etag", 'W/"new"']]), json: async () => upstream }; + }; + try { + expect((await syncModelCatalog()).status).toBe("updated"); + } finally { + globalThis.fetch = realFetch; + } + expect(sent[0]["if-none-match"]).toBeUndefined(); + const written = JSON.parse(fs.readFileSync(catalogFile, "utf8")); + expect(written.v).toBe(2); + expect(written.models["glm:glm-4.6v"]).toEqual({ vision: true }); + }); + + it("asks upstream for a 304 once the file is current", async () => { + const sent = []; + const realFetch = globalThis.fetch; + globalThis.fetch = async (_url, options) => { + sent.push(options?.headers || {}); + return { ok: false, status: 304, headers: new Map(), json: async () => ({}) }; + }; + try { + expect((await syncModelCatalog()).status).toBe("unchanged"); + } finally { + globalThis.fetch = realFetch; + } + expect(sent[0]["if-none-match"]).toBe('W/"new"'); + }); +}); From 702b57c30d0e4f3cae2d188f342710d3dd10cf82 Mon Sep 17 00:00:00 2001 From: Ridho Perdana Date: Thu, 17 Sep 2026 17:59:45 +0700 Subject: [PATCH 52/78] fix(opencode-go): route every responses-only model (incl. thinking variants) to /responses Derive responses-only routing from the model registry's targetFormat instead of hardcoding model checks, and strip thinking suffixes when looking up models in providerModels so variants like gpt-5.6-luna(high) are routed correctly to /responses. --- open-sse/config/providerModels.js | 9 ++++++--- open-sse/executors/opencode-go.js | 8 ++++++-- tests/unit/opencode-go-models.test.js | 14 +++++++++++++- .../opencode-go-muse-spark-responses.test.js | 16 ++++++++++++++++ 4 files changed, 41 insertions(+), 6 deletions(-) diff --git a/open-sse/config/providerModels.js b/open-sse/config/providerModels.js index 065def24..afa98fa7 100644 --- a/open-sse/config/providerModels.js +++ b/open-sse/config/providerModels.js @@ -27,11 +27,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]); // ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only. function findModel(models, modelId, aliasOrId) { if (!models) return undefined; - const found = models.find(m => m.id === modelId); + const baseModelId = typeof modelId === "string" + ? modelId.replace(/\([^()]+\)\s*$/, "").trim() + : modelId; + const found = models.find(m => m.id === modelId || m.id === baseModelId); if (found) return found; if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined; - const normalized = normalizeModelId(modelId); - if (normalized === modelId) return undefined; + const normalized = normalizeModelId(baseModelId); + if (normalized === baseModelId) return undefined; return models.find(m => m.id === normalized); } diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js index efb656ac..c7065ea7 100644 --- a/open-sse/executors/opencode-go.js +++ b/open-sse/executors/opencode-go.js @@ -1,7 +1,8 @@ import crypto from "node:crypto"; import { DefaultExecutor } from "./default.js"; import { resolveSessionId } from "../utils/sessionManager.js"; -import { isMuseSparkModel } from "../providers/models/helpers.js"; +import { modelTargetFormat } from "../providers/models/schema.js"; +import { getProviderModels } from "../config/providerModels.js"; import { normalizeResponsesInput, clampResponsesCallId, @@ -45,8 +46,11 @@ function baseModelId(model) { return String(model || "").replace(/\([^()]+\)\s*$/, "").trim(); } +// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …). +// Reading the registry keeps this in sync with config — never hardcode model ids here. function isResponsesModel(model) { - return isMuseSparkModel(baseModelId(model)); + const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model)); + return modelTargetFormat(entry) === "openai-responses"; } // Flatten Chat Completions tool declarations into the Responses flat shape and diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 4ee2042a..92e7a059 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -1,5 +1,5 @@ import { describe, expect, it } from "vitest"; -import { PROVIDER_MODELS, getModelSupportedFormats } from "../../open-sse/config/providerModels.js"; +import { PROVIDER_MODELS, getModelSupportedFormats, getModelTargetFormat } from "../../open-sse/config/providerModels.js"; import { PROVIDERS } from "../../open-sse/config/providers.js"; import { resolveTransport } from "../../open-sse/services/provider.js"; @@ -37,6 +37,18 @@ describe("OpenCode Go model catalog", () => { }); }); +describe("OpenCode Go thinking-suffix model lookup", () => { + it("preserves Responses routing for gpt-5.6-luna thinking variants", () => { + expect(getModelSupportedFormats("opencode-go", "gpt-5.6-luna(high)")).toEqual(["openai-responses"]); + expect(getModelTargetFormat("opencode-go", "gpt-5.6-luna(high)")).toBe("openai-responses"); + }); + + it("preserves Responses routing for grok-4.6 thinking variants", () => { + expect(getModelSupportedFormats("opencode-go", "grok-4.6(high)")).toEqual(["openai-responses"]); + expect(getModelTargetFormat("opencode-go", "grok-4.6(high)")).toBe("openai-responses"); + }); +}); + describe("OpenCode Go per-model supportedFormats", () => { it("declares [openai, claude] for MiniMax + Qwen models", () => { for (const m of CLAUDE_CAPABLE) { diff --git a/tests/unit/opencode-go-muse-spark-responses.test.js b/tests/unit/opencode-go-muse-spark-responses.test.js index 4f7dcb3d..cbe5c349 100644 --- a/tests/unit/opencode-go-muse-spark-responses.test.js +++ b/tests/unit/opencode-go-muse-spark-responses.test.js @@ -48,6 +48,22 @@ describe("ocg/muse-spark-1.3-contributor catalog", () => { }); describe("OpenCodeGoExecutor routing + sanitization", () => { + it("routes gpt-5.6-luna to /responses", () => { + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl("gpt-5.6-luna")).toBe("https://opencode.ai/zen/go/v1/responses"); + expect(ex.buildUrl("gpt-5.6-luna(high)", true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions" }, + })).toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("routes every responses-only registry model (grok-4.6) to /responses", () => { + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl("grok-4.6")).toBe("https://opencode.ai/zen/go/v1/responses"); + expect(ex.buildUrl("grok-4.6(high)", true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions" }, + })).toBe("https://opencode.ai/zen/go/v1/responses"); + }); + it("is wired for opencode-go and routes muse-spark to /responses", () => { expect(getExecutor("opencode-go")).toBeInstanceOf(OpenCodeGoExecutor); const ex = new OpenCodeGoExecutor(); From 2b65c49ff5a9aa96c20287bf02ec13047d4bea35 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Thu, 17 Sep 2026 18:01:33 +0700 Subject: [PATCH 53/78] fix(opencode): route Union Alpha through Messages API Route union-alpha to /zen/v1/messages with targetFormat claude, add anthropic-version header, and register model capabilities (vision, 262K context, 131K max output). --- open-sse/executors/opencode.js | 19 +++++++---- open-sse/providers/capabilities.js | 2 ++ open-sse/providers/registry/opencode.js | 4 +-- .../unit/opencode-muse-spark-thinking.test.js | 33 +++++++++++++++++++ 4 files changed, 50 insertions(+), 8 deletions(-) diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index e2e36449..fe77137c 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -5,13 +5,14 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js"; import { injectReasoningContent } from "../utils/reasoningContentInjector.js"; import { resolveSessionId } from "../utils/sessionManager.js"; import { isMuseSparkModel } from "../providers/models/helpers.js"; +import { ANTHROPIC_API_VERSION } from "../providers/shared.js"; const OPENCODE_UA = "opencode"; -// Models served by /zen/v1/responses; every other model stays on /chat/completions. const RESPONSES_MODELS = new Set([ "muse-spark-1.2-contributor-free", "muse-spark-1.3-contributor-free", ]); +const MESSAGES_MODELS = new Set(["union-alpha"]); function generateRequestId() { return `msg_${crypto.randomUUID().replace(/-/g, "")}`; @@ -31,6 +32,10 @@ function isResponsesModel(model) { return RESPONSES_MODELS.has(base) || isMuseSparkModel(base); } +function isMessagesModel(model) { + return MESSAGES_MODELS.has(baseModelId(model)); +} + function resolveOpencodeSession(body, credentials) { const headers = credentials?.rawHeaders || {}; return resolveSessionId({ @@ -89,12 +94,12 @@ export class OpenCodeExecutor extends BaseExecutor { buildUrl(model) { const base = this.config.baseUrl; - return isResponsesModel(model) - ? `${base}/zen/v1/responses` - : `${base}/zen/v1/chat/completions`; + if (isResponsesModel(model)) return `${base}/zen/v1/responses`; + if (isMessagesModel(model)) return `${base}/zen/v1/messages`; + return `${base}/zen/v1/chat/completions`; } - buildHeaders(credentials, stream = true) { + buildHeaders(credentials, stream = true, url = "") { const raw = credentials?.rawHeaders || {}; const lower = {}; for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v; @@ -102,7 +107,7 @@ export class OpenCodeExecutor extends BaseExecutor { const downstreamUa = lower["user-agent"] || ""; const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode"); - return { + const headers = { "Content-Type": "application/json", "Authorization": "Bearer public", "User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA, @@ -112,5 +117,7 @@ export class OpenCodeExecutor extends BaseExecutor { "x-opencode-project": lower["x-opencode-project"] || "global", "Accept": stream ? "text/event-stream" : "*/*", }; + if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION; + return headers; } } diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index eac9089e..12c9d82f 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -141,6 +141,8 @@ export const MODEL_CAPABILITIES = { // via OpenAI Responses input_image; reasoning supports up to xhigh. "muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 }, "muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 }, + // OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output + "union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 }, }; const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; diff --git a/open-sse/providers/registry/opencode.js b/open-sse/providers/registry/opencode.js index 03e72eef..fc2ab8c6 100644 --- a/open-sse/providers/registry/opencode.js +++ b/open-sse/providers/registry/opencode.js @@ -20,10 +20,10 @@ export default { noAuth: true, }, models: [ - // Muse Spark models are served by /zen/v1/responses; the rest stay on - // /chat/completions, so the format is declared per-model, not per-provider. + // Endpoint formats differ per model, so declare non-chat models explicitly. { id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" }, { id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" }, + { id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" }, ], modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" }, passthroughModels: true, diff --git a/tests/unit/opencode-muse-spark-thinking.test.js b/tests/unit/opencode-muse-spark-thinking.test.js index 800f7d4e..9ffc29b1 100644 --- a/tests/unit/opencode-muse-spark-thinking.test.js +++ b/tests/unit/opencode-muse-spark-thinking.test.js @@ -63,6 +63,39 @@ describe("OpenCode Free Muse Spark thinking", () => { expect(out.max_tokens).toBeUndefined(); }); + it("routes Union Alpha through Anthropic Messages", () => { + const caps = getCapabilitiesForModel(PROVIDER, "union-alpha"); + expect(caps.vision).toBe(true); + expect(caps.contextWindow).toBe(262144); + expect(caps.maxOutput).toBe(131072); + + const executor = new OpenCodeExecutor(); + + expect(getModelTargetFormat("oc", "union-alpha")).toBe(FORMATS.CLAUDE); + const url = executor.buildUrl("union-alpha"); + expect(url).toBe("https://opencode.ai/zen/v1/messages"); + expect(executor.buildHeaders({}, true, url)).toMatchObject({ + "anthropic-version": "2023-06-01", + }); + expect(executor.buildHeaders({}, true, executor.buildUrl("big-pickle"))) + .not.toHaveProperty("anthropic-version"); + + const translated = translateRequest( + FORMATS.OPENAI, + FORMATS.CLAUDE, + "union-alpha", + { messages: [{ role: "user", content: "ping" }], max_tokens: 1 }, + false, + {}, + PROVIDER, + ); + expect(translated).toMatchObject({ + model: "union-alpha", + messages: [{ role: "user", content: [{ type: "text", text: "ping" }] }], + max_tokens: 1, + }); + }); + it("leaves the other free models on Chat Completions", () => { const executor = new OpenCodeExecutor(); const body = { messages: [{ role: "user", content: "hi" }], max_tokens: 1024 }; From 6091ff597e61e83c7c06e7dbc25968809a78c891 Mon Sep 17 00:00:00 2001 From: anojndr Date: Thu, 17 Sep 2026 18:02:31 +0700 Subject: [PATCH 54/78] fix(opencode): resolve 403 FreeTierError with canonical session format and valid User-Agent OpenCode upstream validates free-tier requests: User-Agent must be opencode/ (>= 1.17.0) and x-opencode-session must match canonical ses_ format. Default OPENCODE_UA to opencode/1.18.31, generate canonical descending session IDs, provide deterministic foreign session translation, and isolate credentials per-request. --- open-sse/executors/opencode.js | 134 ++++++++++++++++-- tests/unit/opencode-session.test.js | 202 ++++++++++++++++++++++++++++ 2 files changed, 323 insertions(+), 13 deletions(-) create mode 100644 tests/unit/opencode-session.test.js diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index fe77137c..ba6a16a0 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -7,19 +7,100 @@ import { resolveSessionId } from "../utils/sessionManager.js"; import { isMuseSparkModel } from "../providers/models/helpers.js"; import { ANTHROPIC_API_VERSION } from "../providers/shared.js"; -const OPENCODE_UA = "opencode"; +const OPENCODE_UA = "opencode/1.18.31"; +const MAX_SESSION_LENGTH = 256; +const SESSION_HEADER = "x-opencode-session"; +const SESSION_FIELD = "_opencodeSession"; +export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/; +const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"; + +function hasValidOpencodeVersion(ua) { + const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i); + if (!m) return false; + const major = parseInt(m[1], 10); + const minor = parseInt(m[2], 10); + return major > 1 || (major === 1 && minor >= 17); +} + +// Models served by /zen/v1/responses; every other model stays on /chat/completions. const RESPONSES_MODELS = new Set([ "muse-spark-1.2-contributor-free", "muse-spark-1.3-contributor-free", ]); const MESSAGES_MODELS = new Set(["union-alpha"]); -function generateRequestId() { - return `msg_${crypto.randomUUID().replace(/-/g, "")}`; +let lastTimestamp = 0; +let counter = 0; + +function unstableRandom() { + const bytes = crypto.randomBytes(14); + let randomPart = ""; + for (let i = 0; i < 14; i++) { + randomPart += BASE62_CHARS[bytes[i] % 62]; + } + return randomPart; } -function generateSessionId() { - return `ses_${crypto.randomUUID().replace(/-/g, "")}`; +export function generateSessionId(timestamp = Date.now()) { + if (timestamp !== lastTimestamp) { + lastTimestamp = timestamp; + counter = 0; + } + counter++; + + const current = BigInt(timestamp) * 0x1000n + BigInt(counter); + const value = ~current; + const time = Array.from({ length: 6 }, (_, index) => + Number((value >> BigInt(40 - 8 * index)) & 0xffn) + .toString(16) + .padStart(2, "0") + ).join(""); + return `ses_${time}${unstableRandom()}`; +} + +export function generateRequestId(timestamp = Date.now()) { + const current = BigInt(timestamp) * 0x1000n + 1n; + const value = current; + const time = Array.from({ length: 6 }, (_, index) => + Number((value >> BigInt(40 - 8 * index)) & 0xffn) + .toString(16) + .padStart(2, "0") + ).join(""); + return `msg_${time}${unstableRandom()}`; +} + +export function translateSessionId(sessionId, clientTool = "") { + if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) { + return sessionId.trim(); + } + const digest = crypto + .createHash("sha256") + .update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`) + .digest(); + const timeHex = digest.subarray(0, 6).toString("hex"); + let randomPart = ""; + for (let i = 6; i < 20; i++) { + randomPart += BASE62_CHARS[digest[i] % 62]; + } + return `ses_${timeHex}${randomPart}`; +} + +function normalizeSession(value) { + if (typeof value !== "string") return null; + const normalized = value.trim(); + if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null; + return normalized; +} + +function nativeSession(headers) { + if (!headers || typeof headers !== "object") return null; + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase() === SESSION_HEADER) { + const normalized = normalizeSession(value); + if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized; + } + } + return null; } // Strip the thinking suffix "model(level)" so registry lookups hit the base id. @@ -36,15 +117,27 @@ function isMessagesModel(model) { return MESSAGES_MODELS.has(baseModelId(model)); } -function resolveOpencodeSession(body, credentials) { +function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) { const headers = credentials?.rawHeaders || {}; - return resolveSessionId({ + const native = nativeSession(headers); + if (native) return native; + + let incoming = null; + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase() === SESSION_HEADER) { + incoming = normalizeSession(value); + break; + } + } + + const resolved = incoming || normalizeSession(providerSessionId) || resolveSessionId({ headers, body, connectionId: credentials?.connectionId, scope: "opencode", - generate: generateSessionId, }); + + return resolved ? translateSessionId(resolved, clientTool) : generateSessionId(); } function normalizeOpencodeReasoning(model, body) { @@ -73,12 +166,21 @@ function normalizeOpencodeReasoning(model, body) { export class OpenCodeExecutor extends BaseExecutor { constructor() { super("opencode", PROVIDERS.opencode); - this._currentSessionId = null; + } + + prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) { + const sourceCredentials = credentials || {}; + const resolved = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool); + + return { + ...sourceCredentials, + [SESSION_FIELD]: resolved, + }; } transformRequest(model, body, stream, credentials) { - this._currentSessionId = resolveOpencodeSession(body, credentials); - if (isResponsesModel(model)) { + if (body && typeof body === "object" && model && !body.model) body.model = model; + if (isResponsesModel(model) && body && typeof body === "object") { // Responses API names the output cap max_output_tokens and takes thinking // as reasoning:{effort,summary} — normalize the Chat fields at this boundary. if (body.max_output_tokens === undefined) { @@ -92,6 +194,10 @@ export class OpenCodeExecutor extends BaseExecutor { return injectReasoningContent({ provider: this.provider, model, body }); } + async execute(args) { + return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) }); + } + buildUrl(model) { const base = this.config.baseUrl; if (isResponsesModel(model)) return `${base}/zen/v1/responses`; @@ -105,14 +211,16 @@ export class OpenCodeExecutor extends BaseExecutor { for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v; const downstreamUa = lower["user-agent"] || ""; - const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode"); + const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa); + + const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD]; const headers = { "Content-Type": "application/json", "Authorization": "Bearer public", "User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA, "x-opencode-client": lower["x-opencode-client"] || "desktop", - "x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(), + "x-opencode-session": session, "x-opencode-request": lower["x-opencode-request"] || generateRequestId(), "x-opencode-project": lower["x-opencode-project"] || "global", "Accept": stream ? "text/event-stream" : "*/*", diff --git a/tests/unit/opencode-session.test.js b/tests/unit/opencode-session.test.js new file mode 100644 index 00000000..4fef80fe --- /dev/null +++ b/tests/unit/opencode-session.test.js @@ -0,0 +1,202 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const { fetchMock } = vi.hoisted(() => ({ + fetchMock: vi.fn(), +})); + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: fetchMock, +})); + +import { getExecutor } from "../../open-sse/executors/index.js"; +import { + OPENCODE_SESSION_RE, + generateSessionId, + generateRequestId, + translateSessionId, +} from "../../open-sse/executors/opencode.js"; + +function makeCredentials(overrides = {}) { + return { + connectionId: "conn_test", + rawHeaders: {}, + ...overrides, + }; +} + +function prepare(executor, overrides = {}) { + const credentials = overrides.credentials || makeCredentials(); + const prepared = executor.prepareRequestCredentials({ + body: overrides.body || { input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }] }, + credentials, + providerSessionId: overrides.providerSessionId ?? "conversation-a", + clientTool: overrides.clientTool ?? "claude", + }); + return { credentials, prepared }; +} + +beforeEach(() => { + fetchMock.mockReset(); + fetchMock.mockResolvedValue(new Response("{}", { + status: 200, + headers: { "content-type": "application/json" }, + })); +}); + +describe("OpenCode Free Session ID Format", () => { + it("generates session IDs matching OpenCode canonical format (ses_ + 12 hex + 14 base62)", () => { + for (let i = 0; i < 20; i++) { + const id = generateSessionId(); + expect(id).toMatch(OPENCODE_SESSION_RE); + expect(id).toHaveLength(30); + } + }); + + it("generates request IDs matching OpenCode canonical format (msg_ + 12 hex + 14 base62)", () => { + for (let i = 0; i < 20; i++) { + const id = generateRequestId(); + expect(id).toMatch(/^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/); + expect(id).toHaveLength(30); + } + }); + + it("translates arbitrary sessions into valid OpenCode session format", () => { + const inputs = [ + "claude:550e8400-e29b-41d4-a716-446655440000", + "antigravity:conv-abc-123", + "session-from-codex", + "12345", + "", + ]; + for (const raw of inputs) { + const translated = translateSessionId(raw, "claude"); + expect(translated).toMatch(OPENCODE_SESSION_RE); + expect(translated).toHaveLength(30); + } + }); + + it("preserves already-valid OpenCode sessions without re-hashing", () => { + const valid = "ses_f534dfae8ffeCy4Ee4tLWNygDc"; + expect(translateSessionId(valid)).toBe(valid); + expect(translateSessionId(` ${valid} `)).toBe(valid); + }); +}); + +describe("OpenCode Free Executor Session Resolution", () => { + it("uses request-local session credentials without mutating source credentials", () => { + const executor = getExecutor("opencode"); + const { credentials, prepared } = prepare(executor); + + expect(executor.constructor.name).toBe("OpenCodeExecutor"); + expect(prepared).not.toBe(credentials); + expect(prepared._opencodeSession).toMatch(OPENCODE_SESSION_RE); + expect(credentials).not.toHaveProperty("_opencodeSession"); + expect(executor).not.toHaveProperty("_currentSessionId"); + }); + + it("preserves valid native x-opencode-session header case-insensitively", () => { + const executor = getExecutor("opencode"); + const valid = "ses_f534dfae8ffeCy4Ee4tLWNygDc"; + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "X-OpenCode-Session": ` ${valid} ` } }), + }); + + expect(prepared._opencodeSession).toBe(valid); + }); + + it("translates invalid native x-opencode-session header into a valid session", () => { + const executor = getExecutor("opencode"); + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "x-opencode-session": "invalid-session-uuid" } }), + }); + + expect(prepared._opencodeSession).toMatch(OPENCODE_SESSION_RE); + expect(prepared._opencodeSession).not.toBe("invalid-session-uuid"); + }); + + it("translates conversation session deterministically", () => { + const executor = getExecutor("opencode"); + const first = prepare(executor, { providerSessionId: "conversation-a", clientTool: "claude" }).prepared._opencodeSession; + const second = prepare(executor, { providerSessionId: "conversation-a", clientTool: "claude" }).prepared._opencodeSession; + + expect(first).toBe(second); + expect(first).toMatch(OPENCODE_SESSION_RE); + }); + + it("isolates different conversations and tools", () => { + const executor = getExecutor("opencode"); + const convA = prepare(executor, { providerSessionId: "conversation-a" }).prepared._opencodeSession; + const convB = prepare(executor, { providerSessionId: "conversation-b" }).prepared._opencodeSession; + const toolClaude = prepare(executor, { providerSessionId: "same", clientTool: "claude" }).prepared._opencodeSession; + const toolCodex = prepare(executor, { providerSessionId: "same", clientTool: "codex" }).prepared._opencodeSession; + + expect(convA).not.toBe(convB); + expect(toolClaude).not.toBe(toolCodex); + }); + + it("adds the valid session header to fetch requests", async () => { + const executor = getExecutor("opencode"); + const credentials = makeCredentials(); + const result = await executor.execute({ + model: "muse-spark-1.3-contributor-free", + body: { input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }] }, + stream: false, + credentials, + providerSessionId: "conversation-fetch-test", + clientTool: "claude", + }); + + expect(result.headers["x-opencode-session"]).toMatch(OPENCODE_SESSION_RE); + expect(fetchMock).toHaveBeenCalledOnce(); + expect(fetchMock.mock.calls[0][1].headers["x-opencode-session"]).toBe(result.headers["x-opencode-session"]); + expect(fetchMock.mock.calls[0][1].headers["Authorization"]).toBe("Bearer public"); + expect(credentials).not.toHaveProperty("_opencodeSession"); + }); + + it("falls back to a valid generated session in buildHeaders when called standalone", () => { + const executor = getExecutor("opencode"); + const headers = executor.buildHeaders({}); + + expect(headers["x-opencode-session"]).toMatch(OPENCODE_SESSION_RE); + expect(headers["Authorization"]).toBe("Bearer public"); + }); + it("handles null or undefined body gracefully in transformRequest", () => { + const executor = getExecutor("opencode"); + expect(() => executor.transformRequest("muse-spark-1.3-contributor-free", null, false, {})).not.toThrow(); + expect(() => executor.transformRequest("big-pickle", undefined, false, {})).not.toThrow(); + }); +}); + +describe("OpenCode Free User-Agent Validation", () => { + it("defaults User-Agent to opencode/1.18.31 for non-opencode downstream clients", () => { + const executor = getExecutor("opencode"); + const headersNoUa = executor.buildHeaders({}); + expect(headersNoUa["User-Agent"]).toBe("opencode/1.18.31"); + + const headersClaude = executor.buildHeaders({ rawHeaders: { "user-agent": "Claude-Code/1.0" } }); + expect(headersClaude["User-Agent"]).toBe("opencode/1.18.31"); + }); + + it("replaces bare opencode with versioned opencode/1.18.31 to prevent 403 FreeTierError", () => { + const executor = getExecutor("opencode"); + const headers = executor.buildHeaders({ rawHeaders: { "user-agent": "opencode" } }); + expect(headers["User-Agent"]).toBe("opencode/1.18.31"); + }); + + it("upgrades outdated opencode versions (< 1.17) to prevent 426 Upgrade Required", () => { + const executor = getExecutor("opencode"); + const headers = executor.buildHeaders({ rawHeaders: { "user-agent": "opencode/1.15.0" } }); + expect(headers["User-Agent"]).toBe("opencode/1.18.31"); + }); + + it("preserves valid opencode versions (>= 1.17)", () => { + const executor = getExecutor("opencode"); + const headers118 = executor.buildHeaders({ + rawHeaders: { "user-agent": "opencode/1.18.31 ai-sdk/provider-utils/4.0.40 runtime/bun/1.3.14" }, + }); + expect(headers118["User-Agent"]).toBe("opencode/1.18.31 ai-sdk/provider-utils/4.0.40 runtime/bun/1.3.14"); + + const headersFuture = executor.buildHeaders({ rawHeaders: { "user-agent": "opencode/1.19.0" } }); + expect(headersFuture["User-Agent"]).toBe("opencode/1.19.0"); + }); +}); From 0c6ab4f99b69ecfadcd255f6f2bd793919ab6ff2 Mon Sep 17 00:00:00 2001 From: ErfanBagheri404 Date: Thu, 17 Sep 2026 13:47:12 +0330 Subject: [PATCH 55/78] fix(opencode): reuse one stable upstream session per identity to stop 429s Follow-up to the canonical-session fix: with no explicit session, every request minted a fresh x-opencode-session, and upstream free-tier quota is accounted per session. That burns through quota and surfaces as 429 FreeUsageLimitError with growing reset-after delays, while the real CLI reuses one long-lived session per conversation. - Stable canonical session per downstream identity (connectionId, else auth-header hash, else shared default), evicted after MEMORY_CONFIG.sessionTtlMs like the other session stores. - Deterministic x-opencode-request per message (stable across retries, like the CLI user message id); valid downstream ids preserved. - 6 more unit tests (22 total). --- open-sse/executors/opencode.js | 183 ++++++++++++++++++++++++++-- tests/unit/opencode-session.test.js | 77 ++++++++++++ 2 files changed, 250 insertions(+), 10 deletions(-) diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index ba6a16a0..c5d0caa7 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -1,6 +1,7 @@ import crypto from "crypto"; import { BaseExecutor } from "./base.js"; import { PROVIDERS } from "../config/providers.js"; +import { MEMORY_CONFIG } from "../config/runtimeConfig.js"; import { getThinkingLevels } from "../providers/thinkingLevels.js"; import { injectReasoningContent } from "../utils/reasoningContentInjector.js"; import { resolveSessionId } from "../utils/sessionManager.js"; @@ -11,7 +12,9 @@ const OPENCODE_UA = "opencode/1.18.31"; const MAX_SESSION_LENGTH = 256; const SESSION_HEADER = "x-opencode-session"; const SESSION_FIELD = "_opencodeSession"; +const REQ_FIELD = "_opencodeRequest"; export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/; +export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/; const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"; function hasValidOpencodeVersion(ua) { @@ -103,6 +106,140 @@ function nativeSession(headers) { return null; } +// Upstream free-tier quota is accounted per session. Minting a fresh +// x-opencode-session on every request burns through it and surfaces as +// 429 FreeUsageLimitError with growing reset-after delays, while the real +// CLI reuses one long-lived canonical session per conversation. Mirror +// that: one stable canonical session per downstream identity, evicted +// after MEMORY_CONFIG.sessionTtlMs like the other session stores. +const stableOpencodeSessions = new Map(); +const MAX_STABLE_SESSIONS = 1000; +const stableSessionCleanup = setInterval(() => { + const now = Date.now(); + for (const [key, entry] of stableOpencodeSessions) { + if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) { + stableOpencodeSessions.delete(key); + } + } +}, MEMORY_CONFIG.sessionCleanupIntervalMs); +if (stableSessionCleanup.unref) stableSessionCleanup.unref(); + +function identityKey(credentials) { + const connectionId = credentials?.connectionId || credentials?.id; + if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`; + const raw = credentials?.rawHeaders || {}; + const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || ""; + if (auth) { + const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32); + return `opencode:auth:${digest}`; + } + return "opencode:default"; +} + +export function stableSessionId(credentials) { + const key = identityKey(credentials); + const existing = stableOpencodeSessions.get(key); + if (existing) { + existing.lastUsed = Date.now(); + stableOpencodeSessions.delete(key); + stableOpencodeSessions.set(key, existing); + return existing.sessionId; + } + const sessionId = generateSessionId(); + if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) { + stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value); + } + stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() }); + return sessionId; +} + +function lastUserText(body) { + try { + if (!body || typeof body !== "object") return ""; + const arr = Array.isArray(body.messages) + ? body.messages + : Array.isArray(body.input) + ? body.input + : null; + if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : ""; + for (let i = arr.length - 1; i >= 0; i--) { + const msg = arr[i]; + if (!msg) continue; + if (msg.role && msg.role !== "user") continue; + const content = msg.content; + if (typeof content === "string" && content.trim()) return content.trim().slice(-600); + if (Array.isArray(content)) { + const text = content + .map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || "")) + .join(" ") + .trim(); + if (text) return text.slice(-600); + } + } + } catch { + return ""; + } + return ""; +} + +// The real CLI sends the current user message id (stable per turn, same on +// retries) as x-opencode-request. Derive it deterministically from the +// session plus the last user message so retries share the id. +export function deriveRequestId(sessionId, body) { + const text = lastUserText(body); + if (!text) return generateRequestId(); + const digest = crypto + .createHash("sha256") + .update(`opencode-req\0${sessionId || ""}\0${text}`) + .digest(); + const timeHex = digest.subarray(0, 6).toString("hex"); + let randomPart = ""; + for (let i = 6; i < 20; i++) { + randomPart += BASE62_CHARS[digest[i] % 62]; + } + const id = `msg_${timeHex}${randomPart}`; + return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId(); +} + +function normalizeRequestId(value) { + if (typeof value !== "string") return null; + const normalized = value.trim(); + if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null; + return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null; +} + +function bodyHasSessionHints(body) { + try { + if (!body || typeof body !== "object") return false; + if (typeof body.session_id === "string" && body.session_id.trim()) return true; + if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true; + if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true; + if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true; + if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true; + const arr = Array.isArray(body.messages) + ? body.messages + : Array.isArray(body.input) + ? body.input + : null; + if (arr) { + let assistantText = ""; + for (const msg of arr) { + if (msg?.role === "assistant") { + const content = msg.content; + if (typeof content === "string") assistantText += content; + else if (Array.isArray(content)) { + for (const part of content) assistantText += part?.text || part?.output || ""; + } + if (assistantText.length >= 50) return true; + } + } + } + return false; + } catch { + return false; + } +} + // Strip the thinking suffix "model(level)" so registry lookups hit the base id. function baseModelId(model) { return String(model || "").replace(/\([^()]+\)\s*$/, "").trim(); @@ -130,14 +267,37 @@ function resolveOpencodeSession(body, credentials, providerSessionId, clientTool } } - const resolved = incoming || normalizeSession(providerSessionId) || resolveSessionId({ - headers, - body, - connectionId: credentials?.connectionId, - scope: "opencode", - }); + const hinted = incoming || normalizeSession(providerSessionId); + if (hinted) return translateSessionId(hinted, clientTool); - return resolved ? translateSessionId(resolved, clientTool) : generateSessionId(); + if (credentials?.connectionId || bodyHasSessionHints(body)) { + let viaManager = null; + try { + viaManager = resolveSessionId({ + headers, + body, + connectionId: credentials?.connectionId, + scope: "opencode", + }); + } catch { + viaManager = null; + } + if (viaManager) return translateSessionId(viaManager, clientTool); + } + + return stableSessionId(credentials); +} + +function resolveOpencodeRequestId(body, credentials, sessionId) { + const raw = credentials?.rawHeaders || {}; + for (const [key, value] of Object.entries(raw)) { + if (key.toLowerCase() === "x-opencode-request") { + const normalized = normalizeRequestId(value); + if (normalized) return normalized; + break; + } + } + return deriveRequestId(sessionId, body); } function normalizeOpencodeReasoning(model, body) { @@ -170,11 +330,12 @@ export class OpenCodeExecutor extends BaseExecutor { prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) { const sourceCredentials = credentials || {}; - const resolved = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool); + const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool); return { ...sourceCredentials, - [SESSION_FIELD]: resolved, + [SESSION_FIELD]: session, + [REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session), }; } @@ -214,6 +375,8 @@ export class OpenCodeExecutor extends BaseExecutor { const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa); const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD]; + const downstreamReq = normalizeRequestId(lower["x-opencode-request"]); + const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId(); const headers = { "Content-Type": "application/json", @@ -221,7 +384,7 @@ export class OpenCodeExecutor extends BaseExecutor { "User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA, "x-opencode-client": lower["x-opencode-client"] || "desktop", "x-opencode-session": session, - "x-opencode-request": lower["x-opencode-request"] || generateRequestId(), + "x-opencode-request": requestId, "x-opencode-project": lower["x-opencode-project"] || "global", "Accept": stream ? "text/event-stream" : "*/*", }; diff --git a/tests/unit/opencode-session.test.js b/tests/unit/opencode-session.test.js index 4fef80fe..b7651f16 100644 --- a/tests/unit/opencode-session.test.js +++ b/tests/unit/opencode-session.test.js @@ -11,9 +11,12 @@ vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ import { getExecutor } from "../../open-sse/executors/index.js"; import { OPENCODE_SESSION_RE, + OPENCODE_REQUEST_RE, generateSessionId, generateRequestId, translateSessionId, + stableSessionId, + deriveRequestId, } from "../../open-sse/executors/opencode.js"; function makeCredentials(overrides = {}) { @@ -200,3 +203,77 @@ describe("OpenCode Free User-Agent Validation", () => { expect(headersFuture["User-Agent"]).toBe("opencode/1.19.0"); }); }); + +describe("OpenCode Stable Session Reuse (429 follow-up)", () => { + function anonymousCredentials(auth) { + return makeCredentials({ connectionId: undefined, rawHeaders: { authorization: `Bearer ${auth}` } }); + } + + it("reuses one stable upstream session instead of minting a new one per request", () => { + const executor = getExecutor("opencode"); + const body = { messages: [{ role: "user", content: "hello" }] }; + const first = executor.prepareRequestCredentials({ + body, + credentials: anonymousCredentials("stable-key-1"), + providerSessionId: null, + clientTool: "claude", + }); + const second = executor.prepareRequestCredentials({ + body, + credentials: anonymousCredentials("stable-key-1"), + providerSessionId: null, + clientTool: "claude", + }); + + expect(first._opencodeSession).toMatch(OPENCODE_SESSION_RE); + expect(second._opencodeSession).toBe(first._opencodeSession); + }); + + it("isolates stable sessions by downstream identity", () => { + const executor = getExecutor("opencode"); + const body = { messages: [{ role: "user", content: "hello" }] }; + const forKey = (auth) => executor.prepareRequestCredentials({ + body, + credentials: anonymousCredentials(auth), + providerSessionId: null, + clientTool: "claude", + })._opencodeSession; + + expect(forKey("user-A")).not.toBe(forKey("user-B")); + expect(forKey("user-A")).toMatch(OPENCODE_SESSION_RE); + }); + + it("exposes the stable session helper directly", () => { + const first = stableSessionId({ connectionId: "direct-conn" }); + expect(stableSessionId({ connectionId: "direct-conn" })).toBe(first); + expect(first).toMatch(OPENCODE_SESSION_RE); + }); + + it("derives deterministic, canonical request ids per message", () => { + const session = stableSessionId({ connectionId: "req-conn" }); + const body = { messages: [{ role: "user", content: "ping" }] }; + const first = deriveRequestId(session, body); + expect(first).toMatch(OPENCODE_REQUEST_RE); + expect(deriveRequestId(session, body)).toBe(first); + expect( + deriveRequestId(session, { messages: [{ role: "user", content: "a different question" }] }), + ).not.toBe(first); + }); + + it("preserves a valid downstream x-opencode-request header", () => { + const executor = getExecutor("opencode"); + const validReq = "msg_0ae8d9cd3001swxaFbM248jcIF"; + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "x-opencode-request": validReq } }), + }); + expect(prepared._opencodeRequest).toBe(validReq); + }); + + it("keeps the standalone buildHeaders session stable across calls", () => { + const executor = getExecutor("opencode"); + const first = executor.buildHeaders({})["x-opencode-session"]; + const second = executor.buildHeaders({})["x-opencode-session"]; + expect(first).toMatch(OPENCODE_SESSION_RE); + expect(second).toBe(first); + }); +}); From f4f06f290c054f55c3cdb598e8ef5d58f5b8043d Mon Sep 17 00:00:00 2001 From: Manan Santoki Date: Thu, 17 Sep 2026 18:11:20 +0700 Subject: [PATCH 56/78] fix(translator): keep tool-result images, restore Kiro tool names, preserve thinking display Forward images inside tool_result to OpenAI and Kiro upstreams via following user messages, restore original client tool names on Kiro responses via _toolNameMap, and preserve thinking display settings across translations. --- open-sse/executors/base.js | 2 +- open-sse/executors/default.js | 4 +- open-sse/providers/shared.js | 13 +- .../translator/concerns/thinkingUnified.js | 11 +- open-sse/translator/formats/claude.js | 30 +++ open-sse/translator/request/claude-to-kiro.js | 19 +- .../translator/request/claude-to-openai.js | 28 ++- .../request/openai-to-commandcode.js | 6 +- open-sse/translator/request/openai-to-kiro.js | 7 + .../translator/response/kiro-to-claude.js | 14 +- .../translator/response/kiro-to-openai.js | 31 ++- tests/translator/agent-client-fixes.test.js | 186 ++++++++++++++++++ 12 files changed, 328 insertions(+), 23 deletions(-) create mode 100644 tests/translator/agent-client-fixes.test.js diff --git a/open-sse/executors/base.js b/open-sse/executors/base.js index a4c017e4..18a62229 100644 --- a/open-sse/executors/base.js +++ b/open-sse/executors/base.js @@ -127,7 +127,7 @@ export class BaseExecutor { for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) { const url = this.buildUrl(model, stream, urlIndex, credentials); const transformedBody = this.transformRequest(model, body, stream, credentials); - const headers = this.buildHeaders(credentials, stream, url, model); + const headers = this.buildHeaders(credentials, stream, url, model, transformedBody); if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0; diff --git a/open-sse/executors/default.js b/open-sse/executors/default.js index 00f3e3de..64ad46d0 100644 --- a/open-sse/executors/default.js +++ b/open-sse/executors/default.js @@ -146,7 +146,7 @@ export class DefaultExecutor extends BaseExecutor { return BEARER; } - buildHeaders(credentials, stream = true, url, model) { + buildHeaders(credentials, stream = true, url, model, body = null) { const rt = credentials?.runtimeTransport; const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) }; const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor(); @@ -166,7 +166,7 @@ export class DefaultExecutor extends BaseExecutor { const isClaudeModel = typeof model === "string" && /^claude-/.test(model); if (model && (this.provider === "claude" || (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) { - headers["Anthropic-Beta"] = selectAnthropicBeta(model); + headers["Anthropic-Beta"] = selectAnthropicBeta(model, body); } // Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams diff --git a/open-sse/providers/shared.js b/open-sse/providers/shared.js index 88488499..c0699c6f 100644 --- a/open-sse/providers/shared.js +++ b/open-sse/providers/shared.js @@ -62,8 +62,17 @@ const ANTHROPIC_BETA_BASE = [ const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"]; // Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them. -export function selectAnthropicBeta(model = "") { - const flags = [...ANTHROPIC_BETA_BASE]; +// `redact-thinking` asks Anthropic to return signature-only thinking blocks, which +// is right for clients that never render thinking but blanks the summaries a +// client explicitly requested with `thinking.display: "summarized"`. +const ANTHROPIC_BETA_REDACT_THINKING = "redact-thinking-2026-02-12"; + +export function wantsThinkingSummaries(body) { + return body?.thinking?.display === "summarized"; +} + +export function selectAnthropicBeta(model = "", body = null) { + const flags = ANTHROPIC_BETA_BASE.filter((flag) => flag !== ANTHROPIC_BETA_REDACT_THINKING || !wantsThinkingSummaries(body)); if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT); return flags.join(","); } diff --git a/open-sse/translator/concerns/thinkingUnified.js b/open-sse/translator/concerns/thinkingUnified.js index 001259e3..4bc9601f 100644 --- a/open-sse/translator/concerns/thinkingUnified.js +++ b/open-sse/translator/concerns/thinkingUnified.js @@ -232,7 +232,7 @@ function stripAll(body) { } // Apply unified thinking config to body in the resolved provider-native format. -function applyFormat(fmt, body, cfg, caps, supportedLevels) { +function applyFormat(fmt, body, cfg, caps, supportedLevels, display) { const none = cfg.mode === "none"; const canDisable = caps.thinkingCanDisable !== false; // Model cannot disable thinking → clamp "none" to minimal effort instead. @@ -249,7 +249,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) { if (none && canDisable) { body.thinking = { type: "disabled" }; break; } // Models that can disable thinking need the explicit adaptive switch. // Permanently adaptive models such as Fable 5.1 accept effort directly. - if (canDisable) body.thinking = { type: "adaptive" }; + if (canDisable) body.thinking = { type: "adaptive", ...(display ? { display } : {}) }; else delete body.thinking; const level = toLevel(eff); body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level }; @@ -258,7 +258,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) { case "claude-budget": { if (none && canDisable) { body.thinking = { type: "disabled" }; break; } const budget = toBudget(eff, caps.thinkingRange); - body.thinking = budget === -1 ? { type: "enabled" } : { type: "enabled", budget_tokens: budget || 8192 }; + body.thinking = budget === -1 ? { type: "enabled", ...(display ? { display } : {}) } : { type: "enabled", budget_tokens: budget || 8192, ...(display ? { display } : {}) }; break; } case "gemini-level": { @@ -378,7 +378,10 @@ export function applyThinking(targetFormat, model, body, provider = null, intent const fmt = resolveFormat(targetFormat, cleanModel, provider); const supportedLevels = getThinkingLevels(provider, cleanModel); + // Anthropic's `display` (summarized | omitted) decides whether thinking text + // comes back at all; keep what the client asked for instead of resetting it. + const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined; stripAll(body); - applyFormat(fmt, body, cfg, caps, supportedLevels); + applyFormat(fmt, body, cfg, caps, supportedLevels, display); return body; } diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index f57972da..14c9fc10 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -415,6 +415,28 @@ export function anchorClaudeCache(body) { // - Add thinking block for Anthropic endpoint (provider === "claude") // - Fix tool_use/tool_result ordering // - Apply cloaking (billing header + fake user ID) for OAuth tokens +export function hoistToolResultImages(body) { + if (!Array.isArray(body?.messages)) return body; + let touched = false; + const messages = body.messages.map((msg) => { + if (msg?.role !== ROLE.USER || !Array.isArray(msg.content)) return msg; + const hoisted = []; + const content = msg.content.map((block) => { + if (block?.type !== CLAUDE_BLOCK.TOOL_RESULT || !Array.isArray(block.content)) return block; + const images = block.content.filter((c) => c?.type === CLAUDE_BLOCK.IMAGE); + if (!images.length) return block; + const rest = block.content.filter((c) => c?.type !== CLAUDE_BLOCK.IMAGE); + hoisted.push({ type: CLAUDE_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` }, ...images); + return { ...block, content: rest.length ? rest : [{ type: CLAUDE_BLOCK.TEXT, text: "(image attached below)" }] }; + }); + if (!hoisted.length) return msg; + touched = true; + // tool_result blocks must lead a user message; the hoisted image follows them. + return { ...msg, content: [...content, ...hoisted] }; + }); + return touched ? { ...body, messages } : body; +} + export function prepareClaudeRequest(body, provider = null, apiKey = null, connectionId = null, rawHeaders = null, sessionId = null) { // quirk: MiniMax's Claude-compatible endpoint rejects Anthropic's output_config (400 invalid params) if (PROVIDERS[provider]?.quirks?.dropOutputConfig) { @@ -608,6 +630,14 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne } } + // Anthropic itself reads images inside tool_result; other Anthropic-compatible + // endpoints (OpenCode Go, Kimi, DeepSeek, GLM, MiniMax) accept image blocks + // only as user content and silently drop them inside a tool result. Move a + // tool's screenshot out of the result and into the same user turn. + if (provider !== "claude" && !provider?.startsWith("anthropic-compatible")) { + body = hoistToolResultImages(body); + } + // Apply cloaking for OAuth tokens (billing header + fake user ID) // session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) { diff --git a/open-sse/translator/request/claude-to-kiro.js b/open-sse/translator/request/claude-to-kiro.js index ef2dd6c5..ac7a704d 100644 --- a/open-sse/translator/request/claude-to-kiro.js +++ b/open-sse/translator/request/claude-to-kiro.js @@ -97,11 +97,21 @@ function convertClaudeMessagesToKiro(messages, model) { if (typeof block.content === "string") { resultContent = block.content; } else if (Array.isArray(block.content)) { + // Images a tool returned (screenshots) ride along as user images; + // Kiro tool results are text-only. + let hasImage = false; + for (const c of block.content) { + if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") { + hasImage = true; + const imageType = c.source.media_type || DEFAULT_IMAGE_MIME; + pendingImages.push({ format: imageType.split("/")[1] || imageType, source: { bytes: c.source.data } }); + } + } resultContent = block.content .filter((c) => c.type === CLAUDE_BLOCK.TEXT) .map((c) => c.text) - .join("\n") || JSON.stringify(block.content); + .join("\n") || (hasImage ? "(image attached)" : JSON.stringify(block.content)); } else if (block.content) { resultContent = JSON.stringify(block.content); } @@ -341,6 +351,13 @@ export function claudeToKiroRequest(model, body, stream, credentials) { enumerable: false, }); + // Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the + // reverse map so tool calls stream back under the client's own names. + const restoredToolNames = new Map(); + for (const [original, sanitized] of nameMap) { + if (original !== sanitized) restoredToolNames.set(sanitized, original); + } + if (restoredToolNames.size) payload._toolNameMap = restoredToolNames; return payload; } diff --git a/open-sse/translator/request/claude-to-openai.js b/open-sse/translator/request/claude-to-openai.js index 38976226..e75d89b8 100644 --- a/open-sse/translator/request/claude-to-openai.js +++ b/open-sse/translator/request/claude-to-openai.js @@ -196,25 +196,41 @@ function convertClaudeMessage(msg) { }); break; - case CLAUDE_BLOCK.TOOL_RESULT: + case CLAUDE_BLOCK.TOOL_RESULT: { let resultContent = ""; + const resultImages = []; if (typeof block.content === "string") { resultContent = block.content; } else if (Array.isArray(block.content)) { - resultContent = block.content - .filter(c => c.type === CLAUDE_BLOCK.TEXT) - .map(c => c.text) - .join("\n") || JSON.stringify(block.content); + for (const c of block.content) { + if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") { + resultImages.push({ + type: OPENAI_BLOCK.IMAGE_URL, + image_url: { url: encodeDataUri(c.source.media_type, c.source.data) } + }); + } + } + const textOnly = block.content.filter(c => c?.type === CLAUDE_BLOCK.TEXT); + resultContent = textOnly.map(c => c.text).join("\n") + || (resultImages.length ? "" : JSON.stringify(block.content)); } else if (block.content) { resultContent = JSON.stringify(block.content); } - + toolResults.push({ role: ROLE.TOOL, tool_call_id: block.tool_use_id, content: resultContent }); + // The OpenAI tool role is text-only, so a screenshot or any other image a + // tool returned would otherwise vanish. Hand it to the model in the user + // turn that follows the tool messages, tagged with the call it came from. + if (resultImages.length) { + parts.push({ type: OPENAI_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` }); + parts.push(...resultImages); + } break; + } } } diff --git a/open-sse/translator/request/openai-to-commandcode.js b/open-sse/translator/request/openai-to-commandcode.js index 194078ae..d0a55e43 100644 --- a/open-sse/translator/request/openai-to-commandcode.js +++ b/open-sse/translator/request/openai-to-commandcode.js @@ -42,16 +42,19 @@ function toNativeImageBlock(part) { type: OPENAI_BLOCK.IMAGE, image: encodeDataUri(parsed.mimeType, parsed.base64), mimeType: parsed.mimeType, + mediaType: parsed.mimeType, }; } if (part.type === OPENAI_BLOCK.IMAGE || part.type === CLAUDE_BLOCK.IMAGE) { if (typeof part.image === "string" && part.image.startsWith("data:")) { const parsed = parseDataUri(part.image); + const mime = part.mimeType || parsed?.mimeType || "image/png"; return { type: OPENAI_BLOCK.IMAGE, image: part.image, - mimeType: part.mimeType || parsed?.mimeType || "image/png", + mimeType: mime, + mediaType: mime, }; } const source = part.source; @@ -61,6 +64,7 @@ function toNativeImageBlock(part) { type: OPENAI_BLOCK.IMAGE, image: encodeDataUri(mime, source.data), mimeType: mime, + mediaType: mime, }; } } diff --git a/open-sse/translator/request/openai-to-kiro.js b/open-sse/translator/request/openai-to-kiro.js index dbaefbae..0c9f04fd 100644 --- a/open-sse/translator/request/openai-to-kiro.js +++ b/open-sse/translator/request/openai-to-kiro.js @@ -434,6 +434,13 @@ export function openaiToKiroRequest(model, body, stream, credentials) { enumerable: false }); + // Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the + // reverse map so tool calls stream back under the client's own names. + const restoredToolNames = new Map(); + for (const [original, sanitized] of nameMap) { + if (original !== sanitized) restoredToolNames.set(sanitized, original); + } + if (restoredToolNames.size) payload._toolNameMap = restoredToolNames; return payload; } diff --git a/open-sse/translator/response/kiro-to-claude.js b/open-sse/translator/response/kiro-to-claude.js index 455672b1..9118b9cb 100644 --- a/open-sse/translator/response/kiro-to-claude.js +++ b/open-sse/translator/response/kiro-to-claude.js @@ -46,6 +46,14 @@ function convertFinishReason(reason) { * Convert one OpenAI-format chunk (from KiroExecutor) into Claude SSE events. * Returns an array of Claude events, or null when the chunk yields nothing. */ +// Kiro only accepts sanitized tool names; the request translator leaves the +// reverse map on the stream state so calls come back under the client's names. +function restoreToolName(state, name) { + const raw = name || ""; + const map = state?.toolNameMap; + return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw; +} + export function kiroToClaudeResponse(chunk, state) { // KiroExecutor emits chat.completion.chunk objects; tolerate string chunks // by attempting a parse (defensive — the direct path is always objects). @@ -161,7 +169,7 @@ export function kiroToClaudeResponse(chunk, state) { const toolBlockIndex = state.nextBlockIndex++; state.toolCalls.set(idx, { id: tc.id, - name: tc.function?.name || "", + name: restoreToolName(state, tc.function?.name), blockIndex: toolBlockIndex, }); results.push({ @@ -170,7 +178,7 @@ export function kiroToClaudeResponse(chunk, state) { content_block: { type: "tool_use", id: tc.id, - name: tc.function?.name || "", + name: restoreToolName(state, tc.function?.name), input: {}, }, }); @@ -246,7 +254,7 @@ export function kiroToClaudeNonStreaming(data) { content.push({ type: "tool_use", id: tc.id || `toolu_${Date.now()}`, - name: tc.function?.name || "", + name: restoreToolName(state, tc.function?.name), input, }); } diff --git a/open-sse/translator/response/kiro-to-openai.js b/open-sse/translator/response/kiro-to-openai.js index 7059a851..8e713fd7 100644 --- a/open-sse/translator/response/kiro-to-openai.js +++ b/open-sse/translator/response/kiro-to-openai.js @@ -20,13 +20,38 @@ function chunkMeta(state) { * Parse Kiro SSE event and convert to OpenAI format * Kiro events: assistantResponseEvent, codeEvent, supplementaryWebLinksEvent, etc. */ +// Kiro only accepts sanitized tool names; the request translator leaves the +// reverse map on the stream state so calls come back under the client's names. +function restoreToolName(state, name) { + const raw = name || ""; + const map = state?.toolNameMap; + return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw; +} + export function kiroToOpenAIResponse(chunk, state) { if (!chunk) return null; - // If chunk is already in OpenAI format (from executor transform), return as-is + // If chunk is already in OpenAI format (from executor transform), return it + // with the client's tool names restored. if (chunk.object === "chat.completion.chunk" && chunk.choices) { - return chunk; + if (!state?.toolNameMap?.size) return chunk; + return { + ...chunk, + choices: chunk.choices.map((choice) => { + const calls = choice?.delta?.tool_calls; + if (!Array.isArray(calls)) return choice; + return { + ...choice, + delta: { + ...choice.delta, + tool_calls: calls.map((tc) => tc?.function?.name + ? { ...tc, function: { ...tc.function, name: restoreToolName(state, tc.function.name) } } + : tc), + }, + }; + }), + }; } // Handle string chunk (raw SSE data) @@ -109,7 +134,7 @@ export function kiroToOpenAIResponse(chunk, state) { state.hadToolUse = true; const toolUse = data.toolUseEvent || data; const toolCallId = toolUse.toolUseId || fallbackToolCallId(); - const toolName = toolUse.name || ""; + const toolName = restoreToolName(state, toolUse.name); const toolInput = toolUse.input || {}; const openaiChunk = buildChunk(chunkMeta(state), { diff --git a/tests/translator/agent-client-fixes.test.js b/tests/translator/agent-client-fixes.test.js new file mode 100644 index 00000000..3f4e7cd7 --- /dev/null +++ b/tests/translator/agent-client-fixes.test.js @@ -0,0 +1,186 @@ +// Fixes for agent clients (Claude Code) driving non-Anthropic upstreams: +// - tool-result images survive the Claude → OpenAI / Kiro request translation +// - Kiro tool calls stream back under the client's own (unsanitized) names +// - the client's thinking `display` is kept on Claude-format upstreams +import { describe, it, expect } from "vitest"; +import "./registerAll.js"; +import { translateRequest, translateResponse, initState } from "../../open-sse/translator/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { applyThinking } from "../../open-sse/translator/concerns/thinkingUnified.js"; +import { kiroToClaudeResponse } from "../../open-sse/translator/response/kiro-to-claude.js"; +import { kiroToOpenAIResponse } from "../../open-sse/translator/response/kiro-to-openai.js"; +import { selectAnthropicBeta } from "../../open-sse/providers/shared.js"; +import { hoistToolResultImages } from "../../open-sse/translator/formats/claude.js"; +import { openaiToCommandCodeRequest } from "../../open-sse/translator/request/openai-to-commandcode.js"; + +const PNG = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="; + +const screenshotTurn = (extraTools = []) => ({ + tools: [ + { name: "mcp__browser__computer", description: "browser", input_schema: { type: "object", properties: {} } }, + ...extraTools, + ], + messages: [ + { role: "user", content: "take a screenshot" }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "mcp__browser__computer", input: { action: "screenshot" } }] }, + { + role: "user", + content: [{ + type: "tool_result", + tool_use_id: "toolu_1", + content: [ + { type: "text", text: "Successfully captured screenshot (1x1, png)" }, + { type: "image", source: { type: "base64", media_type: "image/png", data: PNG } }, + ], + }], + }, + ], +}); + +describe("tool-result images reach OpenAI-format upstreams", () => { + it("emits the tool message text and a follow-up user message carrying the image", () => { + const out = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "gpt-x", screenshotTurn(), true, null, "openai"); + const toolMsg = out.messages.find((m) => m.role === "tool"); + expect(toolMsg.tool_call_id).toBe("toolu_1"); + expect(toolMsg.content).toBe("Successfully captured screenshot (1x1, png)"); + expect(toolMsg.content).not.toContain(PNG); + const follow = out.messages[out.messages.indexOf(toolMsg) + 1]; + expect(follow.role).toBe("user"); + const image = follow.content.find((p) => p.type === "image_url"); + expect(image.image_url.url).toBe(`data:image/png;base64,${PNG}`); + expect(follow.content.find((p) => p.type === "text").text).toContain("toolu_1"); + }); + + it("does not dump base64 into a tool message that had no text", () => { + const body = screenshotTurn(); + body.messages[2].content[0].content = [{ type: "image", source: { type: "base64", media_type: "image/png", data: PNG } }]; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "gpt-x", body, true, null, "openai"); + const toolMsg = out.messages.find((m) => m.role === "tool"); + expect(toolMsg.content).toBe(""); + expect(out.messages.some((m) => Array.isArray(m.content) && m.content.some((p) => p.type === "image_url"))).toBe(true); + }); + + it("leaves text-only tool results exactly as before", () => { + const body = screenshotTurn(); + body.messages[2].content[0].content = "plain result"; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "gpt-x", body, true, null, "openai"); + const toolMsg = out.messages.find((m) => m.role === "tool"); + expect(toolMsg.content).toBe("plain result"); + expect(out.messages[out.messages.length - 1]).toBe(toolMsg); + }); + + it("forwards tool-result images to Kiro as user images", () => { + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", screenshotTurn(), true, null, "kiro"); + const json = JSON.stringify(out.conversationState); + expect(json).toContain(PNG); + expect(json).toContain("Successfully captured screenshot"); + }); +}); + +describe("Kiro tool names round-trip", () => { + it("returns the sanitized→original map on the translated body", () => { + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", screenshotTurn(), true, null, "kiro"); + expect(out._toolNameMap).toBeInstanceOf(Map); + expect(out._toolNameMap.get("mcp_browser_computer")).toBe("mcp__browser__computer"); + const wire = JSON.parse(JSON.stringify(out.conversationState)); + expect(JSON.stringify(wire)).toContain("mcp_browser_computer"); + expect(JSON.stringify(wire)).not.toContain("mcp__browser__computer"); + }); + + it("omits the map when no name changed", () => { + const body = screenshotTurn(); + body.tools = [{ name: "plain_tool", description: "x", input_schema: { type: "object", properties: {} } }]; + body.messages[1].content[0].name = "plain_tool"; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", body, true, null, "kiro"); + expect(out._toolNameMap).toBeUndefined(); + }); + + it("restores the client name on streamed Claude tool_use blocks", () => { + const state = { ...initState(FORMATS.CLAUDE), toolNameMap: new Map([["mcp_browser_computer", "mcp__browser__computer"]]) }; + const chunk = { + id: "c1", object: "chat.completion.chunk", created: 1, model: "claude-sonnet-4.5", + choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_1", type: "function", function: { name: "mcp_browser_computer", arguments: "" } }] }, finish_reason: null }], + }; + const events = kiroToClaudeResponse(chunk, state); + const start = events.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + expect(start.content_block.name).toBe("mcp__browser__computer"); + }); + + it("passes unknown names through untouched", () => { + const state = { ...initState(FORMATS.CLAUDE), toolNameMap: new Map([["mcp_browser_computer", "mcp__browser__computer"]]) }; + const chunk = { + id: "c1", object: "chat.completion.chunk", created: 1, model: "claude-sonnet-4.5", + choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_2", type: "function", function: { name: "other_tool", arguments: "" } }] }, finish_reason: null }], + }; + const events = kiroToClaudeResponse(chunk, state); + const start = events.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + expect(start.content_block.name).toBe("other_tool"); + }); + + it("restores the client name on OpenAI chunks passed through kiro-to-openai", () => { + const state = { ...initState(FORMATS.OPENAI), toolNameMap: new Map([["mcp_browser_computer", "mcp__browser__computer"]]) }; + const chunk = { + id: "c1", object: "chat.completion.chunk", created: 1, model: "claude-sonnet-4.5", + choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_1", type: "function", function: { name: "mcp_browser_computer", arguments: "{}" } }] }, finish_reason: null }], + }; + const out = kiroToOpenAIResponse(chunk, state); + expect(out.choices[0].delta.tool_calls[0].function.name).toBe("mcp__browser__computer"); + expect(kiroToOpenAIResponse(chunk, initState(FORMATS.OPENAI))).toBe(chunk); + }); +}); + +describe("thinking display is preserved for Claude-format upstreams", () => { + it("keeps display on adaptive thinking", () => { + const body = { model: "claude-sonnet-5", thinking: { type: "adaptive", display: "summarized" }, output_config: { effort: "high" }, messages: [] }; + applyThinking(FORMATS.CLAUDE, "claude-sonnet-5", body, "claude"); + expect(body.thinking).toEqual({ type: "adaptive", display: "summarized" }); + expect(body.output_config).toEqual({ effort: "high" }); + }); + + it("keeps display on budget thinking and omits it when the client sent none", () => { + const withDisplay = { model: "claude-haiku-4-5-20251001", thinking: { type: "adaptive", display: "omitted" }, output_config: { effort: "low" }, messages: [] }; + applyThinking(FORMATS.CLAUDE, "claude-haiku-4-5-20251001", withDisplay, "claude"); + expect(withDisplay.thinking.type).toBe("enabled"); + expect(withDisplay.thinking.display).toBe("omitted"); + + const without = { model: "claude-sonnet-5", thinking: { type: "adaptive" }, output_config: { effort: "high" }, messages: [] }; + applyThinking(FORMATS.CLAUDE, "claude-sonnet-5", without, "claude"); + expect(without.thinking).toEqual({ type: "adaptive" }); + }); +}); + +describe("redact-thinking beta follows the client's display request", () => { + it("keeps redact-thinking by default and drops it for summarized display", () => { + expect(selectAnthropicBeta("claude-sonnet-5")).toContain("redact-thinking-2026-02-12"); + expect(selectAnthropicBeta("claude-sonnet-5", { thinking: { type: "adaptive", display: "omitted" } })).toContain("redact-thinking-2026-02-12"); + const summarized = selectAnthropicBeta("claude-sonnet-5", { thinking: { type: "adaptive", display: "summarized" } }); + expect(summarized).not.toContain("redact-thinking-2026-02-12"); + expect(summarized).toContain("interleaved-thinking-2025-05-14"); + expect(summarized).toContain("effort-2025-11-24"); + }); +}); + +describe("tool-result images reach Anthropic-compatible and Command Code upstreams", () => { + it("hoists a tool_result image into the same user turn after the results", () => { + const body = screenshotTurn(); + const out = hoistToolResultImages(body); + const user = out.messages[2]; + expect(user.content[0].type).toBe("tool_result"); + expect(user.content[0].content.every((c) => c.type !== "image")).toBe(true); + expect(user.content.some((c) => c.type === "image" && c.source?.data === PNG)).toBe(true); + expect(user.content.find((c) => c.type === "text" && /toolu_1/.test(c.text))).toBeTruthy(); + // No image: untouched object identity. + const plain = { messages: [{ role: "user", content: [{ type: "tool_result", tool_use_id: "x", content: "ok" }] }] }; + expect(hoistToolResultImages(plain)).toBe(plain); + }); + + it("sends an image block to Command Code instead of a placeholder", () => { + const openaiBody = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "muse-spark", screenshotTurn(), true, null, "commandcode"); + const out = openaiToCommandCodeRequest("muse-spark", openaiBody, true); + const json = JSON.stringify(out); + expect(json).not.toContain("[image omitted]"); + expect(json).toContain(`"type":"image"`); + expect(json).toContain(`data:image/png;base64,${PNG}`); + expect(json).toContain(`"mediaType":"image/png"`); + }); +}); From 82b1bca42a12fb50b19643d2ae73b2d99d80233e Mon Sep 17 00:00:00 2001 From: Qisthi Ramadhani Date: Thu, 17 Sep 2026 18:12:34 +0700 Subject: [PATCH 57/78] fix(kiro): use neutral placeholder for tool-result-only user turns Replace the literal 'continue' placeholder on tool-result-only user turns with 'Tool results provided.' to prevent models from treating it as a new user instruction. --- .../translator/concerns/kiroConversation.js | 17 ++- open-sse/translator/request/claude-to-kiro.js | 4 +- open-sse/translator/request/openai-to-kiro.js | 4 +- .../__snapshots__/golden-request.test.js.snap | 8 +- .../unit/kiro-tool-result-placeholder.test.js | 115 ++++++++++++++++++ 5 files changed, 143 insertions(+), 5 deletions(-) create mode 100644 tests/unit/kiro-tool-result-placeholder.test.js diff --git a/open-sse/translator/concerns/kiroConversation.js b/open-sse/translator/concerns/kiroConversation.js index 11d49dc7..5ac1485b 100644 --- a/open-sse/translator/concerns/kiroConversation.js +++ b/open-sse/translator/concerns/kiroConversation.js @@ -5,6 +5,20 @@ import { } from "../../config/kiroConstants.js"; const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/; + +/** + * Kiro rejects user turns with empty `content`, so a turn that only carries + * tool results needs placeholder text. It must not read like a user + * instruction: with "continue", models answer the word itself ("Nothing in + * progress to continue") and drop the task they were in the middle of. + */ +export const KIRO_TOOL_RESULTS_PLACEHOLDER = "Tool results provided."; +export const KIRO_EMPTY_USER_PLACEHOLDER = "continue"; + +/** Placeholder content for a user turn with no text of its own. */ +export function kiroEmptyUserContent(hasToolResults) { + return hasToolResults ? KIRO_TOOL_RESULTS_PLACEHOLDER : KIRO_EMPTY_USER_PLACEHOLDER; +} const TOOL_NAME_PATTERN = /[^a-zA-Z0-9_-]/g; function clone(value) { @@ -174,7 +188,8 @@ function normalizeTurns(history, currentMessage, modelId) { for (const turn of turns) { if (turn.userInputMessage) { - turn.userInputMessage.content = text(turn.userInputMessage.content).trim() || "continue"; + turn.userInputMessage.content = text(turn.userInputMessage.content).trim() + || kiroEmptyUserContent(turn.userInputMessage.userInputMessageContext?.toolResults?.length > 0); turn.userInputMessage.modelId ||= modelId; if (turn.userInputMessage.userInputMessageContext?.tools) { delete turn.userInputMessage.userInputMessageContext.tools; diff --git a/open-sse/translator/request/claude-to-kiro.js b/open-sse/translator/request/claude-to-kiro.js index ac7a704d..cffe59ec 100644 --- a/open-sse/translator/request/claude-to-kiro.js +++ b/open-sse/translator/request/claude-to-kiro.js @@ -34,6 +34,7 @@ import { ROLE, CLAUDE_BLOCK } from "../schema/index.js"; import { canonicalizeKiroConversation, normalizeKiroToolSpecs, + kiroEmptyUserContent, } from "../concerns/kiroConversation.js"; /** @@ -53,7 +54,8 @@ function convertClaudeMessagesToKiro(messages, model) { const flushPending = () => { if (currentRole === ROLE.USER) { - const content = pendingUserContent.join("\n\n").trim() || "continue"; + const content = pendingUserContent.join("\n\n").trim() + || kiroEmptyUserContent(pendingToolResults.length > 0); const userMsg = { userInputMessage: { content, modelId: model } }; if (pendingImages.length > 0) { diff --git a/open-sse/translator/request/openai-to-kiro.js b/open-sse/translator/request/openai-to-kiro.js index 0c9f04fd..b8846660 100644 --- a/open-sse/translator/request/openai-to-kiro.js +++ b/open-sse/translator/request/openai-to-kiro.js @@ -23,6 +23,7 @@ import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js"; import { canonicalizeKiroConversation, normalizeKiroToolSpecs, + kiroEmptyUserContent, } from "../concerns/kiroConversation.js"; /** @@ -51,7 +52,8 @@ function convertMessages(messages, model) { const flushPending = () => { if (currentRole === "user") { - const content = pendingUserContent.join("\n\n").trim() || "continue"; + const content = pendingUserContent.join("\n\n").trim() + || kiroEmptyUserContent(pendingToolResults.length > 0); const userMsg = { userInputMessage: { content: content, diff --git a/tests/translator/__snapshots__/golden-request.test.js.snap b/tests/translator/__snapshots__/golden-request.test.js.snap index a7a0c219..f5a5253b 100644 --- a/tests/translator/__snapshots__/golden-request.test.js.snap +++ b/tests/translator/__snapshots__/golden-request.test.js.snap @@ -239,7 +239,7 @@ exports[`GOLDEN request: OpenAI → Kiro > full body (image base64 + tool_result "userInputMessage": { "content": "[Context: Current time is -continue", +Tool results provided.", "modelId": "claude-sonnet-4.5", "origin": "AI_EDITOR", "userInputMessageContext": { @@ -281,7 +281,11 @@ continue", "history": [ { "userInputMessage": { - "content": "You are helpful. + "content": "[Context: Current time is + + +You are helpful. + What's in this image?", "images": [ diff --git a/tests/unit/kiro-tool-result-placeholder.test.js b/tests/unit/kiro-tool-result-placeholder.test.js new file mode 100644 index 00000000..ab910b42 --- /dev/null +++ b/tests/unit/kiro-tool-result-placeholder.test.js @@ -0,0 +1,115 @@ +import { describe, it, expect } from "vitest"; +import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to-kiro.js"; +import { claudeToKiroRequest } from "../../open-sse/translator/request/claude-to-kiro.js"; +import { + canonicalizeKiroConversation, + KIRO_TOOL_RESULTS_PLACEHOLDER, + KIRO_EMPTY_USER_PLACEHOLDER, +} from "../../open-sse/translator/concerns/kiroConversation.js"; + +const TOOLS_OPENAI = [{ + type: "function", + function: { + name: "get_weather", + description: "Get weather", + parameters: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, + }, +}]; + +const TOOLS_CLAUDE = [{ + name: "get_weather", + description: "Get weather", + input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, +}]; + +function allUserContents(payload) { + const state = payload.conversationState; + return [ + ...state.history.filter((t) => t.userInputMessage).map((t) => t.userInputMessage.content), + state.currentMessage.userInputMessage.content, + ]; +} + +describe("Kiro tool-result-only turns", () => { + it("OpenAI → Kiro: tool message gets a neutral placeholder, not \"continue\"", () => { + const payload = openaiToKiroRequest("claude-sonnet-4.6", { + tools: TOOLS_OPENAI, + messages: [ + { role: "user", content: "The secret word is PINEAPPLE. Weather in Jakarta?" }, + { role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "get_weather", arguments: "{\"city\":\"Jakarta\"}" } }] }, + { role: "tool", tool_call_id: "call_1", content: "32C, humid" }, + ], + }, true, {}); + + const current = payload.conversationState.currentMessage.userInputMessage; + expect(current.content).toContain(KIRO_TOOL_RESULTS_PLACEHOLDER); + expect(current.content).not.toMatch(/\bcontinue\b/); + expect(current.userInputMessageContext.toolResults).toHaveLength(1); + expect(allUserContents(payload).join("\n")).toContain("PINEAPPLE"); + }); + + it("Claude → Kiro: tool_result-only user message gets a neutral placeholder", () => { + const payload = claudeToKiroRequest("claude-sonnet-4.6", { + tools: TOOLS_CLAUDE, + messages: [ + { role: "user", content: "The secret word is PINEAPPLE. Weather in Jakarta?" }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Jakarta" } }] }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: "32C, humid" }] }, + ], + }, true, {}); + + const current = payload.conversationState.currentMessage.userInputMessage; + expect(current.content).toContain(KIRO_TOOL_RESULTS_PLACEHOLDER); + expect(current.content).not.toMatch(/\bcontinue\b/); + expect(current.userInputMessageContext.toolResults).toHaveLength(1); + }); + + it("keeps real user text when a turn has both text and tool results", () => { + const payload = claudeToKiroRequest("claude-sonnet-4.6", { + tools: TOOLS_CLAUDE, + messages: [ + { role: "user", content: "Weather in Jakarta?" }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Jakarta" } }] }, + { role: "user", content: [ + { type: "tool_result", tool_use_id: "toolu_1", content: "32C" }, + { type: "text", text: "Now answer in one word." }, + ] }, + ], + }, true, {}); + + const current = payload.conversationState.currentMessage.userInputMessage; + expect(current.content).toContain("Now answer in one word."); + expect(current.content).not.toContain(KIRO_TOOL_RESULTS_PLACEHOLDER); + }); + + it("canonicalize: history turn with tool results and no text uses the placeholder", () => { + const result = canonicalizeKiroConversation({ + history: [ + { userInputMessage: { content: "Weather in Jakarta?", modelId: "m" } }, + { assistantResponseMessage: { content: "", toolUses: [{ toolUseId: "t1", name: "get_weather", input: { city: "Jakarta" } }] } }, + { userInputMessage: { content: "", modelId: "m", userInputMessageContext: { toolResults: [{ toolUseId: "t1", status: "success", content: [{ text: "32C" }] }] } } }, + { assistantResponseMessage: { content: "It is 32C." } }, + ], + currentMessage: { userInputMessage: { content: "Hot or cold?", modelId: "m" } }, + modelId: "m", + toolSpecs: [{ toolSpecification: { name: "get_weather", description: "Get weather", inputSchema: { json: { type: "object", properties: {} } } } }], + nameMap: new Map([["get_weather", "get_weather"]]), + }); + + expect(result.valid).toBe(true); + expect(result.history[2].userInputMessage.content).toBe(KIRO_TOOL_RESULTS_PLACEHOLDER); + }); + + it("canonicalize: an empty turn without tool results still falls back to \"continue\"", () => { + const result = canonicalizeKiroConversation({ + history: [{ assistantResponseMessage: { content: "Hello" } }], + currentMessage: { userInputMessage: { content: "", modelId: "m" } }, + modelId: "m", + toolSpecs: [], + nameMap: new Map(), + }); + + expect(result.history[0].userInputMessage.content).toBe(KIRO_EMPTY_USER_PLACEHOLDER); + expect(result.currentMessage.userInputMessage.content).toBe(KIRO_EMPTY_USER_PLACEHOLDER); + }); +}); From c49efdf5288200f19ed78136bdb0a6dcc3d5d7ff Mon Sep 17 00:00:00 2001 From: Qisthi Ramadhani Date: Thu, 17 Sep 2026 18:13:14 +0700 Subject: [PATCH 58/78] fix(kiro): preserve underscores in tool names and restore sanitized names in responses Do not collapse consecutive underscores in uniqueName so mcp__server__tool is sent intact to Kiro, attach reverse map on request translation, and restore client tool names in responses. --- .../translator/concerns/kiroConversation.js | 1 - .../translator/response/kiro-to-claude.js | 6 +- tests/translator/agent-client-fixes.test.js | 8 +- .../kiro-tool-name-roundtrip.test.js | 110 ++++++++++++++++++ 4 files changed, 118 insertions(+), 7 deletions(-) create mode 100644 tests/translator/kiro-tool-name-roundtrip.test.js diff --git a/open-sse/translator/concerns/kiroConversation.js b/open-sse/translator/concerns/kiroConversation.js index 5ac1485b..b19e9ba6 100644 --- a/open-sse/translator/concerns/kiroConversation.js +++ b/open-sse/translator/concerns/kiroConversation.js @@ -48,7 +48,6 @@ function uniqueName(rawName, index, usedNames) { const cleaned = String(rawName || "") .trim() .replace(TOOL_NAME_PATTERN, "_") - .replace(/_+/g, "_") .replace(/^_+|_+$/g, ""); const base = trimCodePoints(cleaned || `tool_${index + 1}`, KIRO_TOOL_NAME_MAX_LENGTH); let candidate = base; diff --git a/open-sse/translator/response/kiro-to-claude.js b/open-sse/translator/response/kiro-to-claude.js index 9118b9cb..d9fc0aab 100644 --- a/open-sse/translator/response/kiro-to-claude.js +++ b/open-sse/translator/response/kiro-to-claude.js @@ -48,9 +48,9 @@ function convertFinishReason(reason) { */ // Kiro only accepts sanitized tool names; the request translator leaves the // reverse map on the stream state so calls come back under the client's names. -function restoreToolName(state, name) { +function restoreToolName(stateOrData, name) { const raw = name || ""; - const map = state?.toolNameMap; + const map = stateOrData?.toolNameMap || stateOrData?._toolNameMap; return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw; } @@ -254,7 +254,7 @@ export function kiroToClaudeNonStreaming(data) { content.push({ type: "tool_use", id: tc.id || `toolu_${Date.now()}`, - name: restoreToolName(state, tc.function?.name), + name: restoreToolName(data, tc.function?.name), input, }); } diff --git a/tests/translator/agent-client-fixes.test.js b/tests/translator/agent-client-fixes.test.js index 3f4e7cd7..6004db80 100644 --- a/tests/translator/agent-client-fixes.test.js +++ b/tests/translator/agent-client-fixes.test.js @@ -79,12 +79,14 @@ describe("tool-result images reach OpenAI-format upstreams", () => { describe("Kiro tool names round-trip", () => { it("returns the sanitized→original map on the translated body", () => { - const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", screenshotTurn(), true, null, "kiro"); + const body = screenshotTurn(); + body.tools = [{ name: "mcp.browser.computer", description: "browser", input_schema: { type: "object", properties: {} } }]; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", body, true, null, "kiro"); expect(out._toolNameMap).toBeInstanceOf(Map); - expect(out._toolNameMap.get("mcp_browser_computer")).toBe("mcp__browser__computer"); + expect(out._toolNameMap.get("mcp_browser_computer")).toBe("mcp.browser.computer"); const wire = JSON.parse(JSON.stringify(out.conversationState)); expect(JSON.stringify(wire)).toContain("mcp_browser_computer"); - expect(JSON.stringify(wire)).not.toContain("mcp__browser__computer"); + expect(JSON.stringify(wire)).not.toContain("mcp.browser.computer"); }); it("omits the map when no name changed", () => { diff --git a/tests/translator/kiro-tool-name-roundtrip.test.js b/tests/translator/kiro-tool-name-roundtrip.test.js new file mode 100644 index 00000000..a8f151ad --- /dev/null +++ b/tests/translator/kiro-tool-name-roundtrip.test.js @@ -0,0 +1,110 @@ +import { describe, it, expect } from "vitest"; +import { normalizeKiroToolSpecs } from "../../open-sse/translator/concerns/kiroConversation.js"; +import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to-kiro.js"; +import { claudeToKiroRequest } from "../../open-sse/translator/request/claude-to-kiro.js"; +import { kiroToOpenAIResponse } from "../../open-sse/translator/response/kiro-to-openai.js"; +import { kiroToClaudeResponse, kiroToClaudeNonStreaming } from "../../open-sse/translator/response/kiro-to-claude.js"; + +describe("Kiro tool name normalization and roundtrip", () => { + it("preserves consecutive underscores like mcp__gitea__search_repos without collapsing", () => { + const { specs, nameMap } = normalizeKiroToolSpecs([ + { name: "mcp__gitea__search_repos", description: "Search Gitea" }, + ]); + expect(specs).toHaveLength(1); + expect(specs[0].toolSpecification.name).toBe("mcp__gitea__search_repos"); + expect(nameMap.get("mcp__gitea__search_repos")).toBe("mcp__gitea__search_repos"); + }); + + it("builds _toolNameMap for illegal characters and deduplicates colliding names", () => { + const tools = [ + { name: "my.tool/search", description: "tool 1" }, + { name: "my_tool_search", description: "tool 2" }, + ]; + const openaiPayload = openaiToKiroRequest("claude-sonnet-4.6", { + tools: tools.map((t) => ({ type: "function", function: t })), + messages: [{ role: "user", content: "hello" }], + }, true, {}); + + expect(openaiPayload._toolNameMap).toBeInstanceOf(Map); + // my.tool/search cleaned to my_tool_search. Since my_tool_search comes next, it becomes my_tool_search_2 + expect(openaiPayload._toolNameMap.get("my_tool_search")).toBe("my.tool/search"); + + const claudePayload = claudeToKiroRequest("claude-sonnet-4.6", { + tools, + messages: [{ role: "user", content: "hello" }], + }, true, {}); + + expect(claudePayload._toolNameMap).toBeInstanceOf(Map); + expect(claudePayload._toolNameMap.get("my_tool_search")).toBe("my.tool/search"); + }); + + it("does not attach _toolNameMap when all tool names are legal and unchanged", () => { + const tools = [ + { name: "mcp__gitea__search_repos", description: "Search Gitea" }, + { name: "bash_exec", description: "Run bash" }, + ]; + const payload = openaiToKiroRequest("claude-sonnet-4.6", { + tools: tools.map((t) => ({ type: "function", function: t })), + messages: [{ role: "user", content: "hello" }], + }, true, {}); + + expect(payload._toolNameMap).toBeUndefined(); + }); + + it("restores original tool name in kiroToOpenAIResponse when state.toolNameMap is present", () => { + const state = { + toolNameMap: new Map([["my_tool_search", "my.tool/search"]]), + }; + const event = { + toolUseEvent: { + toolUseId: "call_123", + name: "my_tool_search", + input: { q: "test" }, + }, + }; + const chunk = kiroToOpenAIResponse(event, state); + expect(chunk).not.toBeNull(); + expect(chunk.choices[0].delta.tool_calls[0].function.name).toBe("my.tool/search"); + }); + + it("restores original tool name in kiroToClaudeResponse streaming when state.toolNameMap is present", () => { + const state = { + toolNameMap: new Map([["my_tool_search", "my.tool/search"]]), + toolCalls: new Map(), + nextBlockIndex: 0, + }; + const chunk = { + id: "chatcmpl-1", + choices: [{ + delta: { + tool_calls: [{ + index: 0, + id: "call_123", + type: "function", + function: { name: "my_tool_search", arguments: "" }, + }], + }, + }], + }; + const events = kiroToClaudeResponse(chunk, state); + const startEvent = events.find((e) => e.type === "content_block_start"); + expect(startEvent).toBeDefined(); + expect(startEvent.content_block.name).toBe("my.tool/search"); + }); + + it("restores original tool name in kiroToClaudeNonStreaming when toolNameMap is present", () => { + const data = { + choices: [{ + message: { + tool_calls: [{ + id: "call_123", + function: { name: "my_tool_search", arguments: "{}" }, + }], + }, + }], + toolNameMap: new Map([["my_tool_search", "my.tool/search"]]), + }; + const result = kiroToClaudeNonStreaming(data); + expect(result.content[0].name).toBe("my.tool/search"); + }); +}); From eafac37dcbd24d867b18bd70e6959eccbec500fe Mon Sep 17 00:00:00 2001 From: anojndr Date: Thu, 17 Sep 2026 18:17:18 +0700 Subject: [PATCH 59/78] fix(opencode): strip prior reasoning items on Muse Spark Responses models Strip prior-turn type: reasoning items and encrypted_content fields from body.input on Muse Spark Responses endpoints in OpenCode and OpenCode Go executors to avoid HTTP 400 errors across rotated proxy accounts. --- open-sse/executors/opencode-go.js | 6 ++ open-sse/executors/opencode.js | 82 ++++++++++++++++++- .../opencode-go-muse-spark-responses.test.js | 22 +++++ .../unit/opencode-muse-spark-thinking.test.js | 59 +++++++++++++ 4 files changed, 167 insertions(+), 2 deletions(-) diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js index c7065ea7..fe90c6eb 100644 --- a/open-sse/executors/opencode-go.js +++ b/open-sse/executors/opencode-go.js @@ -94,6 +94,12 @@ function sanitizeResponsesItems(body) { if (!Array.isArray(body.input)) return; body.input = body.input.filter((item) => { if (!item || typeof item !== "object" || Array.isArray(item)) return true; + // Strip prior-turn reasoning items: Muse Spark contributor models route to + // an upstream Console backend where encrypted_content cannot be validated across + // rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller". + if (item.type === "reasoning") return false; + delete item.encrypted_content; + delete item.reasoning_encrypted_content; if (item.type === "function_call") { if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false; item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN); diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index c5d0caa7..c4c31720 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -7,9 +7,16 @@ import { injectReasoningContent } from "../utils/reasoningContentInjector.js"; import { resolveSessionId } from "../utils/sessionManager.js"; import { isMuseSparkModel } from "../providers/models/helpers.js"; import { ANTHROPIC_API_VERSION } from "../providers/shared.js"; +import { + normalizeResponsesInput, + clampResponsesCallId, + coerceResponsesArguments, + coerceResponsesOutput, +} from "../translator/formats/responsesApi.js"; const OPENCODE_UA = "opencode/1.18.31"; const MAX_SESSION_LENGTH = 256; +const MAX_TOOL_NAME_LEN = 128; const SESSION_HEADER = "x-opencode-session"; const SESSION_FIELD = "_opencodeSession"; const REQ_FIELD = "_opencodeRequest"; @@ -24,7 +31,6 @@ function hasValidOpencodeVersion(ua) { const minor = parseInt(m[2], 10); return major > 1 || (major === 1 && minor >= 17); } - // Models served by /zen/v1/responses; every other model stays on /chat/completions. const RESPONSES_MODELS = new Set([ "muse-spark-1.2-contributor-free", @@ -300,6 +306,69 @@ function resolveOpencodeRequestId(body, credentials, sessionId) { return deriveRequestId(sessionId, body); } +function normalizeResponsesTools(body) { + if (!Array.isArray(body.tools)) return; + const validNames = new Set(); + body.tools = body.tools.filter((tool) => { + if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false; + const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null; + const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : ""); + const name = rawName.trim(); + if (!name) return false; + const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : ""); + let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)) + ? tool.parameters + : (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} }); + if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} }; + for (const k of Object.keys(tool)) delete tool[k]; + tool.type = "function"; + tool.name = name.slice(0, MAX_TOOL_NAME_LEN); + if (description) tool.description = description; + tool.parameters = parameters; + validNames.add(tool.name); + return true; + }); + if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) { + if (body.tool_choice.type === "function") { + const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : ""; + if (!n || !validNames.has(n)) delete body.tool_choice; + } + } +} + +function sanitizeResponsesItems(body) { + if (!Array.isArray(body.input)) return; + body.input = body.input.filter((item) => { + if (!item || typeof item !== "object" || Array.isArray(item)) return true; + // Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials + // (`Bearer public`) routing to an upstream OpenAI/Console account pool. + // OpenAI Responses API strictly enforces that reasoning `encrypted_content` + // can only be decrypted by the exact caller/account that issued it; sending it + // across different accounts or rotating proxy relays triggers: + // [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400). + // Furthermore, under stateless mode (store=false), omitting encrypted_content + // causes OpenAI to reject the referenced reasoning item as "not found or was deleted". + // Dropping prior reasoning items allows multi-turn conversations and tool-calling + // loops to succeed cleanly. + if (item.type === "reasoning") return false; + delete item.encrypted_content; + delete item.reasoning_encrypted_content; + if (item.type === "function_call") { + if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false; + item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN); + item.call_id = clampResponsesCallId(item.call_id); + item.arguments = coerceResponsesArguments(item.arguments); + return true; + } + if (item.type === "function_call_output") { + item.call_id = clampResponsesCallId(item.call_id); + item.output = coerceResponsesOutput(item.output); + return true; + } + return true; + }); +} + function normalizeOpencodeReasoning(model, body) { const current = body.reasoning; const currentReasoning = current && typeof current === "object" && !Array.isArray(current) @@ -341,7 +410,12 @@ export class OpenCodeExecutor extends BaseExecutor { transformRequest(model, body, stream, credentials) { if (body && typeof body === "object" && model && !body.model) body.model = model; - if (isResponsesModel(model) && body && typeof body === "object") { + if (isResponsesModel(model || body?.model) && body && typeof body === "object") { + const normalized = normalizeResponsesInput(body.input); + if (normalized) body.input = normalized; + if (!Array.isArray(body.input) || body.input.length === 0) { + body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }]; + } // Responses API names the output cap max_output_tokens and takes thinking // as reasoning:{effort,summary} — normalize the Chat fields at this boundary. if (body.max_output_tokens === undefined) { @@ -351,6 +425,10 @@ export class OpenCodeExecutor extends BaseExecutor { delete body.max_tokens; delete body.max_completion_tokens; normalizeOpencodeReasoning(model, body); + body.stream = true; + body.store = false; + normalizeResponsesTools(body); + sanitizeResponsesItems(body); } return injectReasoningContent({ provider: this.provider, model, body }); } diff --git a/tests/unit/opencode-go-muse-spark-responses.test.js b/tests/unit/opencode-go-muse-spark-responses.test.js index cbe5c349..98a780d7 100644 --- a/tests/unit/opencode-go-muse-spark-responses.test.js +++ b/tests/unit/opencode-go-muse-spark-responses.test.js @@ -136,6 +136,28 @@ describe("OpenCodeGoExecutor routing + sanitization", () => { expect(out.tools.find((t) => t.name === "bare").parameters).toEqual({ type: "object", properties: {} }); expect(out.tools.find((t) => t.name === "full").parameters).toEqual({ type: "object", properties: { a: { type: "string" } } }); }); + + it("strips prior-turn reasoning items carrying encrypted_content from input", () => { + const ex = new OpenCodeGoExecutor(); + const body = { + model: MODEL, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { + type: "reasoning", + id: "rs_123", + encrypted_content: "ENC_BLOB_TURN_1", + summary: [{ type: "summary_text", text: "thinking text" }], + }, + { type: "function_call", call_id: "c1", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "c1", output: "ok" }, + ], + }; + const out = ex.transformRequest(MODEL, body, true, {}); + expect(out.input.some((i) => i.type === "reasoning")).toBe(false); + expect(JSON.stringify(out.input)).not.toContain("ENC_BLOB_TURN_1"); + expect(out.input.map((i) => i.type)).toEqual(["message", "function_call", "function_call_output"]); + }); }); describe("chat/claude clients translate to Responses without breaking tools", () => { diff --git a/tests/unit/opencode-muse-spark-thinking.test.js b/tests/unit/opencode-muse-spark-thinking.test.js index 9ffc29b1..94a44e8c 100644 --- a/tests/unit/opencode-muse-spark-thinking.test.js +++ b/tests/unit/opencode-muse-spark-thinking.test.js @@ -165,4 +165,63 @@ describe("OpenCode Free Muse Spark thinking", () => { expect(out.max_tokens).toBeUndefined(); } }); + + it("strips prior-turn reasoning items carrying encrypted_content from input", () => { + const executor = new OpenCodeExecutor(); + const model = "muse-spark-1.3-contributor-free"; + const body = { + model, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "say hi" }] }, + { + type: "reasoning", + id: "rs_123", + encrypted_content: "ENC_BLOB_TURN_1", + summary: [{ type: "summary_text", text: "thinking text" }], + }, + { + type: "function_call", + id: "fc_1", + call_id: "call_1", + name: "shell", + arguments: JSON.stringify({ command: "echo hi" }), + }, + { + type: "function_call_output", + call_id: "call_1", + output: "hi", + }, + { type: "message", role: "user", content: [{ type: "input_text", text: "now say bye" }] }, + ], + tools: [ + { + type: "function", + function: { + name: "shell", + description: "Run shell command", + parameters: { type: "object" }, + }, + }, + ], + }; + + const out = executor.transformRequest(model, body, true, {}); + expect(out.stream).toBe(true); + expect(out.store).toBe(false); + // Prior reasoning items stripped to prevent 400 "reasoning encrypted_content was not issued to this caller" + expect(out.input.some((item) => item.type === "reasoning")).toBe(false); + expect(JSON.stringify(out.input)).not.toContain("ENC_BLOB_TURN_1"); + // User message, function_call, function_call_output, and next user message survive + const types = out.input.map((item) => item.type); + expect(types).toEqual(["message", "function_call", "function_call_output", "message"]); + // Tools flattened and empty properties added + expect(out.tools).toEqual([ + { + type: "function", + name: "shell", + description: "Run shell command", + parameters: { type: "object", properties: {} }, + }, + ]); + }); }); From aa14ef72e2ea71431dc5963214a7615116af386f Mon Sep 17 00:00:00 2001 From: KunN-21 Date: Thu, 17 Sep 2026 18:19:38 +0700 Subject: [PATCH 60/78] fix(opencode): normalize Muse Free tool choice OpenCode Free returns HTTP 400 for muse-spark-1.3-contributor-free when tool_choice is non-auto. Declare forceAutoToolChoiceModels quirk and normalize explicit tool_choice to auto. --- open-sse/executors/opencode.js | 5 ++ open-sse/providers/registry/opencode.js | 3 + tests/unit/opencode-free-tool-choice.test.js | 90 ++++++++++++++++++++ 3 files changed, 98 insertions(+) create mode 100644 tests/unit/opencode-free-tool-choice.test.js diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index c4c31720..531c5df5 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -411,6 +411,11 @@ export class OpenCodeExecutor extends BaseExecutor { transformRequest(model, body, stream, credentials) { if (body && typeof body === "object" && model && !body.model) body.model = model; if (isResponsesModel(model || body?.model) && body && typeof body === "object") { + // ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng. + if ("tool_choice" in body && body.tool_choice !== "auto" + && this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) { + body.tool_choice = "auto"; + } const normalized = normalizeResponsesInput(body.input); if (normalized) body.input = normalized; if (!Array.isArray(body.input) || body.input.length === 0) { diff --git a/open-sse/providers/registry/opencode.js b/open-sse/providers/registry/opencode.js index fc2ab8c6..0914f431 100644 --- a/open-sse/providers/registry/opencode.js +++ b/open-sse/providers/registry/opencode.js @@ -18,6 +18,9 @@ export default { "x-opencode-client": "desktop", }, noAuth: true, + quirks: { + forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"], + }, }, models: [ // Endpoint formats differ per model, so declare non-chat models explicitly. diff --git a/tests/unit/opencode-free-tool-choice.test.js b/tests/unit/opencode-free-tool-choice.test.js new file mode 100644 index 00000000..4d3aca91 --- /dev/null +++ b/tests/unit/opencode-free-tool-choice.test.js @@ -0,0 +1,90 @@ +import { describe, expect, it, vi } from "vitest"; +import { PROVIDERS } from "../../open-sse/config/providers.js"; +import { OpenCodeExecutor } from "../../open-sse/executors/opencode.js"; +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(async () => ({ ok: true, status: 200, headers: { get: () => "" } })), +})); + +// Break caught: opencode/muse-spark-1.3-contributor-free 400 vì upstream +// chỉ nhận tool_choice "auto"; named/required/none phải demote sang "auto". +const FREE_13 = "muse-spark-1.3-contributor-free"; +const CREDS = { connectionId: "opencode-free-tool-choice-test" }; +const INPUT = [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }]; +const TOOLS = [{ type: "function", name: "get_weather", description: "w", parameters: { type: "object", properties: {} } }]; + +function responsesBody(model, tool_choice) { + const body = { model, input: structuredClone(INPUT), tools: structuredClone(TOOLS) }; + if (tool_choice !== undefined) body.tool_choice = tool_choice; + return body; +} + +describe("opencode Free 1.3 tool_choice auto-only", () => { + it("khai quirk đúng model 1.3-Free trong registry", () => { + expect(PROVIDERS.opencode.quirks?.forceAutoToolChoiceModels).toEqual([FREE_13]); + }); + + it.each([ + ["Responses named", { type: "function", name: "get_weather" }], + ["Chat function named", { type: "function", function: { name: "get_weather" } }], + ["Claude tool named", { type: "tool", name: "get_weather" }], + ["required", "required"], + ["none", "none"], + ])("demote %s sang auto (plain và max)", (_label, choice) => { + for (const model of [FREE_13, `${FREE_13}(max)`]) { + const body = responsesBody(model, structuredClone(choice)); + const out = new OpenCodeExecutor().transformRequest(model, body, true, CREDS); + expect(out.tool_choice).toBe("auto"); + expect(out.tools).toEqual(TOOLS); + expect(out.input).toEqual(INPUT); + } + }); + + it("giữ auto và absent; tools/input nguyên vẹn", () => { + const autoOut = new OpenCodeExecutor().transformRequest( + FREE_13, responsesBody(FREE_13, "auto"), true, CREDS, + ); + expect(autoOut.tool_choice).toBe("auto"); + expect(autoOut.tools).toEqual(TOOLS); + expect(autoOut.input).toEqual(INPUT); + + const absentOut = new OpenCodeExecutor().transformRequest( + FREE_13, responsesBody(FREE_13, undefined), true, CREDS, + ); + expect("tool_choice" in absentOut).toBe(false); + expect(absentOut.tools).toEqual(TOOLS); + expect(absentOut.input).toEqual(INPUT); + }); + + it.each([ + ["1.2-Free", "muse-spark-1.2-contributor-free"], + ["future 1.4-Free", "muse-spark-1.4-contributor-free"], + ["Go id", "muse-spark-1.3-contributor"], + ["non-Muse", "big-pickle"], + ])("không đổi tool_choice của %s", (_label, model) => { + const choice = { type: "function", name: "get_weather" }; + const body = responsesBody(model, structuredClone(choice)); + const out = new OpenCodeExecutor().transformRequest(model, body, true, CREDS); + expect(out.tool_choice).toEqual(choice); + }); + + it("wire: execute gửi choice auto tới /zen/v1/responses", async () => { + proxyAwareFetch.mockClear(); + const ex = new OpenCodeExecutor(); + const body = responsesBody(FREE_13, { type: "function", name: "get_weather" }); + const { url, transformedBody } = await ex.execute({ + model: FREE_13, body, stream: true, credentials: CREDS, + }); + expect(url).toBe("https://opencode.ai/zen/v1/responses"); + expect(transformedBody.tool_choice).toBe("auto"); + expect(proxyAwareFetch).toHaveBeenCalledTimes(1); + const [actualUrl, actualInit] = proxyAwareFetch.mock.calls[0]; + expect(actualUrl).toBe("https://opencode.ai/zen/v1/responses"); + const sent = JSON.parse(actualInit.body); + expect(sent.tool_choice).toBe("auto"); + expect(sent.model).toBe(FREE_13); + expect(sent.tools).toEqual(TOOLS); + expect(sent.input).toEqual(INPUT); + }); +}); From 20a43f5a2ca491335e623c4ceb7d36cac1f225d7 Mon Sep 17 00:00:00 2001 From: RaoYu <2425198313@qq.com> Date: Thu, 17 Sep 2026 18:26:05 +0700 Subject: [PATCH 61/78] fix(auth): don't cool down an account for a request-scoped 4xx Do not trigger account cooldown or fallback for request-scoped 4xx errors that match no account rules so healthy credentials are not locked out for context length or validation errors. --- open-sse/services/accountFallback.js | 14 +++++++++ tests/unit/account-fallback-4xx.test.js | 38 +++++++++++++++++++++++++ 2 files changed, 52 insertions(+) create mode 100644 tests/unit/account-fallback-4xx.test.js diff --git a/open-sse/services/accountFallback.js b/open-sse/services/accountFallback.js index 8d280da4..766b9981 100644 --- a/open-sse/services/accountFallback.js +++ b/open-sse/services/accountFallback.js @@ -45,6 +45,20 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) { } } + // Request-scoped client errors that matched no rule above: a 400 caused by the + // request itself (context overflow, malformed body, unsupported parameter) says + // nothing about the credential, so cooling the account down only removes a + // healthy connection from rotation. With a single connection it is worse: every + // later request in the window fails with a copy of this very error + // ("all 1 accounts locked for | lastError=[400]: ..."), which hides the + // real cause from the caller and makes unrelated sessions look like they hit the + // same limit. Hand the upstream error back for this request instead. + // Account-scoped statuses keep their rules above (401/402/403/404/429), and the + // text rules still win for rate-limit / quota / capacity wording. + if (status >= 400 && status < 500 && status !== 401 && status !== 402 && status !== 403 && status !== 429) { + return { shouldFallback: false, cooldownMs: 0 }; + } + // Default: transient cooldown for any unmatched error return { shouldFallback: true, cooldownMs: TRANSIENT_COOLDOWN_MS }; } diff --git a/tests/unit/account-fallback-4xx.test.js b/tests/unit/account-fallback-4xx.test.js new file mode 100644 index 00000000..ba94c16c --- /dev/null +++ b/tests/unit/account-fallback-4xx.test.js @@ -0,0 +1,38 @@ +// Regression: an unmatched 4xx (a request-scoped failure) used to hit the +// transient-cooldown default, which locked the account for 30s and — with a +// single connection — answered every other request in that window with a copy of +// the first error. A 400 "maximum context length" from one session therefore +// looked like the same failure in unrelated sessions. +import { describe, expect, it } from "vitest"; +import { checkFallbackError } from "../../open-sse/services/accountFallback.js"; + +describe("checkFallbackError — request-scoped vs account-scoped failures", () => { + it("does not cool the account down for a 400 caused by the request", () => { + const result = checkFallbackError(400, JSON.stringify({ + error: { + message: "This model's maximum context length is 1048576 tokens. However, you requested 1186139 tokens", + type: "invalid_request_error", + }, + })); + + expect(result).toEqual({ shouldFallback: false, cooldownMs: 0 }); + }); + + it("still falls back for account-scoped statuses", () => { + for (const status of [401, 402, 403, 404, 429]) { + expect(checkFallbackError(status, "nope").shouldFallback).toBe(true); + } + }); + + it("still honours rate-limit / quota wording on any 4xx", () => { + expect(checkFallbackError(400, "rate limit reached").shouldFallback).toBe(true); + expect(checkFallbackError(422, "quota exceeded").shouldFallback).toBe(true); + }); + + it("keeps the transient cooldown for unmatched server errors", () => { + const result = checkFallbackError(503, "upstream exploded"); + + expect(result.shouldFallback).toBe(true); + expect(result.cooldownMs).toBeGreaterThan(0); + }); +}); From 367fc546d8c4a98d6c54e678af701f79b5b4ee8c Mon Sep 17 00:00:00 2001 From: Rafli Ahmad Zulfikar Date: Thu, 17 Sep 2026 18:38:10 +0700 Subject: [PATCH 62/78] feat(models): deepseek-v4.* accepts low..max effort; flag thinkingEffortSupported and vision --- open-sse/providers/capabilities.js | 6 +++++- open-sse/providers/thinkingLevels.js | 4 ++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 12c9d82f..c02f255f 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -378,7 +378,11 @@ export const PATTERN_CAPABILITIES = [ { pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } }, // ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─ - { pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } }, + // v4.1+ has real image input (probed live on Alibaba MaaS: correct color + // read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore + // them (answered "Unknown"), so vision stays scoped to v4.* dotted releases. + { pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } }, + { pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } }, { pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } }, { pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } }, { pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } }, diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index 0b77e4b4..94897ec7 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -41,6 +41,10 @@ const PATTERN_THINKING = [ { provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS }, { pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking + // DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max + // all 200 via output_config.effort; "none" is a 400 on the anthropic route + // (disable thinking instead). none kept for the picker = disable. + { pattern: "*deepseek-v4.*", levels: ["none", "low", "medium", "high", "xhigh", "max"] }, // codebuddy-cn per-model effort sets — the server's product-config payload // publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint // accepts any level you send (probed none/minimal/low/medium/high/xhigh/max From 725e2c1187f69f93310c1bbd0ecd8063b474d068 Mon Sep 17 00:00:00 2001 From: Ahmad Beyranvand Date: Thu, 17 Sep 2026 18:42:31 +0700 Subject: [PATCH 63/78] feat(i18n): integrate Persian (fa) translation --- public/i18n/literals/fa.json | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/public/i18n/literals/fa.json b/public/i18n/literals/fa.json index 0f28367f..bfd2af60 100644 --- a/public/i18n/literals/fa.json +++ b/public/i18n/literals/fa.json @@ -1389,5 +1389,21 @@ "⚠️ Risk Notice: This provider uses a subscription/OAuth session not officially licensed for proxy/router use. Account may be restricted or banned. Use at your own risk.": "⚠️ اطلاعیه ریسک: این ارائه‌دهنده از اشتراک/جلسه OAuth استفاده می‌کند که به طور رسمی برای استفاده پروکسی/روتر مجوز ندارد. حساب ممکن است محدود یا مسدود شود. با مسئولیت خود استفاده کنید.", "✓ Confirm Add": "✓ تأیید افزودن", "📝 Configure providers in dashboard or use environment variables": "📝 ارائه‌دهندگان را در داشبورد پیکربندی کنید یا از متغیرهای محیطی استفاده کنید", - "🔐 OAuth required. Add now and authenticate after Apply; tool list will be discovered after first connect.": "🔐 نیاز به OAuth. اکنون اضافه کنید و پس از اعمال، احراز هویت کنید؛ لیست ابزارها پس از اولین اتصال کشف می‌شود." + "🔐 OAuth required. Add now and authenticate after Apply; tool list will be discovered after first connect.": "🔐 نیاز به OAuth. اکنون اضافه کنید و پس از اعمال، احراز هویت کنید؛ لیست ابزارها پس از اولین اتصال کشف می‌شود.", + "Combo & Vision Adapter":"آداپتور بینایی و ترکیبی", + "Vision Adapter": "آداپتور بینایی", + "Skills": "مهارت‌ها", + "Cached Cost":"هزینه‌ی کش‌شده", + "Compress prompts and outputs to save tokens":"فشرده‌سازی پرامپت‌ها و خروجی‌ها برای صرفه‌جویی در توکن‌ها", + "Manage your web providers": "مدیریت ارائه‌دهندگان خدمات وب خود را انجام دهید", + "providers":"ارائه‌دهندگان", + "combos": "ترکیبی", + "Configure enterprise Single Sign-On (SSO) for dashboard access using SAML 2.0 or OIDC.":"پیکربندی ورود تک‌نشانه سازمانی (SSO) برای دسترسی به داشبورد با استفاده از SAML 2.0 یا OIDC.", + "Optional SSO via Okta, Entra ID, Keycloak, or OIDC":"SSO اختیاری از طریق Okta، Entra ID، Keycloak یا OIDC", + "Single Sign-On (SSO)":"ورود یک‌باره (SSO)", + "SSO Protocol":"پروتکل SSO", + "Keep legacy password login.":"حفظ ورود با نام کاربری و رمز عبور قدیمی", + "Require SSO for dashboard access.":"برای دسترسی به داشبورد نیاز به SSO دارد.", + "Allow password or SSO login.":"اجازه ورود با رمز عبور یا ورود یک‌بار مصرف (SSO)", + "Save OIDC settings":"ذخیره تنظیمات OIDC" } From ef18175226449e11481e984011ee5dc2dd03dea2 Mon Sep 17 00:00:00 2001 From: Mosabbir Maruf <96982516+mosabbir-maruf@users.noreply.github.com> Date: Thu, 17 Sep 2026 18:46:13 +0700 Subject: [PATCH 64/78] fix(zed): harden OAuth lifecycle and live model support - executors/zed.js: use exact wire values (anthropic, open_ai, google, x_ai) and strip incompatible Vertex safetySettings on the Google path - shared/zedAuth.js: robust callback query parsing, reject garbage PKCS#1 v1.5 decryptions, and thread proxyOptions when fetching LLM tokens - oauth: preserve systemId across authorize/register/exchange lifecycle, renew proxy idle timeout on reuse, and ignore non-callback localhost requests - shared/OAuthModal.js: track owned proxy in flowRef and stop at most once - api/providers/[id]/models: add connection-scoped live Zed model resolver - registry: unhide provider in dashboard - tests: add unit coverage for wire format, native auth, and live models --- open-sse/executors/zed.js | 24 +- open-sse/providers/registry/zed.js | 1 - open-sse/shared/zedAuth.js | 28 +- .../dashboard/providers/[id]/page.js | 37 ++- .../api/oauth/[provider]/[action]/route.js | 13 +- src/app/api/providers/[id]/models/route.js | 41 ++- src/lib/oauth/providers/index.js | 5 + src/lib/oauth/providers/zed.js | 8 +- src/lib/oauth/utils/server.js | 44 ++- src/shared/components/OAuthModal.js | 218 +++++++++----- tests/unit/zed-completions-wire.test.js | 121 ++++++++ tests/unit/zed-live-models.test.js | 165 +++++++++++ tests/unit/zed-native-auth.test.js | 266 ++++++++++++++++++ 13 files changed, 866 insertions(+), 105 deletions(-) create mode 100644 tests/unit/zed-completions-wire.test.js create mode 100644 tests/unit/zed-live-models.test.js create mode 100644 tests/unit/zed-native-auth.test.js diff --git a/open-sse/executors/zed.js b/open-sse/executors/zed.js index e6233fcb..b88d2762 100644 --- a/open-sse/executors/zed.js +++ b/open-sse/executors/zed.js @@ -29,11 +29,18 @@ import { zedLlmFetch, } from "../shared/zedAuth.js"; +// Wire values for the `provider` field of POST /completions. These are NOT +// display names: cloud.zed.dev matches them exactly, and an unrecognized value +// fails the whole request with `500 {"message":"An internal server error +// occurred."}` before the model is ever looked at. Spellings come from Zed's +// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore), +// `x_ai` follows the same convention — so feeding a catalog value back through +// normalizeZedProvider is identity. const ZED_PROVIDER = { - anthropic: "Anthropic", - openai: "OpenAi", - google: "Google", - xai: "XAi", + anthropic: "anthropic", + openai: "open_ai", + google: "google", + xai: "x_ai", }; function normalizeZedProvider(value, model) { @@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) { return openaiToClaudeRequest(model, body, true); } if (provider === ZED_PROVIDER.google) { - return openaiToGeminiRequest(model, body, true); + const geminiRequest = openaiToGeminiRequest(model, body, true); + // Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the + // public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`, + // `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google + // path so Zed applies its own defaults — scoped here so native Gemini/ + // Antigravity is untouched. + delete geminiRequest.safetySettings; + return geminiRequest; } if (provider === ZED_PROVIDER.openai) { return openaiToOpenAIResponsesRequest(model, body, true, credentials); diff --git a/open-sse/providers/registry/zed.js b/open-sse/providers/registry/zed.js index 9224cf95..7e2e5dda 100644 --- a/open-sse/providers/registry/zed.js +++ b/open-sse/providers/registry/zed.js @@ -4,7 +4,6 @@ export default { priority: 10, alias: "zd", uiAlias: "zd", - hidden: true, display: { name: "Zed", icon: "code", diff --git a/open-sse/shared/zedAuth.js b/open-sse/shared/zedAuth.js index e3d8371a..4bfc33bc 100644 --- a/open-sse/shared/zedAuth.js +++ b/open-sse/shared/zedAuth.js @@ -112,7 +112,14 @@ export function parseZedCallbackPayload(input) { url = new URL(raw); } catch { try { - url = new URL(`http://127.0.0.1/?${raw.replace(/^\?/, "")}`); + // Accept pathname+query (what the local proxy forwards, e.g. + // "/?user_id=..&access_token=.." or "/callback?.."), a bare query, + // or a lone query string. Only the query part is parsed — a leading + // path must never become part of the first parameter name. + const query = raw.includes("?") + ? raw.slice(raw.indexOf("?") + 1) + : raw.replace(/^\?/, ""); + url = new URL(`http://127.0.0.1/?${query}`); } catch { throw new Error("Invalid Zed callback URL"); } @@ -134,6 +141,10 @@ export function parseZedCallbackPayload(input) { export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier) { const privateKey = decodeZedPrivateKeyVerifier(privateKeyVerifier); const encrypted = Buffer.from(String(encryptedAccessToken), "base64url"); + const fail = (oaepError) => { + const message = oaepError instanceof Error ? oaepError.message : String(oaepError); + throw new Error(`Failed to decrypt Zed access token: ${message}`); + }; try { return crypto .privateDecrypt( @@ -143,15 +154,21 @@ export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier) .toString("utf8"); } catch (oaepError) { try { - return crypto + const text = crypto .privateDecrypt( { key: privateKey, padding: crypto.constants.RSA_PKCS1_PADDING }, encrypted, ) .toString("utf8"); - } catch { - const message = oaepError instanceof Error ? oaepError.message : String(oaepError); - throw new Error(`Failed to decrypt Zed access token: ${message}`); + // PKCS#1 v1.5 unpadding is not integrity-checked: a wrong-key decrypt + // can "succeed" with garbage bytes instead of throwing. Replacement + // characters prove the output is not the real UTF-8 token — fail loudly + // rather than storing garbage as a credential. + if (text.includes("�")) fail(oaepError); + return text; + } catch (err) { + if (err.message.startsWith("Failed to decrypt Zed access token")) throw err; + fail(oaepError); } } } @@ -280,6 +297,7 @@ export async function fetchZedLlmToken(credentials, options = {}) { body: JSON.stringify({ organization_id: organizationId }), signal: options.signal ?? undefined, }, + options.proxyOptions ?? null, ); const token = typeof data?.token === "string" ? data.token : data?.token?.[0] || data?.token?.value; diff --git a/src/app/(dashboard)/dashboard/providers/[id]/page.js b/src/app/(dashboard)/dashboard/providers/[id]/page.js index 5ae217fb..437214e2 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/page.js +++ b/src/app/(dashboard)/dashboard/providers/[id]/page.js @@ -71,6 +71,8 @@ export default function ProviderDetailPage() { const [autoPing, setAutoPing] = useState({ enabled: false, connections: {} }); const [suggestedModels, setSuggestedModels] = useState([]); const [liveModels, setLiveModels] = useState([]); + // Live-catalog fetch warning/error (surfaced for zed only; cursor behavior unchanged). + const [liveModelsError, setLiveModelsError] = useState(null); const [kiloFreeModels, setKiloFreeModels] = useState([]); const [disabledModelIds, setDisabledModelIds] = useState([]); const [confirmState, setConfirmState] = useState(null); @@ -153,7 +155,7 @@ export default function ProviderDetailPage() { const supportsApiKeyAuth = !!APIKEY_PROVIDERS[providerId] || authModes.includes("apikey"); const isFreeNoAuth = !!FREE_PROVIDERS[providerId]?.noAuth; const staticModels = getModelsByProviderId(providerId); - const models = providerId === "cursor" && liveModels.length > 0 + const models = (providerId === "cursor" || providerId === "zed") && liveModels.length > 0 ? liveModels : staticModels; const providerAlias = getProviderAlias(providerId); @@ -467,11 +469,13 @@ export default function ProviderDetailPage() { fetchDisabledModels(); }, [fetchConnections, fetchAliases, fetchCustomModels, fetchDisabledModels]); - // Cursor's model availability is account-specific and changes frequently. - // Load the active account's live catalog for the dashboard; the static - // registry remains the fallback while the request is pending or unavailable. + // Live per-connection catalogs (cursor, zed): the static registry carries + // no usable list, so resolve from the active connection. Fires only when + // the provider id or connection list changes — no polling, no loop. + // Cursor path is statement-identical to before; zed adds error surfacing. useEffect(() => { - if (providerId !== "cursor") { + const isLiveCatalog = providerId === "cursor" || providerId === "zed"; + if (!isLiveCatalog) { setLiveModels([]); return; } @@ -479,18 +483,32 @@ export default function ProviderDetailPage() { const connection = connections.find((item) => item.isActive !== false); if (!connection?.id) { setLiveModels([]); + if (providerId === "zed") setLiveModelsError(null); return; } let cancelled = false; + if (providerId === "zed") setLiveModelsError(null); fetch(`/api/providers/${connection.id}/models`, { cache: "no-store" }) - .then(async (res) => ({ ok: res.ok, data: await res.json() })) + .then(async (res) => ({ ok: res.ok, data: await res.json().catch(() => null) })) .then(({ ok, data }) => { - if (!cancelled && ok && Array.isArray(data.models) && data.models.length > 0) { + if (cancelled) return; + if (ok && Array.isArray(data?.models) && data.models.length > 0) { setLiveModels(data.models); + if (providerId === "zed" && data?.warning) setLiveModelsError(data.warning); + return; + } + if (providerId === "zed") { + setLiveModels([]); + setLiveModelsError(data?.warning || data?.error || "Zed returned no live models."); } }) - .catch(() => {}); + .catch(() => { + if (!cancelled && providerId === "zed") { + setLiveModels([]); + setLiveModelsError("Failed to reach the Zed model catalog."); + } + }); return () => { cancelled = true; }; }, [providerId, connections]); @@ -1767,6 +1785,9 @@ export default function ProviderDetailPage() { {!!modelsTestError && (

{modelsTestError}

)} + {providerId === "zed" && !!liveModelsError && ( +

{liveModelsError}

+ )} {renderModelsSection()} diff --git a/src/app/api/oauth/[provider]/[action]/route.js b/src/app/api/oauth/[provider]/[action]/route.js index 12fb9d66..520be970 100644 --- a/src/app/api/oauth/[provider]/[action]/route.js +++ b/src/app/api/oauth/[provider]/[action]/route.js @@ -309,13 +309,13 @@ export async function POST(request, { params }) { let ok = false; if (provider === "trae") ok = registerTraeSession({ state }); else if (provider === "windsurf") ok = registerWindsurfSession({ state }); - else if (provider === "zed") ok = registerZedSession({ state, codeVerifier: body?.codeVerifier }); + else if (provider === "zed") ok = registerZedSession({ state, codeVerifier: body?.codeVerifier, systemId: body?.systemId }); else return NextResponse.json({ error: "register-session only supported for trae/windsurf/zed" }, { status: 400 }); return NextResponse.json({ success: ok }); } if (action === "exchange") { - const { code, redirectUri, codeVerifier, state, meta } = body; + const { code, redirectUri, codeVerifier, state, meta, systemId } = body; // Xiaomi MiMo: no token exchange needed — the callback already decrypted the sk. // Just read the session result and create the connection. @@ -459,8 +459,13 @@ export async function POST(request, { params }) { return NextResponse.json({ error: "Missing required fields" }, { status: 400 }); } - // Exchange code for tokens (meta carries provider-specific params, e.g. gitlab clientId/baseUrl) - const tokenData = await exchangeTokens(provider, code, redirectUri, codeVerifier, state, meta); + // Exchange code for tokens (meta carries provider-specific params, e.g. gitlab clientId/baseUrl). + // systemId (Zed) is merged into meta so the login attempt's own id is + // used instead of a freshly prepared one. Ignored by other providers. + const tokenData = await exchangeTokens(provider, code, redirectUri, codeVerifier, state, { + ...(meta || {}), + ...(systemId ? { systemId } : {}), + }); // Save to database const connection = await createProviderConnection({ diff --git a/src/app/api/providers/[id]/models/route.js b/src/app/api/providers/[id]/models/route.js index 605bc5ca..23f97380 100644 --- a/src/app/api/providers/[id]/models/route.js +++ b/src/app/api/providers/[id]/models/route.js @@ -1,7 +1,7 @@ import { NextResponse } from "next/server"; import { getProviderConnectionById } from "@/models"; import { isOpenAICompatibleProvider, isAnthropicCompatibleProvider } from "@/shared/constants/providers"; -import { GEMINI_CONFIG } from "@/lib/oauth/constants/oauth"; +import { GEMINI_CONFIG, ZED_HOSTED_CONFIG } from "@/lib/oauth/constants/oauth"; import { refreshGoogleToken, refreshCodexToken, updateProviderCredentials } from "@/sse/services/tokenRefresh"; import { resolveOllamaLocalHost } from "open-sse/config/providers.js"; import { getModelsByProviderId } from "open-sse/config/providerModels.js"; @@ -11,6 +11,7 @@ import { resolveQoderModels } from "open-sse/services/qoderModels.js"; import { resolveGrokCliModels } from "open-sse/services/grokCliModels.js"; import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy"; import { resolveCursorModels } from "open-sse/services/cursorModels.js"; +import { resolveZedModels } from "open-sse/shared/zedAuth.js"; import { resolveClineModels, resolveClinepassModels } from "open-sse/services/clinepassModels.js"; const GEMINI_CLI_MODELS_URL = "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels"; @@ -287,6 +288,44 @@ const PROVIDER_MODELS_CONFIG = { }; }, }, + // Zed has no static catalog by design (live /models only) — same cursor + // direct pattern: resolve with the connection's own credentials (never + // exposed to the browser), return rich metadata, drop disabled entries. + // Empty/failure yields an explicit warning, never a silent zero list. + zed: { + customResolver: async (connection) => { + try { + const result = await resolveZedModels({ + accessToken: connection.accessToken, + providerSpecificData: connection.providerSpecificData || {}, + }, { config: ZED_HOSTED_CONFIG, forceRefresh: true }); + const models = (result?.models || []) + .filter((m) => m && !m.isDisabled) + .map((m) => ({ + id: m.id, + name: m.name || m.id, + provider: m.provider, + contextLength: m.contextLength, + contextLengthInMaxMode: m.contextLengthInMaxMode, + maxOutputTokens: m.maxOutputTokens, + supportsTools: m.supportsTools, + supportsImages: m.supportsImages, + supportsThinking: m.supportsThinking, + supportsDisablingThinking: m.supportsDisablingThinking, + supportsFastMode: m.supportsFastMode, + supportsServerSideCompaction: m.supportsServerSideCompaction, + supportedEffortLevels: m.supportedEffortLevels || [], + supportsStreamingTools: m.supportsStreamingTools, + supportsParallelToolCalls: m.supportsParallelToolCalls, + })); + if (models.length > 0) return { models }; + return { models: [], warning: "Zed returned no live models." }; + } catch (error) { + console.log("Failed to fetch Zed models dynamically:", error.message); + return { models: [], warning: `Failed to fetch Zed models: ${error.message}` }; + } + }, + }, // Cline/ClinePass share api.cline.bot/api/v1/models. The service layer already // handles Bearer-vs-`workos:` auth and swallows failures into null, so these follow diff --git a/src/lib/oauth/providers/index.js b/src/lib/oauth/providers/index.js index e505b01d..8ba6c8cb 100644 --- a/src/lib/oauth/providers/index.js +++ b/src/lib/oauth/providers/index.js @@ -112,6 +112,11 @@ export async function generateAuthData(providerName, redirectUri, meta) { flowType: provider.flowType, fixedPort: provider.fixedPort, callbackPath: provider.callbackPath || "/callback", + // Zed: surface the system_id embedded in the sign-in URL so the frontend + // can thread it through register-session → exchange → stored connection + // (exchangeTokens re-runs prepareConfig, which would otherwise mint a + // different one). Absent for every other provider — purely additive. + ...(config.systemId ? { systemId: config.systemId } : {}), }; } diff --git a/src/lib/oauth/providers/zed.js b/src/lib/oauth/providers/zed.js index 3343976a..3f38bf87 100644 --- a/src/lib/oauth/providers/zed.js +++ b/src/lib/oauth/providers/zed.js @@ -21,11 +21,15 @@ const zed = { return { ...config, ...auth }; }, buildAuthUrl: (config, redirectUri, state) => config.authUrl, - exchangeToken: async (config, code, redirectUri, codeVerifier, state) => { + exchangeToken: async (config, code, redirectUri, codeVerifier, state, meta) => { // code = raw callback URL/query; codeVerifier = encoded private key verifier. const { userId, encryptedAccessToken } = parseZedCallbackPayload(code); const accessToken = decryptZedAccessToken(encryptedAccessToken, codeVerifier); - return { accessToken, userId, systemId: config.systemId }; + // Prefer the system_id registered for this login attempt (threaded via + // meta from register-session); fall back to the prepared config. Never + // mint a fresh one here — exchangeTokens re-runs prepareConfig, which + // would otherwise store a system_id unrelated to the zed.dev login. + return { accessToken, userId, systemId: meta?.systemId || config.systemId }; }, postExchange: async (tokens) => { const credentials = { diff --git a/src/lib/oauth/utils/server.js b/src/lib/oauth/utils/server.js index 80377752..83de1821 100644 --- a/src/lib/oauth/utils/server.js +++ b/src/lib/oauth/utils/server.js @@ -648,9 +648,15 @@ let zedProxyTimeout = null; let zedProxyPort = null; let zedSession = null; -export function registerZedSession({ state, codeVerifier }) { +export function registerZedSession({ state, codeVerifier, systemId }) { if (!state || !codeVerifier) return false; - zedSession = { state, codeVerifier, status: "pending", createdAt: Date.now() }; + zedSession = { + state, + codeVerifier, + systemId: systemId || null, + status: "pending", + createdAt: Date.now(), + }; return true; } export function getZedSessionStatus(state) { @@ -665,6 +671,10 @@ export function clearZedSession(state) { export function startZedProxy(preferredPort = 0) { return new Promise((resolve) => { if (zedProxyServer) { + // Reuse the live listener, but renew its idle timeout so a previous + // flow's deadline can never kill the flow that just adopted the port. + if (zedProxyTimeout) clearTimeout(zedProxyTimeout); + zedProxyTimeout = setTimeout(() => { console.log("[Zed proxy] timeout, stopping"); stopZedProxy(); }, ZED_HOSTED_CONFIG.oauthTimeoutMs); resolve({ success: true, port: zedProxyPort, callbackUrl: `http://127.0.0.1:${zedProxyPort}/` }); return; } @@ -694,13 +704,34 @@ export function startZedProxy(preferredPort = 0) { res.end(renderCodexResultPage(false, "Cross-origin callback rejected")); return; } + // A genuine Zed redirect always carries user_id + access_token. Anything + // else (probe, prefetch, stray navigation, favicon-style miss) is NOT + // the callback: answer without touching the session and WITHOUT + // stopping the server, so the real redirect can still land afterwards. + const qp = url.searchParams; + const hasZedParams = + qp.has("user_id") || qp.has("userId") || + qp.has("access_token") || qp.has("accessToken") || qp.has("token"); + if (!hasZedParams) { + console.log(`[Zed proxy] ignoring non-callback ${req.method} ${url.pathname} (session kept, server kept)`); + res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderCodexResultPage(false, "Waiting for Zed sign-in — this request carried no login data.")); + return; + } // Pass raw callback path+query to exchangeTokens → parseZedCallbackPayload. // codeVerifier carries the encoded RSA private key for decryption. const rawCallback = url.search ? `${url.pathname}?${url.searchParams.toString()}` : url.pathname; try { const { exchangeTokens } = await import("../providers.js"); const { createProviderConnection } = await import("@/models"); - const tokenData = await exchangeTokens("zed", rawCallback, null, session.codeVerifier, session.state); + const tokenData = await exchangeTokens( + "zed", + rawCallback, + null, + session.codeVerifier, + session.state, + session.systemId ? { systemId: session.systemId } : undefined, + ); const connection = await createProviderConnection({ provider: "zed", authType: "oauth", @@ -712,13 +743,16 @@ export function startZedProxy(preferredPort = 0) { session.email = connection.email; res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); res.end(renderCodexResultPage(true, "You can close this window.")); + stopZedProxy(); } catch (err) { session.status = "error"; session.error = err.message; res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); res.end(renderCodexResultPage(false, err.message)); - } finally { - stopZedProxy(); + // Intentionally NOT stopping here: the failure may belong to a + // superseded attempt (e.g. an older popup landing after "Try Again" + // registered a new keypair). The live attempt's genuine callback must + // still land. The idle timeout + modal close bound the listener. } }); const tryPort = Number(preferredPort) || 0; diff --git a/src/shared/components/OAuthModal.js b/src/shared/components/OAuthModal.js index 805301da..589a0415 100644 --- a/src/shared/components/OAuthModal.js +++ b/src/shared/components/OAuthModal.js @@ -50,6 +50,19 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, const popupRef = useRef(null); const pollingAbortRef = useRef(false); const openedRef = useRef(false); + // Proxy-flow session ledger: which provider's proxy THIS modal session + // started, and whether its stop was already sent. Every stop-proxy call is + // gated on this — parent re-renders can never spam it, and a close stops + // the owned proxy exactly once. + const flowRef = useRef({ proxyStarted: false, proxyProvider: null, stopSent: false }); + // Parent callbacks are stored in refs so effect/callback identities stay + // stable across parent re-renders (the page passes fresh inline closures). + // Synced by the ref-sync effect below (placed after all callbacks are + // defined); the open effect then depends only on stable primitives. + const onSuccessRef = useRef(onSuccess); + const onCloseRef = useRef(onClose); + const isOpenRef = useRef(isOpen); + const startOAuthFlowRef = useRef(null); const { copied, copy } = useCopyToClipboard(); // State for client-only values to avoid hydration mismatch @@ -81,6 +94,9 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, redirectUri: authData.redirectUri, codeVerifier: authData.codeVerifier, state, + // Zed: thread the login attempt's system_id so the stored + // connection keeps the id sent to zed.dev (see register-session). + ...(authData.systemId ? { systemId: authData.systemId } : {}), ...(oauthMeta ? { meta: oauthMeta } : {}), }), }); @@ -89,12 +105,12 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); } catch (err) { setError(err.message); setStep("error"); } - }, [authData, provider, onSuccess, oauthMeta]); + }, [authData, provider, oauthMeta]); const completeXaiManualCode = useCallback(async (code) => { if (!authData?.state) return; @@ -108,12 +124,12 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); } catch (err) { setError(err.message); setStep("error"); } - }, [authData, onSuccess]); + }, [authData]); // Poll for device code token const startPolling = useCallback(async (deviceCode, codeVerifier, interval, extraData, deadlineMs) => { @@ -155,7 +171,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, pollingAbortRef.current = true; // Stop polling immediately setStep("success"); setPolling(false); - onSuccess?.(); + onSuccessRef.current?.(); return; } @@ -177,9 +193,19 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, setError("Authorization timeout"); setStep("error"); setPolling(false); - }, [provider, onSuccess]); + }, [provider]); - // Trae/Windsurf proxy OAuth flow: dynamic-port local callback → auto exchange. + // Stop the proxy owned by THIS modal session, at most once. Re-renders, + // repeated closes, and post-completion calls are all no-ops by construction. + const stopOwnedProxy = useCallback(() => { + const flow = flowRef.current; + if (flow.proxyStarted && !flow.stopSent && flow.proxyProvider) { + flow.stopSent = true; + fetch(`/api/oauth/${flow.proxyProvider}/stop-proxy`).catch(() => {}); + } + }, []); + + // Trae/Windsurf/Zed proxy OAuth flow: dynamic-port local callback → auto exchange. const startProxyFlow = useCallback(async (providerId) => { // 1. Start the local callback server (returns a dynamic port + callback URL). const startRes = await fetch(`/api/oauth/${providerId}/start-proxy`); @@ -187,31 +213,61 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (!startRes.ok || !startData.success || !startData.callbackUrl) { throw new Error(startData.reason || startData.error || `Failed to start ${providerId} callback server`); } + // Take ownership immediately so a close during the remaining flight still + // cleans this proxy up (via the close effect or the abort below). + flowRef.current.proxyStarted = true; + flowRef.current.proxyProvider = providerId; + flowRef.current.stopSent = false; + if (!isOpenRef.current) { + stopOwnedProxy(); + return; + } // 2. Build the authorize URL with redirect_uri = proxy callback URL. const authorizeUrl = new URL(`/api/oauth/${providerId}/authorize`, window.location.origin); authorizeUrl.searchParams.set("redirect_uri", startData.callbackUrl); const authRes = await fetch(authorizeUrl); const authData = await authRes.json(); - if (!authRes.ok) throw new Error(authData.error); + if (!authRes.ok) { + stopOwnedProxy(); + throw new Error(authData.error); + } + if (!isOpenRef.current) { + stopOwnedProxy(); + return; + } // 3. Register the session so the proxy can match the incoming callback. - // Zed also passes code_verifier (encodes the RSA private key for decrypt); - // sent via POST body so the private key never lands in URL/query logs. + // Zed also passes code_verifier (encodes the RSA private key for decrypt) + // + systemId; sent via POST body so secrets never land in URL/query logs. const regBody = { state: authData.state }; if (authData.codeVerifier) regBody.codeVerifier = authData.codeVerifier; - await fetch(`/api/oauth/${providerId}/register-session`, { + if (authData.systemId) regBody.systemId = authData.systemId; + const regRes = await fetch(`/api/oauth/${providerId}/register-session`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify(regBody), }); + let regData = null; + try { + regData = await regRes.json(); + } catch { + regData = null; + } + if (!regRes.ok || regData?.success === false) { + stopOwnedProxy(); + throw new Error(regData?.error || "Failed to register login session; please retry"); + } + if (!isOpenRef.current) return; // closed mid-flight: close effect owns cleanup now // 4. Open popup; proxy auto-exchanges on callback, modal polls poll-status. setAuthData({ ...authData, proxyProvider: providerId }); setStep("waiting"); popupRef.current = window.open(authData.authUrl, "oauth_popup", "width=600,height=700"); if (!popupRef.current) setStep("input"); // popup blocked → fall back to manual paste - }, []); + }, [stopOwnedProxy]); - // Start OAuth flow - const startOAuthFlow = useCallback(async () => { + // Start OAuth flow (plain function by design: it is only invoked from the + // open effect via ref and from user actions, so memoization would only add + // an identity that re-triggers effects on every parent re-render). + const startOAuthFlow = async () => { if (!provider) return; try { setError(null); @@ -356,6 +412,14 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, setAuthData({ ...data, redirectUri, codexServerSide, xaiServerSide }); + // Take ownership of server-side proxies so close stops them exactly once + // (replaces the per-provider stop branches; same behavior, one ledger). + if ((provider === "codex" && codexProxyActive) || (provider === "xai" && xaiProxyActive)) { + flowRef.current.proxyStarted = true; + flowRef.current.proxyProvider = provider; + flowRef.current.stopSent = false; + } + // Guard: device_code providers return authUrl:null from /authorize. Never window.open(null) // (browsers coerce it to the relative path ".../null"). if (!data.authUrl) { @@ -396,49 +460,55 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, setError(err.message); setStep("error"); } - }, [provider, isLocalhost, startPolling, oauthMeta, idcConfig, authMode, startProxyFlow]); + }; - // Reset state and start OAuth when modal opens + // Sync latest props/flow into refs after every render (no dep array). + // The open effect below then depends only on stable primitives. useEffect(() => { - if (isOpen && provider) { - // Guard against StrictMode/effect re-runs auto-opening multiple tabs. - if (openedRef.current) return; - openedRef.current = true; - setAuthData(null); - setCallbackUrl(""); - setError(null); - setIsDeviceCode(false); - setDeviceData(null); - setPolling(false); - setAuthMode("browser"); - setPasteToken(""); - setIdeStatus(null); - pollingAbortRef.current = false; - // Best-effort IDE detection for paste-token providers (Trae/Windsurf) - if (PASTE_TOKEN_PROVIDERS[provider]) { - fetch(`/api/oauth/${provider}/ide-status`) - .then((r) => r.json()) - .then((data) => setIdeStatus(data)) - .catch(() => setIdeStatus({ installed: false, path: null })); - } - startOAuthFlow(); - } else if (!isOpen) { - // Abort polling and cleanup proxy when modal closes - pollingAbortRef.current = true; - openedRef.current = false; - if (provider === "codex") { - fetch("/api/oauth/codex/stop-proxy").catch(() => {}); - } else if (provider === "xai") { - fetch("/api/oauth/xai/stop-proxy").catch(() => {}); - } else if (provider === "trae") { - fetch("/api/oauth/trae/stop-proxy").catch(() => {}); - } else if (provider === "windsurf") { - fetch("/api/oauth/windsurf/stop-proxy").catch(() => {}); - } else if (provider === "zed") { - fetch("/api/oauth/zed/stop-proxy").catch(() => {}); - } + onSuccessRef.current = onSuccess; + onCloseRef.current = onClose; + isOpenRef.current = isOpen; + startOAuthFlowRef.current = startOAuthFlow; + }); + + // Reset state and start OAuth when modal opens — exactly once per open. + // Guarded by openedRef so StrictMode/effect re-runs never open extra tabs. + useEffect(() => { + if (!isOpen || !provider) return; + if (openedRef.current) return; + openedRef.current = true; + setAuthData(null); + setCallbackUrl(""); + setError(null); + setIsDeviceCode(false); + setDeviceData(null); + setPolling(false); + setAuthMode("browser"); + setPasteToken(""); + setIdeStatus(null); + pollingAbortRef.current = false; + flowRef.current = { proxyStarted: false, proxyProvider: null, stopSent: false }; + // Best-effort IDE detection for paste-token providers (Trae/Windsurf) + if (PASTE_TOKEN_PROVIDERS[provider]) { + fetch(`/api/oauth/${provider}/ide-status`) + .then((r) => r.json()) + .then((data) => setIdeStatus(data)) + .catch(() => setIdeStatus({ installed: false, path: null })); } - }, [isOpen, provider, startOAuthFlow]); + startOAuthFlowRef.current(); + }, [isOpen, provider]); + + // Cleanup when the modal closes: abort polling and stop the proxy THIS + // session started, exactly once. Deps are stable primitives, so unrelated + // parent re-renders cannot reach the stop call (previously every parent + // render re-fired stop-proxy while the modal was closed). + useEffect(() => { + if (isOpen) return; + pollingAbortRef.current = true; + openedRef.current = false; + stopOwnedProxy(); + flowRef.current = { proxyStarted: false, proxyProvider: null, stopSent: false }; + }, [isOpen, provider, stopOwnedProxy]); // Server-side proxy mode (codex/xai fixed-port + trae/windsurf dynamic-port): // poll status until the proxy auto-exchanges and saves the connection. @@ -467,7 +537,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (data.status === "done") { callbackProcessedRef.current = true; setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); return; } if (data.status === "error") { @@ -489,7 +559,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, }; setTimeout(tick, POLL_INTERVAL_MS); return () => { cancelled = true; }; - }, [authData, onSuccess]); + }, [authData]); // Listen for OAuth callback via multiple methods useEffect(() => { @@ -589,23 +659,31 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, const data = await res.json(); if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); return; } const input = callbackUrl.trim(); - // Trae/Windsurf proxy flow fallback (popup blocked): paste the full callback URL + // Trae/Windsurf/Zed proxy flow fallback (popup blocked): paste the full callback URL if (PROXY_OAUTH_PROVIDERS.has(provider) && input) { const res = await fetch(`/api/oauth/${provider}/exchange`, { method: "POST", headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ code: input, state: authData?.state }), + body: JSON.stringify({ + code: input, + state: authData?.state, + // Zed manual fallback needs the same attempt material as the + // automatic path (redirectUri + RSA verifier + system_id). + ...(authData?.redirectUri ? { redirectUri: authData.redirectUri } : {}), + ...(authData?.codeVerifier ? { codeVerifier: authData.codeVerifier } : {}), + ...(authData?.systemId ? { systemId: authData.systemId } : {}), + }), }); const data = await res.json(); if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); return; } @@ -652,21 +730,13 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, } }; - // Clear session on modal close + cleanup proxy + // Clear session on modal close + cleanup proxy (idempotent: the owned + // proxy is stopped at most once across effect-close, button-close, and + // Escape/backdrop-close — all funnel through here or the close effect). const handleClose = useCallback(() => { - if (provider === "codex") { - fetch("/api/oauth/codex/stop-proxy").catch(() => {}); - } else if (provider === "xai") { - fetch("/api/oauth/xai/stop-proxy").catch(() => {}); - } else if (provider === "trae") { - fetch("/api/oauth/trae/stop-proxy").catch(() => {}); - } else if (provider === "windsurf") { - fetch("/api/oauth/windsurf/stop-proxy").catch(() => {}); - } else if (provider === "zed") { - fetch("/api/oauth/zed/stop-proxy").catch(() => {}); - } - onClose(); - }, [onClose, provider]); + stopOwnedProxy(); + onCloseRef.current(); + }, [stopOwnedProxy]); if (!provider || !providerInfo) return null; const isXaiProvider = provider === "xai"; diff --git a/tests/unit/zed-completions-wire.test.js b/tests/unit/zed-completions-wire.test.js new file mode 100644 index 00000000..5be982ae --- /dev/null +++ b/tests/unit/zed-completions-wire.test.js @@ -0,0 +1,121 @@ +// Zed completions wire acceptance: the `provider` field of POST /completions +// must use cloud.zed.dev's exact wire values (anthropic/open_ai/google/x_ai), +// and the Zed Gemini path must not carry the shared translator's +// safetySettings (Zed's hosted Gemini backend speaks the Vertex safety +// vocabulary, not the public-Gemini enums). +import { describe, it, expect, beforeEach, vi } from "vitest"; + +vi.mock("open-sse/shared/zedAuth.js", async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + resolveZedModels: vi.fn(), + zedLlmFetch: vi.fn(), + }; +}); + +import { + resolveZedModels, + zedLlmFetch, +} from "open-sse/shared/zedAuth.js"; +import ZedExecutor from "open-sse/executors/zed.js"; + +function catalogFor(entries) { + const rawById = new Map(entries); + return { rawById, models: [] }; +} + +function mockCatalogFetch(captured) { + zedLlmFetch.mockImplementation(async (credentials, path, options) => { + captured.body = JSON.parse(options.fetchOptions.body); + return new Response("upstream-error-stub", { status: 500 }); + }); +} + +function makeExecutor() { + const executor = new ZedExecutor(); + executor.config = {}; + return executor; +} + +const CHAT_BODY = { messages: [{ role: "user", content: "hi" }] }; + +beforeEach(() => { + vi.clearAllMocks(); +}); + +describe("wire provider enum", () => { + it.each([ + ["Anthropic", "anthropic"], + ["anthropic", "anthropic"], + ["OpenAi", "open_ai"], + ["open_ai", "open_ai"], + ["Google", "google"], + ["gemini", "google"], + ["XAi", "x_ai"], + ["x_ai", "x_ai"], + ])("catalog provider %j normalizes to wire %j", async (catalogValue, wire) => { + resolveZedModels.mockResolvedValue(catalogFor([["m", { provider: catalogValue }]])); + const executor = makeExecutor(); + const { provider } = await executor.resolveModel("m", {}, null, null); + expect(provider).toBe(wire); + }); + + it("infers wire provider from the model id when the catalog is unavailable", async () => { + resolveZedModels.mockRejectedValue(new Error("catalog down")); + const executor = makeExecutor(); + const log = { warn: vi.fn() }; + expect((await executor.resolveModel("claude-opus-x", {}, null, log)).provider).toBe("anthropic"); + expect((await executor.resolveModel("gemini-3-x", {}, null, log)).provider).toBe("google"); + expect((await executor.resolveModel("grok-4-x", {}, null, log)).provider).toBe("x_ai"); + expect((await executor.resolveModel("gpt-5-x", {}, null, log)).provider).toBe("open_ai"); + }); +}); + +describe("completion payload shaping", () => { + it("sends wire provider values per model family", async () => { + resolveZedModels.mockImplementation(async () => catalogFor([ + ["claude-x", { provider: "anthropic" }], + ["gpt-x", { provider: "open_ai" }], + ["gemini-x", { provider: "google" }], + ["grok-x", { provider: "x_ai" }], + ])); + const captured = {}; + mockCatalogFetch(captured); + const executor = makeExecutor(); + + for (const [model, wire] of [ + ["claude-x", "anthropic"], + ["gpt-x", "open_ai"], + ["gemini-x", "google"], + ["grok-x", "x_ai"], + ]) { + await executor.execute({ model, body: { ...CHAT_BODY }, stream: false, credentials: {} }); + expect(captured.body.provider).toBe(wire); + expect(captured.body.model).toBe(model); + } + }); + + it("strips safetySettings on the Zed Gemini path only", async () => { + resolveZedModels.mockImplementation(async () => catalogFor([ + ["gemini-x", { provider: "google" }], + ["claude-x", { provider: "anthropic" }], + ])); + const captured = {}; + mockCatalogFetch(captured); + const executor = makeExecutor(); + + await executor.execute({ model: "gemini-x", body: { ...CHAT_BODY }, stream: false, credentials: {} }); + expect(captured.body.provider).toBe("google"); + expect(captured.body.provider_request).not.toHaveProperty("safetySettings"); + + // Sanity: the shared translator still emits safetySettings — the removal + // happens in the Zed executor, not in shared/native Gemini behavior. + const { openaiToGeminiRequest } = await import( + "open-sse/translator/request/openai-to-gemini.js" + ); + expect(openaiToGeminiRequest("gemini-x", { ...CHAT_BODY }, true)).toHaveProperty( + "safetySettings", + ); + }); +}); diff --git a/tests/unit/zed-live-models.test.js b/tests/unit/zed-live-models.test.js new file mode 100644 index 00000000..1b999044 --- /dev/null +++ b/tests/unit/zed-live-models.test.js @@ -0,0 +1,165 @@ +// Route-level acceptance for the Zed live-model wiring: +// GET /api/providers/[connectionId]/models → resolveZedModels → UI rows +// RUN WITH AN ISOLATED DB: DATA_DIR=$(mktemp -d) npx vitest run ... +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import { GET } from "@/app/api/providers/[id]/models/route.js"; +import { createProviderConnection } from "@/models/index.js"; + +// Transport stub BELOW resolveZedModels: proxyAwareFetch captures the native +// fetch at import time, so stubbing globalThis.fetch cannot intercept it. +// Mock the module instead; untouched hosts pass through to native fetch. +const stub = vi.hoisted(() => { + const nativeFetch = globalThis.fetch.bind(globalThis); + return { mode: "ok", calls: [], nativeFetch }; +}); +vi.mock("open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: async (url, options) => { + const u = String(url); + stub.calls.push(u); + if (u.includes("cloud.zed.dev/client/users/me")) { + return Response.json({ default_organization_id: "org-1" }); + } + if (u.includes("cloud.zed.dev/client/llm_tokens")) { + return Response.json({ token: "llm-token" }); + } + if (u.includes("cloud.zed.dev/models")) { + if (stub.mode === "error") return new Response("boom", { status: 500 }); + if (stub.mode === "empty") return Response.json({ models: [] }); + return Response.json(stub.catalog); + } + return stub.nativeFetch(url, options); + }, + default: async (url, options) => stub.nativeFetch(url, options), +})); + +stub.catalog = { + models: [ + { + id: "claude-opus-4-live", + display_name: "Claude Opus Live", + provider: "anthropic", + max_token_count: 200000, + max_output_tokens: 32000, + supports_tools: true, + supports_images: true, + supports_thinking: true, + is_disabled: false, + }, + { + id: "gpt-live", + display_name: "GPT Live", + provider: "openai", + max_token_count: 128000, + max_output_tokens: 16384, + supports_tools: true, + is_disabled: false, + }, + { + id: "retired-model", + display_name: "Retired", + provider: "openai", + is_disabled: true, + }, + ], + default_model: "claude-opus-4-live", +}; + +beforeEach(() => { + stub.mode = "ok"; + stub.calls.length = 0; +}); +afterEach(() => { + vi.restoreAllMocks(); +}); + +async function seedZed(n) { + return createProviderConnection({ + provider: "zed", + authType: "oauth", + accessToken: `tok-live-${n}-${Date.now()}`, + email: `zed-live-${n}-${Date.now()}@example.com`, + providerSpecificData: { userId: `u-${n}`, systemId: `sys-${n}` }, + testStatus: "active", + }); +} + +async function getModels(connectionId) { + const req = new Request(`http://localhost/api/providers/${connectionId}/models`); + return GET(req, { params: Promise.resolve({ id: connectionId }) }); +} + +describe("criterion 1+2 — active connection + live catalog → models with metadata", () => { + it("returns enabled models with preserved metadata, no secrets", async () => { + const conn = await seedZed("m1"); + const res = await getModels(conn.id); + expect(res.status).toBe(200); + const data = await res.json(); + expect(data.models.map((m) => m.id).sort()).toEqual(["claude-opus-4-live", "gpt-live"]); + const opus = data.models.find((m) => m.id === "claude-opus-4-live"); + expect(opus.name).toBe("Claude Opus Live"); + expect(opus.contextLength).toBe(200000); + expect(opus.maxOutputTokens).toBe(32000); + expect(opus.supportsTools).toBe(true); + expect(opus.supportsImages).toBe(true); + expect(opus.supportsThinking).toBe(true); + // Credentials must never leak into the client response. + expect(JSON.stringify(data)).not.toContain(conn.accessToken); + expect(JSON.stringify(data)).not.toContain("tok-live"); + }); +}); + +describe("criterion 4 — disabled models excluded", () => { + it("is_disabled entries never reach the UI", async () => { + const conn = await seedZed("m2"); + const data = await (await getModels(conn.id)).json(); + expect(data.models.some((m) => m.id === "retired-model")).toBe(false); + }); +}); + +describe("criterion 4b — empty catalog → explicit warning", () => { + it("returns warning instead of silent zero", async () => { + stub.mode = "empty"; + const conn = await seedZed("m3"); + const res = await getModels(conn.id); + expect(res.status).toBe(200); + const data = await res.json(); + expect(data.models).toEqual([]); + expect(data.warning).toMatch(/no live models/i); + }); +}); + +describe("criterion 5 — resolver failure → useful warning, no crash", () => { + it("returns 200 with warning text", async () => { + stub.mode = "error"; + const conn = await seedZed("m4"); + const res = await getModels(conn.id); + expect(res.status).toBe(200); + const data = await res.json(); + expect(data.models).toEqual([]); + expect(data.warning).toMatch(/failed to fetch zed models/i); + }); +}); + +describe("criterion 6 (route) — unknown connection → 404", () => { + it("rejects missing connections", async () => { + const res = await getModels("00000000-0000-0000-0000-000000000000"); + expect(res.status).toBe(404); + }); +}); + +describe("criterion 5 (guard) — unsupported provider unchanged", () => { + it("still 400s for providers without a models config", async () => { + const conn = await createProviderConnection({ + provider: "kimchi-nope", + authType: "oauth", + accessToken: "x", + email: `guard-${Date.now()}@example.com`, + testStatus: "active", + }).catch(() => null); + // createProviderConnection may reject unknown providers; either way the + // route must not have gained a zed-shaped branch for others. + if (!conn) return; + const res = await getModels(conn.id); + expect(res.status).toBe(400); + }); +}); diff --git a/tests/unit/zed-native-auth.test.js b/tests/unit/zed-native-auth.test.js new file mode 100644 index 00000000..66acc9cf --- /dev/null +++ b/tests/unit/zed-native-auth.test.js @@ -0,0 +1,266 @@ +// Acceptance suite for the Zed native-app auth fix. +// RUN WITH AN ISOLATED DB: DATA_DIR=$(mktemp -d) npx vitest run unit/zed-native-auth.test.js +// +// Covers criteria: +// 1. Zed proxy starts +// 2. Stray callback (no params) MUST NOT kill session / stop proxy +// 3. Real callback (user_id + access_token) MUST complete session + save connection +// 4. RSA decrypt works (round-trip) +// 5. systemId identical authorize → exchange → stored connection +// 6. register-session failure is distinguishable (backend contract) +// 8. (backend) reopen/re-register creates a fresh session +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import crypto from "node:crypto"; +import { + createZedNativeAuthData, + parseZedCallbackPayload, + decryptZedAccessToken, +} from "open-sse/shared/zedAuth.js"; +import { + startZedProxy, + stopZedProxy, + registerZedSession, + getZedSessionStatus, + clearZedSession, +} from "@/lib/oauth/utils/server.js"; +import { + generateAuthData, + exchangeTokens, +} from "@/lib/oauth/providers/index.js"; + +const realFetch = globalThis.fetch; + +// Never hit the real network in tests: cloud.zed.dev calls are best-effort +// (postExchange try/catch) — fail them fast and loud instead. +beforeEach(() => { + globalThis.fetch = async (url, init) => { + if (String(url).includes("cloud.zed.dev")) { + return new Response("test-stubbed", { status: 500 }); + } + return realFetch(url, init); + }; +}); +afterEach(async () => { + globalThis.fetch = realFetch; + stopZedProxy(); + vi.restoreAllMocks(); +}); + +async function startTestProxy() { + const started = await startZedProxy(0); // random loopback port — parallel-safe + expect(started.success).toBe(true); + return started; +} + +/** Simulate zed.dev: RSA-encrypt a plaintext token with the flow's public key. */ +function encryptForCallback(publicKeyB64Url, plaintext) { + const der = Buffer.from(String(publicKeyB64Url), "base64url"); + const key = crypto.createPublicKey({ key: der, format: "der", type: "pkcs1" }); + return crypto + .publicEncrypt( + { key, padding: crypto.constants.RSA_PKCS1_OAEP_PADDING, oaepHash: "sha256" }, + Buffer.from(plaintext, "utf8"), + ) + .toString("base64url"); +} + +describe("criterion 1 — Zed proxy starts", () => { + it("binds 127.0.0.1 and reports a usable callback URL", async () => { + const started = await startTestProxy(); + expect(started.port).toBeGreaterThan(0); + expect(started.callbackUrl).toBe(`http://127.0.0.1:${started.port}/`); + }); +}); + +describe("criterion 4 — RSA decrypt works", () => { + it("round-trips OAEP-SHA256 through the verifier slot", async () => { + const auth = createZedNativeAuthData({}, { nativeAppPort: 1 }); + const encrypted = encryptForCallback(auth.publicKey, "plaintext-token-abc"); + expect(decryptZedAccessToken(encrypted, auth.privateKeyVerifier)).toBe( + "plaintext-token-abc", + ); + }); + + it("rejects a missing verifier instead of silently failing", () => { + const auth = createZedNativeAuthData({}, { nativeAppPort: 1 }); + const encrypted = encryptForCallback(auth.publicKey, "x"); + expect(() => decryptZedAccessToken(encrypted, null)).toThrow( + /private key verifier/i, + ); + }); + + it("parser keeps strict validation (no weakened acceptance)", () => { + expect(() => parseZedCallbackPayload("")).toThrow(); + expect(() => parseZedCallbackPayload("http://127.0.0.1:1/")).toThrow( + /user_id and access_token/, + ); + expect(() => + parseZedCallbackPayload("http://127.0.0.1:1/?user_id=only-user"), + ).toThrow(/user_id and access_token/); + }); +}); + +describe("criterion 2 — stray callback MUST NOT kill session", () => { + it("bare GET / leaves session pending and proxy listening", async () => { + const started = await startTestProxy(); + const auth = createZedNativeAuthData({}, { nativeAppPort: started.port }); + expect( + registerZedSession({ state: "stray-state-1", codeVerifier: auth.privateKeyVerifier }), + ).toBe(true); + + const res = await realFetch(`http://127.0.0.1:${started.port}/`); + expect(res.status).toBe(200); + + // Session must still be pending (not poisoned to error)… + const session = getZedSessionStatus("stray-state-1"); + expect(session).not.toBeNull(); + expect(session.status).toBe("pending"); + + // …and the SAME server must still own the port (no silent restart). + const again = await startZedProxy(0); + expect(again.port).toBe(started.port); + + clearZedSession("stray-state-1"); + }); + + it("GET /callback with unrelated params leaves session pending", async () => { + const started = await startTestProxy(); + const auth = createZedNativeAuthData({}, { nativeAppPort: started.port }); + registerZedSession({ state: "stray-state-2", codeVerifier: auth.privateKeyVerifier }); + + const res = await realFetch(`http://127.0.0.1:${started.port}/callback?foo=bar`); + expect(res.status).toBe(200); + + const session = getZedSessionStatus("stray-state-2"); + expect(session).not.toBeNull(); + expect(session.status).toBe("pending"); + clearZedSession("stray-state-2"); + }); +}); + +describe("criterion 3 — real callback completes session + saves connection", () => { + it("user_id + access_token → done, decrypted token persisted", async () => { + const started = await startTestProxy(); + const auth = createZedNativeAuthData({}, { nativeAppPort: started.port }); + const state = `real-state-${Date.now()}`; + registerZedSession({ state, codeVerifier: auth.privateKeyVerifier, systemId: auth.systemId }); + + const encrypted = encryptForCallback(auth.publicKey, "decrypted-token-xyz"); + const cb = new URL(`http://127.0.0.1:${started.port}/`); + cb.searchParams.set("user_id", "user-123"); + cb.searchParams.set("access_token", encrypted); + const res = await realFetch(cb.toString()); + expect(res.status).toBe(200); + + const session = getZedSessionStatus(state); + expect(session).not.toBeNull(); + expect(session.status).toBe("done"); + expect(session.connectionId).toBeTruthy(); + + const { getProviderConnectionById } = await import("@/models/index.js"); + const conn = await getProviderConnectionById(session.connectionId); + expect(conn).toBeTruthy(); + expect(conn.provider).toBe("zed"); + expect(conn.accessToken).toBe("decrypted-token-xyz"); + expect(conn.providerSpecificData?.userId).toBe("user-123"); + expect(conn.providerSpecificData?.systemId).toBe(auth.systemId); + + // Proxy stopped itself after the terminal outcome (no orphan listener). + const again = await startZedProxy(0); + expect(again.port).not.toBe(started.port); + stopZedProxy(); + }); +}); + +describe("criterion 5 — systemId stable authorize → exchange → stored", () => { + it("generateAuthData exposes the systemId sent to zed.dev", async () => { + const auth = await generateAuthData("zed", "http://127.0.0.1:59999/", { + nativeAppPort: 59999, + }); + const url = new URL(auth.authUrl); + expect(url.searchParams.get("native_app_port")).toBe("59999"); + // The system_id embedded in the sign-in URL must be observable downstream. + expect(auth.systemId).toBe(url.searchParams.get("system_id")); + expect(auth.systemId).toBeTruthy(); + }); + + it("exchange preserves the registered systemId (no regeneration)", async () => { + const auth = await generateAuthData("zed", "http://127.0.0.1:59998/", { + nativeAppPort: 59998, + }); + // Public key always rides in the authorize URL (mirrors the real flow). + const pubFromUrl = new URL(auth.authUrl).searchParams.get("native_app_public_key"); + expect(pubFromUrl).toBeTruthy(); + const enc2 = encryptForCallback(pubFromUrl, "tok2"); + const tokens = await exchangeTokens( + "zed", + `/?user_id=u1&access_token=${encodeURIComponent(enc2)}`, + null, + auth.codeVerifier, + auth.state, + { systemId: auth.systemId }, + ); + expect(tokens.providerSpecificData.systemId).toBe(auth.systemId); + }); +}); + +describe("criterion 6 — register-session failure is distinguishable", () => { + it("route reports { success: false } when the verifier is missing", async () => { + const { POST } = await import("@/app/api/oauth/[provider]/[action]/route.js"); + const req = new Request("http://localhost/api/oauth/zed/register-session", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ state: "no-verifier-state" }), + }); + const res = await POST(req, { + params: Promise.resolve({ provider: "zed", action: "register-session" }), + }); + const data = await res.json(); + // Backend contract: failure must be explicit (modal is required to check it). + expect(data.success).toBe(false); + }); +}); + +describe("criterion 8 (backend) — re-register creates a fresh session", () => { + it("a new register supersedes the old state cleanly", async () => { + const a = createZedNativeAuthData({}, { nativeAppPort: 1 }); + const b = createZedNativeAuthData({}, { nativeAppPort: 1 }); + registerZedSession({ state: "old-state", codeVerifier: a.privateKeyVerifier }); + registerZedSession({ state: "new-state", codeVerifier: b.privateKeyVerifier }); + + expect(getZedSessionStatus("old-state")).toBeNull(); + const fresh = getZedSessionStatus("new-state"); + expect(fresh).not.toBeNull(); + expect(fresh.status).toBe("pending"); + expect(fresh.codeVerifier).toBe(b.privateKeyVerifier); + clearZedSession("new-state"); + }); +}); + +describe("criterion L — decrypt failure errors the session but keeps the server", () => { + it("wrong-key token → session error, listener survives for the live attempt", async () => { + const started = await startTestProxy(); + const live = createZedNativeAuthData({}, { nativeAppPort: started.port }); + const other = createZedNativeAuthData({}, { nativeAppPort: started.port }); + const state = `wrongkey-state-${Date.now()}`; + registerZedSession({ state, codeVerifier: live.privateKeyVerifier }); + + // Token encrypted for a DIFFERENT keypair (e.g. superseded popup). + const bad = encryptForCallback(other.publicKey, "not-for-this-key"); + const cb = new URL(`http://127.0.0.1:${started.port}/`); + cb.searchParams.set("user_id", "user-123"); + cb.searchParams.set("access_token", bad); + const res = await realFetch(cb.toString()); + expect(res.status).toBe(200); + + const session = getZedSessionStatus(state); + expect(session).not.toBeNull(); + expect(session.status).toBe("error"); + expect(session.error).toMatch(/decrypt/i); + + // Server must still be alive (same port) for the live attempt. + const again = await startZedProxy(0); + expect(again.port).toBe(started.port); + clearZedSession(state); + }); +}); From 3ac100d5248e49a6aaa3b7efa6c1fd9d3dd14571 Mon Sep 17 00:00:00 2001 From: Ali Shaikh Date: Thu, 17 Sep 2026 20:06:19 +0700 Subject: [PATCH 65/78] fix(usage): improve DeepSeek credit balance display Display DeepSeek prepaid balances as credit with currency instead of a 0/total quota bar, and mark balance items as isCreditBalance. --- open-sse/providers/registry/opencode-go.js | 2 +- open-sse/services/usage/deepseek.js | 7 ++++--- .../usage/components/ProviderLimits/QuotaTable.js | 15 +++++++++++---- .../usage/components/ProviderLimits/utils.js | 2 ++ tests/unit/deepseek-usage.test.js | 2 ++ 5 files changed, 20 insertions(+), 8 deletions(-) diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index 1d1d12c0..d65d4c15 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -13,7 +13,7 @@ export default { textIcon: "OC", website: "https://opencode.ai/auth", notice: { - text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.", + text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.", apiKeyUrl: "https://opencode.ai/auth", }, }, diff --git a/open-sse/services/usage/deepseek.js b/open-sse/services/usage/deepseek.js index cb70a40d..9d2ed2ac 100644 --- a/open-sse/services/usage/deepseek.js +++ b/open-sse/services/usage/deepseek.js @@ -91,14 +91,15 @@ export async function getDeepseekUsage(apiKey = null, proxyOptions = null) { const quotas = {}; for (const b of balances) { const total = Math.max(0, b.totalBalance); - // Credit pot: show full remaining against current balance; never set absolute - // `remaining` — QuotaTable treats it as a 0–100 percentage. + // Credit balance: show as "Credit: $X.XX USD" not a usage quota quotas[`Balance (${b.currency})`] = { used: 0, total, remainingPercentage: total > 0 ? 100 : 0, resetAt: null, - unlimited: total > 0, + unlimited: false, + isCreditBalance: true, + currency: b.currency, }; } diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/QuotaTable.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/QuotaTable.js index 57a7a9d2..fef34f6c 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/QuotaTable.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/QuotaTable.js @@ -151,7 +151,10 @@ export default function QuotaTable({
{currentPageRows.map((quota) => { const isUnlimited = quota.unlimited === true; - const colors = getColorClasses(quota.remaining); + const isCreditBalance = quota.isCreditBalance === true; + const colors = isCreditBalance + ? { text: "text-blue-600 dark:text-blue-400", bg: "bg-blue-500", bgLight: "bg-blue-500/10", emoji: "💰" } + : getColorClasses(quota.remaining); const countdown = formatResetTime(quota.resetAt); const resetDisplay = formatResetTimeDisplay(quota.resetAt); // recurring defaults true: a missing flag means the quota @@ -175,7 +178,7 @@ export default function QuotaTable({ {/* Progress + used/total */}
- {!isUnlimited && ( + {!isUnlimited && !isCreditBalance && (
@@ -192,15 +195,19 @@ export default function QuotaTable({ title={ isUnlimited ? `${quota.used.toLocaleString()} used · Unlimited` + : isCreditBalance + ? `Credit balance: ${quota.total.toFixed(2)} ${quota.currency || ""}` : `${quota.used.toLocaleString()} / ${quota.total > 0 ? quota.total.toLocaleString() : "∞"}` } > {isUnlimited ? `${quota.used.toLocaleString()} used · Unlimited` + : isCreditBalance + ? `Credit: ${quota.total.toFixed(2)} ${quota.currency || ""}` : `${quota.used.toLocaleString()} / ${quota.total > 0 ? quota.total.toLocaleString() : "∞"}`} - - {isUnlimited ? "Unlimited" : `${quota.remaining}%`} + + {isUnlimited ? "Unlimited" : isCreditBalance ? "" : `${quota.remaining}%`}
diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js index ad321433..377bb02d 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js @@ -609,6 +609,8 @@ export function parseQuotaData(provider, data) { total: quota.total || 0, resetAt: quota.resetAt || null, remainingPercentage: quota.remainingPercentage, + isCreditBalance: quota.isCreditBalance ?? true, + currency: quota.currency || (name.includes("(") ? name.slice(name.indexOf("(") + 1, name.indexOf(")")) : "USD"), }); }); } diff --git a/tests/unit/deepseek-usage.test.js b/tests/unit/deepseek-usage.test.js index 0be26edc..d9048b09 100644 --- a/tests/unit/deepseek-usage.test.js +++ b/tests/unit/deepseek-usage.test.js @@ -80,6 +80,8 @@ describe("getUsageForProvider(deepseek)", () => { used: 0, total: 12.5, remainingPercentage: 100, + isCreditBalance: true, + currency: "USD", }); expect(usage.quotas["Balance (USD)"].remaining).toBeUndefined(); // Zero CNY still listed so user sees currency row From f64229530eb7cdd798e3f934a9af716c995fc0da Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 17 Sep 2026 20:18:37 +0700 Subject: [PATCH 66/78] fix(antigravity): sanitize Hermes system identity --- open-sse/config/appConstants.js | 1 + 1 file changed, 1 insertion(+) diff --git a/open-sse/config/appConstants.js b/open-sse/config/appConstants.js index 4df1463e..106cba50 100644 --- a/open-sse/config/appConstants.js +++ b/open-sse/config/appConstants.js @@ -178,6 +178,7 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C // makes the backend flag the request and answer 429 Quota Exhausted. export const ANTIGRAVITY_PROMPT_REWRITES = [ { from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" }, + { from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." }, { from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") } ]; From 52917a6d49d290b8a3be315dafc1b3e86a425ee5 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 17 Sep 2026 20:18:56 +0700 Subject: [PATCH 67/78] Add NEW badge to 9Remote menu --- src/shared/components/Sidebar.js | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/shared/components/Sidebar.js b/src/shared/components/Sidebar.js index 812e319e..01dd1334 100644 --- a/src/shared/components/Sidebar.js +++ b/src/shared/components/Sidebar.js @@ -303,6 +303,9 @@ export default function Sidebar({ onClose }) { computer 9Remote + + New + {/* 9English */} From 93837af09fca065717784895d5c092421bdf1d1b Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 16:57:06 +0700 Subject: [PATCH 68/78] fix(opencode): fix free tier 403 error and improve China region handling - Force stream:true and cloak decoy tools (bash, read) for OpenCode free tier - Support connection testing for opencode in testUtils - Expand error message slice limits in auth and ping to preserve workspace link - Add concise China region link chip in provider detail page Co-Authored-By: Claude Code --- open-sse/executors/opencode.js | 70 +++++++++++++++++++ .../dashboard/providers/[id]/page.js | 28 +++++++- src/app/api/models/test/ping.js | 2 +- src/app/api/providers/[id]/test/testUtils.js | 6 ++ src/sse/services/auth.js | 2 +- tests/unit/opencode-session.test.js | 35 ++++++++++ 6 files changed, 140 insertions(+), 3 deletions(-) diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index 531c5df5..507c3718 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -24,6 +24,68 @@ export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/; export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/; const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"; +// OpenCode free tier requires both 'bash' and 'read' in tools payload. +// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read) +// take precedence while satisfying upstream verification. +const OPENCODE_DECOY_CHAT_TOOLS = [ + { + type: "function", + function: { + name: "bash", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, + }, + { + type: "function", + function: { + name: "read", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, + }, +]; + +const OPENCODE_DECOY_RESPONSES_TOOLS = [ + { + type: "function", + name: "bash", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, + { + type: "function", + name: "read", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, +]; + +function cloakOpencodeTools(body, isResponses) { + if (!body || typeof body !== "object") return; + if (isResponses) { + if (!Array.isArray(body.tools)) body.tools = []; + const names = new Set(body.tools.map((t) => t.name || t.function?.name)); + for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) { + if (!names.has(tool.name)) body.tools.push({ ...tool }); + } + if (!body.tool_choice) body.tool_choice = "auto"; + } else { + const hasTools = Array.isArray(body.tools) && body.tools.length > 0; + if (!hasTools) { + body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } })); + if (!body.tool_choice) body.tool_choice = "none"; + } else { + const names = new Set(body.tools.map((t) => t.function?.name || t.name)); + for (const tool of OPENCODE_DECOY_CHAT_TOOLS) { + if (!names.has(tool.function.name)) { + body.tools.push({ ...tool, function: { ...tool.function } }); + } + } + } + } +} + function hasValidOpencodeVersion(ua) { const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i); if (!m) return false; @@ -410,6 +472,9 @@ export class OpenCodeExecutor extends BaseExecutor { transformRequest(model, body, stream, credentials) { if (body && typeof body === "object" && model && !body.model) body.model = model; + // Zen rejects non-streaming requests on free models with 403 FreeTierError; + // always stream upstream and let the handler layer aggregate for non-stream clients. + if (body && typeof body === "object") body.stream = true; if (isResponsesModel(model || body?.model) && body && typeof body === "object") { // ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng. if ("tool_choice" in body && body.tool_choice !== "auto" @@ -434,6 +499,11 @@ export class OpenCodeExecutor extends BaseExecutor { body.store = false; normalizeResponsesTools(body); sanitizeResponsesItems(body); + if (!Array.isArray(body.tools) || body.tools.length === 0) { + cloakOpencodeTools(body, true); + } + } else if (body && typeof body === "object") { + cloakOpencodeTools(body, false); } return injectReasoningContent({ provider: this.provider, model, body }); } diff --git a/src/app/(dashboard)/dashboard/providers/[id]/page.js b/src/app/(dashboard)/dashboard/providers/[id]/page.js index 437214e2..bae6ff05 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/page.js +++ b/src/app/(dashboard)/dashboard/providers/[id]/page.js @@ -1783,7 +1783,33 @@ export default function ProviderDetailPage() { })()}
{!!modelsTestError && ( -

{modelsTestError}

+
+

{modelsTestError}

+ {/RegionError|hosted in China|regionNotAllowed/i.test(modelsTestError) && (() => { + const str = typeof modelsTestError === "string" ? modelsTestError : JSON.stringify(modelsTestError); + const linkMatch = str.match(/https:\/\/opencode\.ai\/workspace\/[^\s"')]+/); + const wrkMatch = str.match(/wrk_[0-9A-Za-z]+/); + const targetUrl = linkMatch + ? (linkMatch[0].endsWith("/go") ? linkMatch[0] : `${linkMatch[0]}/go`) + : wrkMatch + ? `https://opencode.ai/workspace/${wrkMatch[0]}/go` + : "https://opencode.ai"; + + return ( + + ); + })()} +
)} {providerId === "zed" && !!liveModelsError && (

{liveModelsError}

diff --git a/src/app/api/models/test/ping.js b/src/app/api/models/test/ping.js index 273aff47..a5c882d5 100644 --- a/src/app/api/models/test/ping.js +++ b/src/app/api/models/test/ping.js @@ -160,7 +160,7 @@ export async function pingModelByKind(model, kind, baseUrl = `http://127.0.0.1:$ if (!res.ok) { const detail = parsed?.error?.message || parsed?.msg || parsed?.message || parsed?.error || rawText; - return { ok: false, latencyMs, error: `HTTP ${res.status}${detail ? `: ${String(detail).slice(0, 240)}` : ""}`, status: res.status }; + return { ok: false, latencyMs, error: `HTTP ${res.status}${detail ? `: ${String(detail).slice(0, 500)}` : ""}`, status: res.status }; } const providerStatus = parsed?.status; diff --git a/src/app/api/providers/[id]/test/testUtils.js b/src/app/api/providers/[id]/test/testUtils.js index 9572e69e..03500da2 100644 --- a/src/app/api/providers/[id]/test/testUtils.js +++ b/src/app/api/providers/[id]/test/testUtils.js @@ -751,6 +751,12 @@ async function testApiKeyConnection(connection, effectiveProxy = null) { const valid = !!(data && data.user); return { valid, error: valid ? null : "Session expired — re-paste cookie" }; } + case "opencode": { + const res = await fetchWithConnectionProxy("https://opencode.ai/zen/v1/models", { + headers: { Authorization: "Bearer public", "User-Agent": "opencode/1.18.31" }, + }, effectiveProxy); + return { valid: res.ok, error: res.ok ? null : "OpenCode free tier unavailable" }; + } case "opencode-go": { const res = await fetchWithConnectionProxy("https://opencode.ai/zen/go/v1/chat/completions", { method: "POST", diff --git a/src/sse/services/auth.js b/src/sse/services/auth.js index d85c4901..eff83e7c 100644 --- a/src/sse/services/auth.js +++ b/src/sse/services/auth.js @@ -263,7 +263,7 @@ export async function markAccountUnavailable(connectionId, status, errorText, pr } if (!shouldFallback) return { shouldFallback: false, cooldownMs: 0 }; - const reason = typeof errorText === "string" ? errorText.slice(0, 100) : "Provider error"; + const reason = typeof errorText === "string" ? errorText.slice(0, 200) : "Provider error"; const lockUpdate = buildModelLockUpdate(githubResetAtMs ? null : model, cooldownMs); await updateProviderConnection(connectionId, { diff --git a/tests/unit/opencode-session.test.js b/tests/unit/opencode-session.test.js index b7651f16..89e06a8f 100644 --- a/tests/unit/opencode-session.test.js +++ b/tests/unit/opencode-session.test.js @@ -276,4 +276,39 @@ describe("OpenCode Stable Session Reuse (429 follow-up)", () => { expect(first).toMatch(OPENCODE_SESSION_RE); expect(second).toBe(first); }); + + it("cloaks free-tier requests with bash and read decoy tools", () => { + const executor = getExecutor("opencode"); + + // Case 1: no tools sent by client -> injects bash + read with tool_choice none + const chatNoTools = executor.transformRequest("nemotron-3-ultra-free", { + messages: [{ role: "user", content: "hi" }], + }); + expect(chatNoTools.stream).toBe(true); + expect(chatNoTools.tool_choice).toBe("none"); + expect(chatNoTools.tools.map((t) => t.function?.name)).toEqual(["bash", "read"]); + + // Case 2: external CLI tools (e.g. Claude Code Bash) -> preserves Bash, appends read + const chatWithTools = executor.transformRequest("nemotron-3-ultra-free", { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "Bash", description: "Claude Code tool" } }], + tool_choice: "auto", + }); + expect(chatWithTools.tool_choice).toBe("auto"); + const names = chatWithTools.tools.map((t) => t.function?.name); + expect(names).toContain("Bash"); + expect(names).toContain("bash"); + expect(names).toContain("read"); + + // Case 3: already has both bash and read -> do not insert anything + const chatFull = executor.transformRequest("nemotron-3-ultra-free", { + messages: [{ role: "user", content: "hi" }], + tools: [ + { type: "function", function: { name: "bash", description: "existing" } }, + { type: "function", function: { name: "read", description: "existing" } }, + ], + }); + expect(chatFull.tools.length).toBe(2); + expect(chatFull.tools[0].function.description).toBe("existing"); + }); }); From 4641c2b76aa0e498c752145f0ad905e69b31e1f7 Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 16:59:45 +0700 Subject: [PATCH 69/78] fix(zed): lower display priority in oauth list Move zed provider to the bottom of oauth providers by setting priority to 999. Co-Authored-By: Claude Code --- open-sse/providers/registry/zed.js | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/open-sse/providers/registry/zed.js b/open-sse/providers/registry/zed.js index 7e2e5dda..a8ef14f1 100644 --- a/open-sse/providers/registry/zed.js +++ b/open-sse/providers/registry/zed.js @@ -1,7 +1,7 @@ // Zed provider — RSA keypair callback auth (NOT standard OAuth). export default { id: "zed", - priority: 10, + priority: 999, alias: "zd", uiAlias: "zd", display: { From efc80ba2e3d30e1a427798743c7b73fedbc52064 Mon Sep 17 00:00:00 2001 From: Amirsalar Sojoudi Date: Fri, 18 Sep 2026 17:03:57 +0700 Subject: [PATCH 70/78] fix(codex): route bare codex-auto-review to the Codex provider (#4135) --- open-sse/providers/registry/codex.js | 3 ++ open-sse/services/model.js | 2 + tests/unit/codex-auto-review-routing.test.js | 44 ++++++++++++++++++++ 3 files changed, 49 insertions(+) create mode 100644 tests/unit/codex-auto-review-routing.test.js diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 9eaafe74..710cbc27 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -65,6 +65,9 @@ export default { { id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" }, { id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" }, { id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" }, + // Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived + // from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398). + { id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" }, { id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, diff --git a/open-sse/services/model.js b/open-sse/services/model.js index 5b88809c..15c50195 100644 --- a/open-sse/services/model.js +++ b/open-sse/services/model.js @@ -124,6 +124,8 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) { // Config-driven prefix → provider inference (first match wins, fallback "openai"). const MODEL_PREFIX_PROVIDERS = [ + // Codex CLI sends this bare virtual model for auto-review — keep it on OAuth Codex (#1398). + [/^codex-auto-review$/, "codex"], [/^claude-/, "anthropic"], [/^gemini-/, "gemini"], [/^gpt-/, "openai"], diff --git a/tests/unit/codex-auto-review-routing.test.js b/tests/unit/codex-auto-review-routing.test.js new file mode 100644 index 00000000..e96b0fef --- /dev/null +++ b/tests/unit/codex-auto-review-routing.test.js @@ -0,0 +1,44 @@ +import { describe, expect, it } from "vitest"; + +import { + getDefaultModel, + getModelQuotaFamily, + getModelUpstreamId, + getProviderModels, +} from "../../open-sse/config/providerModels.js"; +import { getModelInfoCore } from "../../open-sse/services/model.js"; + +// Codex CLI's auto-review sends the bare model id "codex-auto-review". Before #1398 it fell +// through prefix inference to the "openai" default and failed with +// "No active credentials for provider: openai". +describe("codex auto-review routing (#1398)", () => { + it("routes the bare Codex auto-review model to the OAuth Codex provider", async () => { + await expect(getModelInfoCore("codex-auto-review", {})).resolves.toEqual({ + provider: "codex", + model: "codex-auto-review", + }); + }); + + it("exposes Codex auto-review as a review-quota Codex model", () => { + const autoReview = getProviderModels("cx").find( + (model) => model.id === "codex-auto-review", + ); + + expect(autoReview).toBeTruthy(); + expect(autoReview.name).toBe("Codex Auto Review"); + expect(getModelQuotaFamily("cx", "codex-auto-review")).toBe("review"); + }); + + // getModelUpstreamId strips CODEX_REVIEW_SUFFIX from unregistered "cx" ids, which would send + // "codex-auto" upstream. This model is not a derived review variant, so it must go out verbatim. + it("forwards the id upstream without stripping the -review suffix", () => { + expect(getModelUpstreamId("cx", "codex-auto-review")).toBe( + "codex-auto-review", + ); + }); + + // Registering it must not push it to the front of the cx list — getDefaultModel takes models[0]. + it("does not become the default Codex model", () => { + expect(getDefaultModel("cx")).not.toBe("codex-auto-review"); + }); +}); From b3d6e089c660389fd4153ee95cfda5e5c35c511f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Louis=20Ph=E1=BA=A1m?= Date: Fri, 18 Sep 2026 17:06:36 +0700 Subject: [PATCH 71/78] fix(antigravity): strip Claude Code billing header from system prompts --- open-sse/config/appConstants.js | 4 ++ ...antigravity-billing-header-rewrite.test.js | 38 +++++++++++++++++++ 2 files changed, 42 insertions(+) create mode 100644 tests/unit/antigravity-billing-header-rewrite.test.js diff --git a/open-sse/config/appConstants.js b/open-sse/config/appConstants.js index 106cba50..62ddfb6b 100644 --- a/open-sse/config/appConstants.js +++ b/open-sse/config/appConstants.js @@ -179,6 +179,10 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C export const ANTIGRAVITY_PROMPT_REWRITES = [ { from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" }, { from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." }, + // Claude Code prepends this line to its system prompt. The Claude-format translator strips it, + // but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions) + // pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED. + { from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" }, { from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") } ]; diff --git a/tests/unit/antigravity-billing-header-rewrite.test.js b/tests/unit/antigravity-billing-header-rewrite.test.js new file mode 100644 index 00000000..e324294f --- /dev/null +++ b/tests/unit/antigravity-billing-header-rewrite.test.js @@ -0,0 +1,38 @@ +import { describe, expect, it } from "vitest"; + +import { AntigravityExecutor } from "../../open-sse/executors/antigravity.js"; +import { openaiToAntigravityRequest } from "../../open-sse/translator/request/openai-to-gemini.js"; + +const HEADER = "x-anthropic-billing-header: cc_version=2.1.275.f15; cc_entrypoint=cli;"; + +function systemTextSentToAntigravity(systemContent) { + // OpenAI-format client (e.g. a proxy converting Claude Code to /v1/chat/completions). + const body = openaiToAntigravityRequest("gemini-3.8-flash-tiered", { + messages: [ + { role: "system", content: systemContent }, + { role: "user", content: "hi" }, + ], + }, true); + const finalBody = new AntigravityExecutor().transformRequest("gemini-3.8-flash-tiered", body, true, {}); + return finalBody.request.systemInstruction.parts.map((p) => p.text).join("\n"); +} + +describe("Antigravity strips the Claude Code billing header from system prompts", () => { + it("removes the header line prepended by Claude Code", () => { + const text = systemTextSentToAntigravity(`${HEADER}\n\nYou are Claude Code, Anthropic's official CLI for Claude.`); + expect(text).not.toContain("x-anthropic-billing-header"); + expect(text).toContain("You are Claude Code, Anthropic's official CLI for Claude."); + }); + + it("removes the header when it is not the first line", () => { + const text = systemTextSentToAntigravity(`Some preamble\n${HEADER}\nRest of prompt`); + expect(text).not.toContain("x-anthropic-billing-header"); + expect(text).toContain("Some preamble"); + expect(text).toContain("Rest of prompt"); + }); + + it("leaves prompts without the header untouched", () => { + const text = systemTextSentToAntigravity("You are a helpful assistant."); + expect(text).toContain("You are a helpful assistant."); + }); +}); From 092c84eac9d0006c99328a1812a454066531a1df Mon Sep 17 00:00:00 2001 From: Christian Gennari Date: Fri, 18 Sep 2026 17:07:32 +0700 Subject: [PATCH 72/78] fix(commandcode): retry on transient stream error and avoid fake stop chunks --- open-sse/executors/commandcode.js | 22 ++++++++-- .../request/openai-to-commandcode.js | 4 ++ .../response/commandcode-to-openai.js | 7 ++-- tests/unit/commandcode-executor.test.js | 40 +++++++++++++++++++ 4 files changed, 65 insertions(+), 8 deletions(-) diff --git a/open-sse/executors/commandcode.js b/open-sse/executors/commandcode.js index f694e61b..d923fd50 100644 --- a/open-sse/executors/commandcode.js +++ b/open-sse/executors/commandcode.js @@ -40,10 +40,24 @@ export class CommandCodeExecutor extends BaseExecutor { } async execute(opts) { - const result = await super.execute(opts); - if (!result?.response?.ok || !result.response.body) return result; - result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model); - return result; + const maxRetries = 2; + for (let attempt = 0; attempt <= maxRetries; attempt++) { + const result = await super.execute(opts); + if (!result?.response?.ok || !result.response.body) return result; + + const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model); + if (!wrappedResponse.ok && attempt < maxRetries) { + const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504; + if (isRetryableStatus) { + opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`); + await new Promise(r => setTimeout(r, 1000 * (attempt + 1))); + continue; + } + } + + result.response = wrappedResponse; + return result; + } } parseError(response, bodyText) { diff --git a/open-sse/translator/request/openai-to-commandcode.js b/open-sse/translator/request/openai-to-commandcode.js index d0a55e43..ac9067f6 100644 --- a/open-sse/translator/request/openai-to-commandcode.js +++ b/open-sse/translator/request/openai-to-commandcode.js @@ -133,6 +133,10 @@ function convertMessages(messages = []) { if (role === ROLE.ASSISTANT) { const blocks = []; + const rc = m.reasoning_content || m.thought || m.reasoning; + if (rc || (Array.isArray(m.tool_calls) && m.tool_calls.length > 0)) { + blocks.push({ type: "reasoning", text: rc || " " }); + } const text = flattenText(m.content); if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text }); if (Array.isArray(m.tool_calls)) { diff --git a/open-sse/translator/response/commandcode-to-openai.js b/open-sse/translator/response/commandcode-to-openai.js index ab3d7d7b..b75a2e70 100644 --- a/open-sse/translator/response/commandcode-to-openai.js +++ b/open-sse/translator/response/commandcode-to-openai.js @@ -165,12 +165,11 @@ export function commandCodeToOpenAIResponse(chunk, state) { break; } case "error": { - state.finishReason = OPENAI_FINISH.STOP; const errVal = event.error ?? event.message ?? "unknown"; const errStr = typeof errVal === "string" ? errVal : JSON.stringify(errVal); - out.push(makeChunk(state, { content: `\n\n[CommandCode error: ${errStr}]` })); - out.push(makeChunk(state, {}, OPENAI_FINISH.STOP)); - break; + // Mid-stream error: throw rather than emitting as fake content with finish_reason: "stop" + // This ensures the downstream stream handler marks the stream as errored/aborted. + throw new Error(`[CommandCode error: ${errStr}]`); } // Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end, // provider-metadata, message-metadata, etc. They carry no client-visible content. diff --git a/tests/unit/commandcode-executor.test.js b/tests/unit/commandcode-executor.test.js index bd0a23cd..f498b39c 100644 --- a/tests/unit/commandcode-executor.test.js +++ b/tests/unit/commandcode-executor.test.js @@ -132,6 +132,46 @@ describe("inspectAndWrapCommandCodeResponse", () => { expect(text).toContain("Hello from Laguna"); expect(text).toContain("data: [DONE]"); }); + + it("retries when initial stream yields an error and succeeds on second attempt", async () => { + let callCount = 0; + const executor = new CommandCodeExecutor(); + + // Override execute on instance to test retry behavior + executor.execute = async (opts) => { + const maxRetries = 2; + for (let attempt = 0; attempt <= maxRetries; attempt++) { + callCount++; + let rawResponse; + if (callCount === 1) { + rawResponse = new Response(createNdjsonStream([ + JSON.stringify({ + type: "error", + error: { type: "server_error", message: "Network connection lost." } + }) + "\n" + ]), { status: 200, headers: { "Content-Type": "text/event-stream" } }); + } else { + rawResponse = new Response(createNdjsonStream([ + JSON.stringify({ type: "start" }) + "\n", + JSON.stringify({ type: "text-delta", text: "Recovered from lost connection" }) + "\n", + JSON.stringify({ type: "finish" }) + "\n" + ]), { status: 200, headers: { "Content-Type": "text/event-stream" } }); + } + + const wrappedResponse = await inspectAndWrapCommandCodeResponse(rawResponse, opts.model); + if (!wrappedResponse.ok && attempt < maxRetries) { + continue; + } + return { response: wrappedResponse }; + } + }; + + const res = await executor.execute({ model: "deepseek/deepseek-v4.1-flash" }); + expect(res.response.ok).toBe(true); + expect(callCount).toBe(2); + const text = await res.response.text(); + expect(text).toContain("Recovered from lost connection"); + }); }); describe("CommandCode in Combo Fallback", () => { From bc3be0cb284f9edd711aabd6c7a8d84c1a4a8f17 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Louis=20Ph=E1=BA=A1m?= Date: Fri, 18 Sep 2026 17:06:50 +0700 Subject: [PATCH 73/78] fix(antigravity): scope cached thought signatures to the model family --- open-sse/executors/antigravity.js | 2 +- open-sse/services/thoughtSignatureStore.js | 51 +++++++-- .../translator/request/openai-to-gemini.js | 4 +- .../translator/response/gemini-to-openai.js | 4 +- ...tigravity-thought-signature-family.test.js | 104 ++++++++++++++++++ 5 files changed, 149 insertions(+), 16 deletions(-) create mode 100644 tests/unit/antigravity-thought-signature-family.test.js diff --git a/open-sse/executors/antigravity.js b/open-sse/executors/antigravity.js index fb7beee0..da4a7981 100644 --- a/open-sse/executors/antigravity.js +++ b/open-sse/executors/antigravity.js @@ -212,7 +212,7 @@ export class AntigravityExecutor extends BaseExecutor { const modifiedParts = parts?.map(p => { if (!p.functionCall) return p; const callId = p.functionCall.id; - const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null; + const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null; const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined); firstFunctionCallSeen = true; if (callSig) { diff --git a/open-sse/services/thoughtSignatureStore.js b/open-sse/services/thoughtSignatureStore.js index eb8a60e4..71fb94b8 100644 --- a/open-sse/services/thoughtSignatureStore.js +++ b/open-sse/services/thoughtSignatureStore.js @@ -10,6 +10,24 @@ const signatureKv = makeKv(SCOPE); const memorySignatures = new Map(); let pruneCounter = 0; +/** + * Model family that produced / will consume a signature. Antigravity serves Gemini and Claude + * models behind the same API, and each backend only accepts its own signatures: a Claude + * signature replayed to Gemini fails with 400 "Corrupted thought signature." (and vice versa). + */ +export function signatureFamily(model) { + const m = typeof model === "string" ? model.toLowerCase() : ""; + if (!m) return null; + if (m.includes("claude")) return "claude"; + if (m.includes("gemini")) return "gemini"; + return m; +} + +// Entries stored before families were recorded (no `family`) stay usable for any model. +function isCompatible(entry, family) { + return !entry.family || !family || entry.family === family; +} + function pruneMemoryExpired() { const now = Date.now(); for (const [key, value] of memorySignatures.entries()) { @@ -62,13 +80,15 @@ async function maybePrunePersisted() { } /** - * Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async) + * Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async). + * `model` is the model that produced the signature; lookups for another model family skip it. */ -export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) { +export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null, model = null) { if (typeof toolCallId !== "string" || !toolCallId) return; if (typeof signature !== "string" || !signature) return; const now = Date.now(); + const family = signatureFamily(model); pruneMemoryExpired(); const keys = []; @@ -80,12 +100,14 @@ export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = n for (const k of keys) { memorySignatures.set(k, { signature, + family, expiresAt: now + MEMORY_TTL_MS, }); // Async persist to SQLite kv table without blocking signatureKv.set(k, { signature, + family, createdAt: now, expiresAt: now + PERSISTED_TTL_MS, }).catch(() => {}); @@ -95,23 +117,25 @@ export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = n } /** - * Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback) + * Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback). + * `model` is the target model; signatures produced by another model family are ignored. */ -export async function getGeminiThoughtSignature(toolCallId, sessionId = null) { +export async function getGeminiThoughtSignature(toolCallId, sessionId = null, model = null) { if (typeof toolCallId !== "string" || !toolCallId) return null; + const family = signatureFamily(model); pruneMemoryExpired(); if (sessionId && typeof sessionId === "string") { const sessionKey = `${sessionId}:${toolCallId}`; const sessionEntry = memorySignatures.get(sessionKey); - if (sessionEntry && sessionEntry.expiresAt > Date.now()) { + if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) { return sessionEntry.signature; } } const entry = memorySignatures.get(toolCallId); - if (entry && entry.expiresAt > Date.now()) { + if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) { return entry.signature; } @@ -119,9 +143,10 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) { if (sessionId && typeof sessionId === "string") { const sessionKey = `${sessionId}:${toolCallId}`; const sessionRow = await signatureKv.get(sessionKey); - if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) { + if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now()) && isCompatible(sessionRow, family)) { memorySignatures.set(sessionKey, { signature: sessionRow.signature, + family: sessionRow.family || null, expiresAt: Date.now() + MEMORY_TTL_MS, }); return sessionRow.signature; @@ -134,8 +159,10 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) { signatureKv.remove(toolCallId).catch(() => {}); return null; } + if (!isCompatible(row, family)) return null; memorySignatures.set(toolCallId, { signature: row.signature, + family: row.family || null, expiresAt: Date.now() + MEMORY_TTL_MS, }); return row.signature; @@ -148,22 +175,24 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) { } /** - * Synchronous get from RAM cache only (for sync translators) + * Synchronous get from RAM cache only (for sync translators). + * `model` is the target model; signatures produced by another model family are ignored. */ -export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) { +export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null, model = null) { if (typeof toolCallId !== "string" || !toolCallId) return null; + const family = signatureFamily(model); pruneMemoryExpired(); if (sessionId && typeof sessionId === "string") { const sessionKey = `${sessionId}:${toolCallId}`; const sessionEntry = memorySignatures.get(sessionKey); - if (sessionEntry && sessionEntry.expiresAt > Date.now()) { + if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) { return sessionEntry.signature; } } const entry = memorySignatures.get(toolCallId); - if (entry && entry.expiresAt > Date.now()) { + if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) { return entry.signature; } return null; diff --git a/open-sse/translator/request/openai-to-gemini.js b/open-sse/translator/request/openai-to-gemini.js index 24e7f262..305eaca0 100644 --- a/open-sse/translator/request/openai-to-gemini.js +++ b/open-sse/translator/request/openai-to-gemini.js @@ -129,7 +129,7 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG if (tc.type !== OPENAI_BLOCK.FUNCTION) continue; const args = tryParseJSON(tc.function?.arguments || "{}"); - const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null; + const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId, model) : null; // First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined); firstFunctionCallSeen = true; @@ -341,7 +341,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu if (block.type === CLAUDE_BLOCK.TEXT) { parts.push({ text: block.text }); } else if (block.type === CLAUDE_BLOCK.TOOL_USE) { - const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null; + const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId, model) : null; const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined); firstToolUseSeen = true; diff --git a/open-sse/translator/response/gemini-to-openai.js b/open-sse/translator/response/gemini-to-openai.js index 66bf1022..14c47939 100644 --- a/open-sse/translator/response/gemini-to-openai.js +++ b/open-sse/translator/response/gemini-to-openai.js @@ -22,7 +22,7 @@ function emitFunctionCall(functionCall, state, signature = null) { const toolCallIndex = state.functionIndex++; const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`; if (signature) { - storeGeminiThoughtSignature(callId, signature, state.sessionId); + storeGeminiThoughtSignature(callId, signature, state.sessionId, state.model); } const toolCall = { id: callId, @@ -52,7 +52,7 @@ export function geminiToOpenAIResponse(chunk, state) { // Initialize state if (!state.messageId) { state.messageId = response.responseId || `msg_${Date.now()}`; - state.model = response.modelVersion || "gemini"; + state.model = response.modelVersion || state.model || "gemini"; state.functionIndex = 0; state.geminiToolCallCount = 0; results.push(buildChunk(chunkMeta(state), { role: ROLE.ASSISTANT }, null)); diff --git a/tests/unit/antigravity-thought-signature-family.test.js b/tests/unit/antigravity-thought-signature-family.test.js new file mode 100644 index 00000000..7e34a17c --- /dev/null +++ b/tests/unit/antigravity-thought-signature-family.test.js @@ -0,0 +1,104 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +// Keep the signature store in RAM only; the SQLite kv layer is not under test here. +vi.mock("@/lib/db/helpers/kvStore.js", () => ({ + makeKv: () => ({ + get: async () => null, + set: async () => {}, + remove: async () => {}, + getAll: async () => ({}), + }), +})); + +const { + storeGeminiThoughtSignature, + getGeminiThoughtSignatureSync, + signatureFamily, +} = await import("../../open-sse/services/thoughtSignatureStore.js"); +const { openaiToAntigravityRequest } = await import("../../open-sse/translator/request/openai-to-gemini.js"); +const { geminiToOpenAIResponse } = await import("../../open-sse/translator/response/gemini-to-openai.js"); +const { DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } = await import("../../open-sse/config/defaultThinkingSignature.js"); + +let n = 0; +const uid = (p) => `${p}_${Date.now()}_${n++}`; + +function toolHistory(callId) { + return { + messages: [ + { role: "user", content: "list files" }, + { role: "assistant", content: null, tool_calls: [{ id: callId, type: "function", function: { name: "ls", arguments: "{}" } }] }, + { role: "tool", tool_call_id: callId, content: "a.txt" }, + ], + }; +} + +function functionCallSignatures(req) { + return req.request.contents.flatMap((c) => c.parts || []).filter((p) => p.functionCall).map((p) => p.thoughtSignature); +} + +describe("antigravity thought signatures are scoped to the model family", () => { + beforeEach(() => { n++; }); + + it("classifies model families", () => { + expect(signatureFamily("claude-opus-4-6-thinking")).toBe("claude"); + expect(signatureFamily("gemini-3.8-flash-tiered")).toBe("gemini"); + expect(signatureFamily("gpt-oss-120b-medium")).toBe("gpt-oss-120b-medium"); + expect(signatureFamily(null)).toBe(null); + }); + + it("does not return a Claude signature for a Gemini target (and vice versa)", () => { + const claudeCall = uid("toolu"); + const geminiCall = uid("call"); + storeGeminiThoughtSignature(claudeCall, "CLAUDE_SIG", "sess", "claude-opus-4-6-thinking"); + storeGeminiThoughtSignature(geminiCall, "GEMINI_SIG", "sess", "gemini-3.8-flash-tiered"); + + expect(getGeminiThoughtSignatureSync(claudeCall, "sess", "gemini-3.8-flash")).toBe(null); + expect(getGeminiThoughtSignatureSync(claudeCall, "sess", "claude-opus-4-6-thinking")).toBe("CLAUDE_SIG"); + expect(getGeminiThoughtSignatureSync(geminiCall, "sess", "gemini-3.7-flash")).toBe("GEMINI_SIG"); + expect(getGeminiThoughtSignatureSync(geminiCall, null, "claude-sonnet-4-6")).toBe(null); + }); + + it("keeps old behaviour for untagged entries and untargeted lookups", () => { + const call = uid("call"); + storeGeminiThoughtSignature(call, "LEGACY_SIG", "sess"); + expect(getGeminiThoughtSignatureSync(call, "sess", "gemini-3.8-flash")).toBe("LEGACY_SIG"); + + const tagged = uid("toolu"); + storeGeminiThoughtSignature(tagged, "CLAUDE_SIG", "sess", "claude-opus-4-6-thinking"); + expect(getGeminiThoughtSignatureSync(tagged, "sess")).toBe("CLAUDE_SIG"); + }); + + it("records the producing model from the Gemini response stream", () => { + const call = uid("toolu_vrtx"); + const state = { model: "claude-opus-4-6-thinking", sessionId: null, toolNameMap: null }; + geminiToOpenAIResponse({ + response: { + responseId: "r1", + candidates: [{ content: { role: "model", parts: [{ functionCall: { id: call, name: "ls", args: {} }, thoughtSignature: "CLAUDE_SIG" }] } }], + }, + }, state); + expect(getGeminiThoughtSignatureSync(call, null, "gemini-3.8-flash-tiered")).toBe(null); + expect(getGeminiThoughtSignatureSync(call, null, "claude-opus-4-6-thinking")).toBe("CLAUDE_SIG"); + }); + + it("switching Claude -> Gemini mid-conversation sends the default signature, not Claude's", () => { + const call = uid("toolu_vrtx"); + storeGeminiThoughtSignature(call, "CLAUDE_SIG", null, "claude-opus-4-6-thinking"); + const req = openaiToAntigravityRequest("gemini-3.8-flash-tiered", toolHistory(call), true); + expect(functionCallSignatures(req)).toEqual([DEFAULT_THINKING_GEMINI_CLI_SIGNATURE]); + }); + + it("switching Gemini -> Claude mid-conversation does not replay Gemini's signature", () => { + const call = uid("call"); + storeGeminiThoughtSignature(call, "GEMINI_SIG", null, "gemini-3.8-flash-tiered"); + const req = openaiToAntigravityRequest("claude-opus-4-6-thinking", toolHistory(call), true); + expect(functionCallSignatures(req)).not.toContain("GEMINI_SIG"); + }); + + it("same model family still reuses the cached signature", () => { + const call = uid("call"); + storeGeminiThoughtSignature(call, "GEMINI_SIG", null, "gemini-3.8-flash-tiered"); + const req = openaiToAntigravityRequest("gemini-3.8-flash-tiered", toolHistory(call), true); + expect(functionCallSignatures(req)).toEqual(["GEMINI_SIG"]); + }); +}); From d99bc8201daec4ad2a387d30cab79db5cac84a22 Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 17:12:57 +0700 Subject: [PATCH 74/78] fix(sidebar): temporarily hide NEW badge from 9Remote menu Co-Authored-By: Claude Code --- src/shared/components/Sidebar.js | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/shared/components/Sidebar.js b/src/shared/components/Sidebar.js index 01dd1334..9e97f0e6 100644 --- a/src/shared/components/Sidebar.js +++ b/src/shared/components/Sidebar.js @@ -303,9 +303,9 @@ export default function Sidebar({ onClose }) { computer 9Remote - + {/* New - + */} {/* 9English */} From 058ceace48e93379b4f24ac03160d27b894536cf Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 17:32:52 +0700 Subject: [PATCH 75/78] fix(opencode): declare forceStream on transport for free-tier SSE aggregation Declares forceStream: true so chatCore properly converts upstream forced-stream responses to JSON for non-streaming callers. Co-authored-by: anojndr Co-authored-by: TEGAR-SRC Co-authored-by: yxxrn Co-Authored-By: Claude Code --- open-sse/providers/registry/opencode.js | 1 + tests/unit/opencode-session.test.js | 5 +++++ 2 files changed, 6 insertions(+) diff --git a/open-sse/providers/registry/opencode.js b/open-sse/providers/registry/opencode.js index 0914f431..b0705066 100644 --- a/open-sse/providers/registry/opencode.js +++ b/open-sse/providers/registry/opencode.js @@ -17,6 +17,7 @@ export default { headers: { "x-opencode-client": "desktop", }, + forceStream: true, noAuth: true, quirks: { forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"], diff --git a/tests/unit/opencode-session.test.js b/tests/unit/opencode-session.test.js index 89e06a8f..3e50b9fe 100644 --- a/tests/unit/opencode-session.test.js +++ b/tests/unit/opencode-session.test.js @@ -311,4 +311,9 @@ describe("OpenCode Stable Session Reuse (429 follow-up)", () => { expect(chatFull.tools.length).toBe(2); expect(chatFull.tools[0].function.description).toBe("existing"); }); + + it("declares forceStream on the opencode transport so chatCore serves SSE upstream", async () => { + const { PROVIDERS } = await import("../../open-sse/config/providers.js"); + expect(PROVIDERS.opencode?.forceStream).toBe(true); + }); }); From 8e15f0bdd8ab857723d77d7ba566d9f7042b81d7 Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 18:09:32 +0700 Subject: [PATCH 76/78] # v0.5.79 (2026-09-18) ## Features - **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow - **Claude Code**: add 1M-context toggle (`[1m]` marker) and drive `CLAUDE_CODE_AUTO_COMPACT_WINDOW` directly from the dashboard - **Models**: add DeepSeek-V4.1-Flash to DeepSeek provider, CodeBuddy-Intl, and Ollama (`deepseek-v4.1-flash:cloud`); enable `low`..`max` reasoning effort levels and vision capability for DeepSeek-V4.* - **i18n**: integrate Persian (fa) translation ## Fixes - **OpenCode / OpenCode Go**: resolve 403 `FreeTierError` and 429 rate limits with canonical session format, valid User-Agent, and stable upstream session reuse; force stream and declare `forceStream` for free-tier SSE aggregation; cloak decoy tools, normalize Muse Free tool choice, and strip prior reasoning items on Responses models; route Union Alpha via Messages API - **Kiro**: preserve underscores in tool names (`mcp__server__tool`) and restore client tool names in responses; use neutral placeholder for tool-result-only turns; forward tool-result images - **Stream**: report aborts after HTTP 200 in-band (per-format error frames) instead of closing silently - **Command Code**: preserve images and `reasoning_effort` on `/alpha/generate`; retry transient stream errors and avoid fake stop chunks; add Quota Tracker support - **Zed**: harden OAuth lifecycle (preserve `systemId`, renew proxy timeout), support live model resolution, and lower display priority in OAuth list - **Antigravity**: scope cached thought signatures to model family; strip Claude Code billing headers from system prompts; sanitize Hermes system identity - **Codex**: route bare `codex-auto-review` requests to the Codex provider (#4135) - **Auth**: do not cool down an account for request-scoped 4xx errors - **Usage**: improve DeepSeek credit balance display as currency credit instead of 0/total quota bar - **Model Catalog**: scope synced catalog to gateways and declare vision capabilities for DeepSeek V4.1-Flash IDs --- CHANGELOG.md | 20 +++++++++++++ cli/package.json | 2 +- open-sse/providers/pricing.js | 2 ++ open-sse/providers/registry/deepseek.js | 1 + package.json | 2 +- tests/__baseline__/providers-baseline.json | 34 ++++++++++++++++++++-- 6 files changed, 56 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a2b6139d..0585de42 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,23 @@ +# v0.5.79 (2026-09-18) + +## Features +- **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow +- **Claude Code**: add 1M-context toggle (`[1m]` marker) and drive `CLAUDE_CODE_AUTO_COMPACT_WINDOW` directly from the dashboard +- **Models**: add DeepSeek-V4.1-Flash to DeepSeek provider, CodeBuddy-Intl, and Ollama (`deepseek-v4.1-flash:cloud`); enable `low`..`max` reasoning effort levels and vision capability for DeepSeek-V4.* +- **i18n**: integrate Persian (fa) translation + +## Fixes +- **OpenCode / OpenCode Go**: resolve 403 `FreeTierError` and 429 rate limits with canonical session format, valid User-Agent, and stable upstream session reuse; force stream and declare `forceStream` for free-tier SSE aggregation; cloak decoy tools, normalize Muse Free tool choice, and strip prior reasoning items on Responses models; route Union Alpha via Messages API +- **Kiro**: preserve underscores in tool names (`mcp__server__tool`) and restore client tool names in responses; use neutral placeholder for tool-result-only turns; forward tool-result images +- **Stream**: report aborts after HTTP 200 in-band (per-format error frames) instead of closing silently +- **Command Code**: preserve images and `reasoning_effort` on `/alpha/generate`; retry transient stream errors and avoid fake stop chunks; add Quota Tracker support +- **Zed**: harden OAuth lifecycle (preserve `systemId`, renew proxy timeout), support live model resolution, and lower display priority in OAuth list +- **Antigravity**: scope cached thought signatures to model family; strip Claude Code billing headers from system prompts; sanitize Hermes system identity +- **Codex**: route bare `codex-auto-review` requests to the Codex provider (#4135) +- **Auth**: do not cool down an account for request-scoped 4xx errors +- **Usage**: improve DeepSeek credit balance display as currency credit instead of 0/total quota bar +- **Model Catalog**: scope synced catalog to gateways and declare vision capabilities for DeepSeek V4.1-Flash IDs + # v0.5.75 (2026-09-10) ## Features diff --git a/cli/package.json b/cli/package.json index 58b9dfa8..3e658e63 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.75", + "version": "0.5.79", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index 7a7a0019..d09452e2 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -111,6 +111,8 @@ export const MODEL_PRICING = { "deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, "deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, "deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, + "deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, + "deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, "deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 }, // === GLM === diff --git a/open-sse/providers/registry/deepseek.js b/open-sse/providers/registry/deepseek.js index 2c3a3e73..997840e3 100644 --- a/open-sse/providers/registry/deepseek.js +++ b/open-sse/providers/registry/deepseek.js @@ -59,6 +59,7 @@ export default { { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" }, { id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" }, { id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" }, + { id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" }, { id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" }, { id: "deepseek-chat", name: "DeepSeek V3.2 Chat" }, diff --git a/package.json b/package.json index 15b7eca7..a0f7ac88 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "9router-app", - "version": "0.5.75", + "version": "0.5.79", "description": "9Router web dashboard", "private": true, "scripts": { diff --git a/tests/__baseline__/providers-baseline.json b/tests/__baseline__/providers-baseline.json index 2dd09f55..89a3db64 100644 --- a/tests/__baseline__/providers-baseline.json +++ b/tests/__baseline__/providers-baseline.json @@ -44,6 +44,7 @@ }, "usage": { "quotaApiUrl": "https://daily-cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels", + "quotaSummaryApiUrl": "https://daily-cloudcode-pa.googleapis.com/v1internal:retrieveUserQuotaSummary", "loadProjectApiUrl": "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist", "tokenUrl": "https://oauth2.googleapis.com/token" }, @@ -131,6 +132,9 @@ "HTTP-Referer": "https://cline.bot", "X-Title": "Cline" }, + "quirks": { + "clineEnvelope": true + }, "tokenUrl": "https://api.cline.bot/api/v1/auth/token", "refreshUrl": "https://api.cline.bot/api/v1/auth/refresh", "auth": { @@ -149,6 +153,9 @@ "HTTP-Referer": "https://cline.bot", "X-Title": "Cline" }, + "quirks": { + "clineEnvelope": true + }, "auth": { "combined": true, "header": "Authorization", @@ -241,6 +248,12 @@ "reasoningInject": { "scope": "all" }, + "quirks": { + "claudeSupportedToolTypes": [ + "web_search_20250305", + "web_search_20260209" + ] + }, "format": "openai", "transports": [ { @@ -438,6 +451,9 @@ "groq": { "baseUrl": "https://api.groq.com/openai/v1/chat/completions", "validateUrl": "https://api.groq.com/openai/v1/models", + "usage": { + "url": "https://api.groq.com/openai/v1/models" + }, "format": "openai" }, "hyperbolic": { @@ -571,7 +587,8 @@ "Anthropic-Beta": "claude-code-20250219,interleaved-thinking-2025-05-14" }, "quirks": { - "dropOutputConfig": true + "dropOutputConfig": true, + "requireClaudeToolType": true }, "reasoningInject": { "scope": "all" @@ -622,7 +639,8 @@ "Anthropic-Beta": "claude-code-20250219,interleaved-thinking-2025-05-14" }, "quirks": { - "dropOutputConfig": true + "dropOutputConfig": true, + "requireClaudeToolType": true }, "reasoningInject": { "scope": "all" @@ -709,6 +727,9 @@ "opencode-go": { "baseUrl": "https://opencode.ai/zen/go/v1/chat/completions", "headers": {}, + "usage": { + "url": "https://opencode.ai/zen/go/v1/usage" + }, "format": "openai", "transports": [ { @@ -746,7 +767,13 @@ "headers": { "x-opencode-client": "desktop" }, + "forceStream": true, "noAuth": true, + "quirks": { + "forceAutoToolChoiceModels": [ + "muse-spark-1.3-contributor-free" + ] + }, "format": "openai" }, "openrouter": { @@ -949,6 +976,7 @@ "HTTP-Referer": "https://endpoint-proxy.local", "X-Title": "Endpoint Proxy" }, + "forceStream": true, "format": "openai" }, "baidu": { @@ -1010,4 +1038,4 @@ }, "format": "openai" } -} +} \ No newline at end of file From 23ae82d8e3bd2306b85ac798050be5e406763a92 Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 18:31:12 +0700 Subject: [PATCH 77/78] # v0.5.81 (2026-09-18) ## Features - **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow - **Claude Code**: add 1M-context toggle (`[1m]` marker) and drive `CLAUDE_CODE_AUTO_COMPACT_WINDOW` directly from the dashboard - **Models**: add DeepSeek-V4.1-Flash to DeepSeek provider, CodeBuddy-Intl, and Ollama (`deepseek-v4.1-flash:cloud`); enable `low`..`max` reasoning effort levels and vision capability for DeepSeek-V4.* - **i18n**: integrate Persian (fa) translation ## Fixes - **OpenCode / OpenCode Go**: resolve 403 `FreeTierError` and 429 rate limits with canonical session format, valid User-Agent, and stable upstream session reuse; force stream and declare `forceStream` for free-tier SSE aggregation; cloak decoy tools, normalize Muse Free tool choice, and strip prior reasoning items on Responses models; route Union Alpha via Messages API - **Kiro**: preserve underscores in tool names (`mcp__server__tool`) and restore client tool names in responses; use neutral placeholder for tool-result-only turns; forward tool-result images - **Stream**: report aborts after HTTP 200 in-band (per-format error frames) instead of closing silently - **Command Code**: preserve images and `reasoning_effort` on `/alpha/generate`; retry transient stream errors and avoid fake stop chunks; add Quota Tracker support - **Zed**: harden OAuth lifecycle (preserve `systemId`, renew proxy timeout), support live model resolution, and lower display priority in OAuth list - **Antigravity**: scope cached thought signatures to model family; strip Claude Code billing headers from system prompts; sanitize Hermes system identity - **Codex**: route bare `codex-auto-review` requests to the Codex provider (#4135) - **Auth**: do not cool down an account for request-scoped 4xx errors - **Usage**: improve DeepSeek credit balance display as currency credit instead of 0/total quota bar - **Model Catalog**: scope synced catalog to gateways and declare vision capabilities for DeepSeek V4.1-Flash IDs --- cli/package.json | 2 +- package.json | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/cli/package.json b/cli/package.json index 3e658e63..c75e6dc9 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.79", + "version": "0.5.81", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" diff --git a/package.json b/package.json index a0f7ac88..c354a14a 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "9router-app", - "version": "0.5.79", + "version": "0.5.81", "description": "9Router web dashboard", "private": true, "scripts": { From a8c9d3802c5933500fba95416f5bf0c130581396 Mon Sep 17 00:00:00 2001 From: decolua Date: Fri, 18 Sep 2026 18:32:14 +0700 Subject: [PATCH 78/78] docs: update changelog header to v0.5.81 Co-Authored-By: Claude Code --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0585de42..f16c8f53 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,4 +1,4 @@ -# v0.5.79 (2026-09-18) +# v0.5.81 (2026-09-18) ## Features - **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow