Merge remote-tracking branch 'origin/master' into gitea/new_feature

# Conflicts:
#	open-sse/executors/qoder.js
#	open-sse/handlers/chatCore.js
#	open-sse/handlers/chatCore/sseToJsonHandler.js
#	open-sse/providers/registry/commandcode.js
#	src/app/(dashboard)/dashboard/combos/page.js
#	src/app/api/v1/models/route.js
#	src/lib/db/repos/usageRepo.js
#	src/shared/components/UsageStats.js
This commit is contained in:
2026-09-25 10:25:56 +07:00
154 changed files with 10246 additions and 1241 deletions

View File

@@ -4,7 +4,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
## Request lifecycle (chat)
`handlers/chatCore.js` → `services/model.js` `parseModel` (resolve `provider/model`) → **pre-translate hooks** (`rtk/` tool_result compress, `rtk/headroom.js` proxy compress, `rtk/caveman.js` system inject — all fail-open) → `executors/index.js` `getExecutor(provider)` → `translator/index.js` `translateRequest` (client format → provider format) → `executor.execute()` (streams upstream) → `translateResponse` (provider chunks → client format) → SSE out.
`handlers/chatCore.js` → `services/model.js` `parseModel` (resolve `provider/model`) → **RTK for `cursor`** (`rtk/` compresses the source-format `tool_result` / `role:tool` in-place — its translator rewrites those shapes, so this one provider must run **before** translate) → `translator/index.js` `translateRequest` (client format → provider format) → **post-translate savers** (`rtk/` compress for every other provider, `rtk/headroom.js` proxy compress, `rtk/caveman.js` / `rtk/ponytail.js` system inject — all fail-open) → `executors/index.js` `getExecutor(provider)` → `executor.execute()` (streams upstream) → `translateResponse` (provider chunks → client format) → SSE out.
## Directory map

View File

@@ -53,7 +53,7 @@ export function findModelName(aliasOrId, modelId) {
}
export function getModelTargetFormat(aliasOrId, modelId) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go" || aliasOrId === "ocz" || aliasOrId === "opencode-zen") && isMuseSparkModel(modelId)) {
return FORMATS.OPENAI_RESPONSES;
}
const models = PROVIDER_MODELS[aliasOrId];

View File

@@ -293,12 +293,19 @@ export class AntigravityExecutor extends BaseExecutor {
this._lastSessionId = transformedRequest.sessionId; // cached for buildHeaders (base.execute order)
// Official Antigravity client omits `requestType` entirely on the agent
// (chat) path. Sending `requestType: "agent"` here (or leaking it through
// from an upstream envelope via the ...body spread below) makes Google
// bucket the request and return a detail-free 429 RESOURCE_EXHAUSTED even
// with quota available. `image_gen` and
// `search` buckets are unaffected and keep their own requestType.
delete body.requestType;
return {
...body,
project: projectId,
model: body.model || model,
userAgent: "antigravity",
requestType: "agent",
requestId: buildIdeRequestId({ body, request: transformedRequest, credentials, model, requestType: "agent" }),
request: transformedRequest
};

View File

@@ -7,13 +7,16 @@ import {
wrapConnectRPCFrame,
decodeMessage,
parseConnectRPCFrame,
extractTextFromResponse
extractTextFromResponse,
encodeMcpTools,
decodeMcpArgs,
} from "../utils/cursorProtobuf.js";
import { buildCursorHeaders } from "../utils/cursorChecksum.js";
import { estimateUsage } from "../utils/usageTracking.js";
import { SSE_DONE, SSE_HEADERS } from "../utils/sseConstants.js";
import { chatChunkSse, sseChunk } from "../utils/sse.js";
import { FORMATS } from "../translator/formats.js";
import { ROLE, OPENAI_BLOCK } from "../translator/schema/index.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import zlib from "zlib";
import crypto from "crypto";
@@ -65,55 +68,74 @@ function textFromContent(content) {
if (typeof content === "string") return content;
if (!Array.isArray(content)) return "";
return content
.filter((part) => part?.type === "text" && typeof part.text === "string")
.filter((part) => part?.type === OPENAI_BLOCK.TEXT && typeof part.text === "string")
.map((part) => part.text)
.join("\n");
}
function isAgentTextRequest(body) {
// Many compatible clients always attach their built-in tool schemas, even
// for a normal text turn. Cursor's retired ChatService rejects those
// requests; AgentService can still answer the text turn, so ignore schemas
// here. A real tool-call/result conversation is kept on the legacy path
// until its AgentService tool protocol is implemented.
return Array.isArray(body?.messages) && body.messages.every((message) => {
if (message?.tool_calls?.length || message?.role === "tool") return false;
return typeof message?.content === "string"
|| Array.isArray(message?.content) && message.content.every((part) => part?.type === "text");
function isTextPart(part) {
return !part || part.type === OPENAI_BLOCK.TEXT || typeof part === "string";
}
export function isAgentCapableRequest(body) {
// ChatService rejects auto/composer and most thinking variants. AgentService
// can answer text turns (including declared tool schemas) and tool-call
// history. Image parts still need the legacy protobuf path.
if (!Array.isArray(body?.messages) || body.messages.length === 0) return false;
return body.messages.every((message) => {
if (Array.isArray(message?.content)) return message.content.every(isTextPart);
return message?.content == null || typeof message.content === "string";
});
}
function encodeHistoryMessage(message) {
const content = textFromContent(message?.content);
if (!content) return null;
const extras = [];
if (message?.role === ROLE.ASSISTANT && message.tool_calls?.length) {
for (const tc of message.tool_calls) {
extras.push(`[tool_call id=${tc.id || ""} name=${tc.function?.name || "tool"} args=${tc.function?.arguments || "{}"}]`);
}
}
if (message?.role === ROLE.TOOL) {
extras.push(`[tool_result id=${message.tool_call_id || ""}]`);
}
const textBody = [content, ...extras].filter(Boolean).join("\n");
if (!textBody) return null;
// ConversationHistoryMessage.user / .assistant -> repeated content -> text.
const text = agentString(1, content);
if (message.role === "assistant") {
const text = agentString(1, textBody);
if (message.role === ROLE.ASSISTANT) {
return agentMessage(2, agentMessage(1, agentMessage(1, text)));
}
return agentMessage(1, agentMessage(1, agentMessage(1, text)));
}
function buildAgentRunFrame(messages, model) {
export function buildAgentRunFrame(messages, model, tools = []) {
// custom_system_prompt (RunRequest field 8) makes AgentService return an
// empty turn. Fold system text into the current user message instead.
const system = messages
.filter((message) => message?.role === "system")
.filter((message) => message?.role === ROLE.SYSTEM)
.map((message) => textFromContent(message.content))
.filter(Boolean)
.join("\n\n");
const chatMessages = messages.filter((message) => message?.role !== "system");
const currentIndex = [...chatMessages].map((message) => message?.role).lastIndexOf("user");
const chatMessages = messages.filter((message) => message?.role !== ROLE.SYSTEM);
const currentIndex = [...chatMessages].map((message) => message?.role).lastIndexOf(ROLE.USER);
const current = currentIndex >= 0 ? chatMessages[currentIndex] : chatMessages.at(-1);
const history = chatMessages
.slice(0, currentIndex >= 0 ? currentIndex : -1)
.map(encodeHistoryMessage)
.filter(Boolean);
const userText = textFromContent(current?.content) || "Continue.";
const rawUser = textFromContent(current?.content) || "Continue.";
const userText = system ? `${system}\n\n${rawUser}` : rawUser;
// agent.v1.UserMessageAction.user_message and its optional history.
// selected_context (3) + mode=1 (4) match cursor-agent's wire format; without
// them the server may accept the RPC and stream an empty turn.
const userMessage = concatBuffers(
agentString(1, userText),
agentString(2, crypto.randomUUID()),
agentMessage(3, new Uint8Array()),
encodeField(4, PROTOBUF_VARINT, 1),
);
const conversationHistory = history.length
? concatBuffers(...history.map((entry) => agentMessage(1, entry)))
@@ -124,11 +146,20 @@ function buildAgentRunFrame(messages, model) {
);
const conversationAction = agentMessage(1, userAction);
const requestedModel = concatBuffers(agentString(1, model), agentBool(7, true));
// ModelDetails (field 3): thinking variants (Composer, Grok, *-thinking)
// return an empty turn when only RequestedModel (field 9) is set.
const modelDetails = concatBuffers(
agentString(1, model),
agentString(3, model),
agentString(4, model),
);
const mcpTools = encodeMcpTools(tools);
const runRequest = concatBuffers(
// An empty ConversationStateStructure starts a fresh local agent session.
agentMessage(1, new Uint8Array()),
agentMessage(2, conversationAction),
...(system ? [agentString(8, system)] : []),
agentMessage(3, modelDetails),
...(mcpTools.length ? [agentMessage(4, mcpTools)] : []),
agentMessage(9, requestedModel),
);
@@ -157,13 +188,51 @@ function decodeAgentFrames(buffer, onFrame) {
return pending;
}
function createRequestContextResponse() {
// AgentService asks every run for client context. 9router has no IDE file
// context, so acknowledge with an empty RequestContext.
function execIds(execRequest) {
const id = Number(execRequest?.get(1)?.[0]?.value || 0);
const execId = extractAgentString(execRequest, 15);
return { id, execId };
}
function wrapExecClientMessage(execMsgId, execId, resultField, resultPayload) {
const parts = [];
if (execMsgId) parts.push(encodeField(1, PROTOBUF_VARINT, execMsgId));
parts.push(agentString(15, execId || ""));
parts.push(encodeField(resultField, PROTOBUF_LEN, resultPayload || new Uint8Array()));
return wrapConnectRPCFrame(agentMessage(2, concatBuffers(...parts)));
}
function createRequestContextResponse(execRequest) {
// Tools already go out on AgentRunRequest.mcp_tools. Echoing them again on
// this ack makes AgentService stall silently (0 SSE bytes until abort).
const { id, execId } = execIds(execRequest);
const requestContextSuccess = agentMessage(1, new Uint8Array());
const requestContextResult = agentMessage(1, requestContextSuccess);
const execClientMessage = agentMessage(10, requestContextResult);
return wrapConnectRPCFrame(agentMessage(2, execClientMessage));
return wrapExecClientMessage(id, execId, 10, requestContextResult);
}
// ExecServerMessage variant → ExecClientMessage result field (same numbers).
const EXEC_RESULT_FIELD = {
2: 2, 3: 3, 4: 4, 5: 5, 7: 7, 8: 8, 9: 9, 16: 16, 20: 20, 23: 23,
};
function rejectExecRequest(execRequest) {
const { id, execId } = execIds(execRequest);
const variant = [...(execRequest?.keys?.() || [])].find((field) => field !== 1 && field !== 15);
const resultField = EXEC_RESULT_FIELD[variant];
if (!resultField) return null;
// Diagnostics has no rejected variant — empty success unblocks the stream.
if (variant === 9) return wrapExecClientMessage(id, execId, 9, new Uint8Array());
const rejected = agentMessage(2, agentString(2, "Tool not available in this environment. Use the MCP tools provided instead."));
return wrapExecClientMessage(id, execId, resultField, rejected);
}
function encodeKvClientMessage(kvId, resultField, resultPayload, metadata) {
const parts = [];
if (kvId) parts.push(encodeField(1, PROTOBUF_VARINT, kvId));
parts.push(encodeField(resultField, PROTOBUF_LEN, resultPayload || new Uint8Array()));
if (metadata && metadata.length) parts.push(encodeField(4, PROTOBUF_LEN, metadata));
return wrapConnectRPCFrame(agentMessage(3, concatBuffers(...parts)));
}
const CURSOR_STREAM_DEBUG = process.env.CURSOR_STREAM_DEBUG === "1";
@@ -479,7 +548,7 @@ export class CursorExecutor extends BaseExecutor {
};
}
async executeAgent({ model, body, stream, credentials, signal }) {
async executeAgent({ model, body, stream, credentials, signal, log }) {
const agentEndpoint = PROVIDER_OAUTH.cursor?.agentEndpoint;
if (!agentEndpoint) throw new Error("Cursor AgentService endpoint is not configured");
@@ -491,9 +560,10 @@ export class CursorExecutor extends BaseExecutor {
}
let session;
const tools = body.tools || [];
try {
session = this.openAgentHttp2Stream(url, headers, requestController.signal);
session.write(buildAgentRunFrame(body.messages || [], model));
session.write(buildAgentRunFrame(body.messages || [], model, tools));
} catch (error) {
throw new Error(`Cursor AgentService request failed: ${error.message}`);
}
@@ -533,8 +603,23 @@ export class CursorExecutor extends BaseExecutor {
// so strict clients such as Claude Code accept the completed stream.
const responseId = `chatcmpl-msg_${Date.now()}`;
const created = Math.floor(Date.now() / 1000);
const composerModel = isComposerModel(model);
let pending = Buffer.alloc(0);
let finished = false;
let thinkingAcc = "";
let emittedVisible = 0;
let emittedText = false;
const flushThinkingFallback = (onEvent) => {
if (emittedText || !thinkingAcc) return;
const fallback = composerModel
? visibleComposerContentFromThinking(thinkingAcc)
: thinkingAcc.trim();
if (fallback) {
emittedText = true;
onEvent({ type: "text", value: fallback });
}
};
const consume = async (onEvent) => {
try {
@@ -553,32 +638,87 @@ export class CursorExecutor extends BaseExecutor {
const update = decodeMessage(serverMessage.get(1)[0].value);
if (update.has(1)) {
const textDelta = extractAgentString(decodeMessage(update.get(1)[0].value), 1);
if (textDelta) onEvent({ type: "text", value: textDelta });
if (textDelta) {
emittedText = true;
onEvent({ type: "text", value: textDelta });
}
}
// Cursor's AgentService emits internal reasoning without the
// cryptographic signature required by Anthropic thinking blocks.
// Forwarding it makes strict Anthropic clients (Claude Code)
// discard or wait on an otherwise complete response. Keep the
// reasoning upstream-only and emit the normal answer text.
// thinking_delta (field 4). Composer (and some Grok variants) put
// the visible answer after </think> here and never send text_delta.
if (update.has(4)) {
const thinkingDelta = extractAgentString(decodeMessage(update.get(4)[0].value), 1);
if (thinkingDelta) {
thinkingAcc += thinkingDelta;
if (composerModel) {
const visible = visibleComposerContentFromThinking(thinkingAcc);
if (visible.length > emittedVisible) {
const deltaContent = visible.slice(emittedVisible);
emittedVisible = visible.length;
emittedText = true;
onEvent({ type: "text", value: deltaContent });
}
}
}
}
// Keep unsigned reasoning upstream-only for Anthropic clients.
if (update.has(14)) {
flushThinkingFallback(onEvent);
finished = true;
onEvent({ type: "done" });
}
}
// KvServerMessage (field 4): get/set blob. Ack so the stream proceeds.
if (serverMessage.has(4)) {
const kv = decodeMessage(serverMessage.get(4)[0].value);
const kvId = kv.get(1)?.[0]?.value || 0;
const metadata = kv.get(4)?.[0]?.value || null;
if (kv.has(2)) {
session.write(encodeKvClientMessage(kvId, 2, agentMessage(1, new Uint8Array()), metadata));
} else if (kv.has(3)) {
session.write(encodeKvClientMessage(kvId, 3, new Uint8Array(), metadata));
}
}
// AgentService requests IDE context before producing a response.
// Return an empty context; 9router is not coupled to an editor.
if (serverMessage.has(2)) {
const execRequest = decodeMessage(serverMessage.get(2)[0].value);
if (execRequest.has(10)) {
session.write(createRequestContextResponse());
log?.info?.("CURSOR", "AgentService request_context ack");
session.write(createRequestContextResponse(execRequest));
} else if (execRequest.has(11)) {
const mcp = decodeMcpArgs(execRequest.get(11)[0].value);
const name = mcp.toolName || mcp.name;
if (name) {
log?.info?.("CURSOR", `AgentService MCP tool_call ${name}`);
finished = true;
onEvent({
type: "tool_call",
value: {
id: mcp.toolCallId || `call_${crypto.randomUUID()}`,
name,
arguments: JSON.stringify(mcp.args || {}),
},
});
onEvent({ type: "done", finishReason: "tool_calls" });
} else {
debugLog(`[CURSOR AGENT] Unsupported exec request fields: ${[...execRequest.keys()].join(",")}`);
finished = true;
onEvent({ type: "error", value: "Cursor AgentService requested an unsupported IDE tool" });
}
} else {
// Every other ExecServerMessage variant is an editor-backed tool
// (shell, read, write, …) that 9router cannot service. Fail the
// turn rather than narrating protocol state as assistant text.
debugLog(`[CURSOR AGENT] Unsupported exec request fields: ${[...execRequest.keys()].join(",")}`);
finished = true;
onEvent({ type: "error", value: "Cursor AgentService requested an unsupported IDE tool" });
// Auto/Composer often probe IDE builtins (shell/read/…). Reject
// them so the model can continue with MCP tools or a text answer
// instead of stalling the h2 stream.
const rejection = rejectExecRequest(execRequest);
if (rejection) {
log?.info?.("CURSOR", `AgentService rejected IDE exec fields=${[...execRequest.keys()].join(",")}`);
session.write(rejection);
} else {
debugLog(`[CURSOR AGENT] Unsupported exec request fields: ${[...execRequest.keys()].join(",")}`);
finished = true;
onEvent({ type: "error", value: "Cursor AgentService requested an unsupported IDE tool" });
}
}
}
});
@@ -586,7 +726,10 @@ export class CursorExecutor extends BaseExecutor {
} finally {
try { session.end(); } catch {}
try { session.close(); } catch {}
if (!finished) onEvent({ type: "done" });
if (!finished) {
flushThinkingFallback(onEvent);
onEvent({ type: "done" });
}
}
};
@@ -594,10 +737,21 @@ export class CursorExecutor extends BaseExecutor {
let content = "";
let reasoning = "";
let agentError = null;
const toolCalls = [];
let finishReason = "stop";
await consume((event) => {
if (event.type === "text") content += event.value;
else if (event.type === "thinking") reasoning += event.value;
else if (event.type === "tool_call") {
toolCalls.push({
id: event.value.id,
type: "function",
function: { name: event.value.name, arguments: event.value.arguments },
});
finishReason = "tool_calls";
}
else if (event.type === "error") agentError = event.value;
else if (event.type === "done" && event.finishReason) finishReason = event.finishReason;
});
if (agentError) {
return {
@@ -611,13 +765,19 @@ export class CursorExecutor extends BaseExecutor {
responseFormat: FORMATS.OPENAI,
};
}
const message = {
role: "assistant",
content: content || null,
...(reasoning ? { reasoning_content: reasoning } : {}),
...(toolCalls.length ? { tool_calls: toolCalls } : {}),
};
return {
response: new Response(JSON.stringify({
id: responseId,
object: "chat.completion",
created,
model,
choices: [{ index: 0, message: { role: "assistant", content: content || null, ...(reasoning ? { reasoning_content: reasoning } : {}) }, finish_reason: "stop" }],
choices: [{ index: 0, message, finish_reason: finishReason }],
usage: estimateUsage(body, content.length, FORMATS.OPENAI),
}), { headers: { "Content-Type": "application/json" } }),
url,
@@ -635,6 +795,18 @@ export class CursorExecutor extends BaseExecutor {
controller.enqueue(encoder.encode(chatChunkSse({ id: responseId, created, model, delta: { content: event.value } })));
} else if (event.type === "thinking") {
controller.enqueue(encoder.encode(chatChunkSse({ id: responseId, created, model, delta: { reasoning_content: event.value } })));
} else if (event.type === "tool_call") {
controller.enqueue(encoder.encode(chatChunkSse({
id: responseId, created, model,
delta: {
tool_calls: [{
index: 0,
id: event.value.id,
type: "function",
function: { name: event.value.name, arguments: event.value.arguments },
}],
},
})));
} else if (event.type === "error") {
// An SSE error frame, not a content delta: a protocol failure must not
// be rendered to the user as the assistant's reply, and downstream
@@ -643,7 +815,10 @@ export class CursorExecutor extends BaseExecutor {
controller.enqueue(encoder.encode(SSE_DONE));
controller.close();
} else if (event.type === "done") {
controller.enqueue(encoder.encode(chatChunkSse({ id: responseId, created, model, delta: {}, finishReason: "stop" })));
controller.enqueue(encoder.encode(chatChunkSse({
id: responseId, created, model, delta: {},
finishReason: event.finishReason || "stop",
})));
controller.enqueue(encoder.encode(SSE_DONE));
controller.close();
}
@@ -664,9 +839,9 @@ export class CursorExecutor extends BaseExecutor {
}
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
if (isAgentTextRequest(body)) {
if (isAgentCapableRequest(body)) {
try {
return await this.executeAgent({ model, body, stream, credentials, signal });
return await this.executeAgent({ model, body, stream, credentials, signal, log });
} catch (error) {
return {
response: new Response(JSON.stringify({

View File

@@ -11,6 +11,7 @@ import { CursorExecutor } from "./cursor.js";
import { VertexExecutor } from "./vertex.js";
import { OpenCodeExecutor } from "./opencode.js";
import { OpenCodeGoExecutor } from "./opencode-go.js";
import { OpenCodeZenExecutor } from "./opencode-zen.js";
import { GrokWebExecutor } from "./grok-web.js";
import { GrokCliExecutor } from "./grok-cli.js";
import { PerplexityWebExecutor } from "./perplexity-web.js";
@@ -34,6 +35,7 @@ const executors = {
github: new GithubExecutor(),
iflow: new IFlowExecutor(),
qoder: new QoderExecutor(),
"qoder-cn": new QoderExecutor("qoder-cn"),
kiro: new KiroExecutor(),
kimchi: new KimchiExecutor(),
codex: new CodexExecutor(),
@@ -43,6 +45,7 @@ const executors = {
"vertex-partner": new VertexExecutor("vertex-partner"),
opencode: new OpenCodeExecutor(),
"opencode-go": new OpenCodeGoExecutor(),
"opencode-zen": new OpenCodeZenExecutor(),
"grok-web": new GrokWebExecutor(),
"grok-cli": new GrokCliExecutor(),
gcli: new GrokCliExecutor(), // Alias
@@ -89,6 +92,7 @@ export { VertexExecutor } from "./vertex.js";
export { DefaultExecutor } from "./default.js";
export { OpenCodeExecutor } from "./opencode.js";
export { OpenCodeGoExecutor } from "./opencode-go.js";
export { OpenCodeZenExecutor } from "./opencode-zen.js";
export { GrokWebExecutor } from "./grok-web.js";
export { GrokCliExecutor } from "./grok-cli.js";
export { PerplexityWebExecutor } from "./perplexity-web.js";

View File

@@ -0,0 +1,315 @@
import crypto from "node:crypto";
import { DefaultExecutor } from "./default.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../translator/formats/responsesApi.js";
const SESSION_HEADER = "x-opencode-session";
const SESSION_FIELD = "_opencodeZenSession";
const MAX_SESSION_LENGTH = 256;
const RESPONSES_BASE_URL = "https://opencode.ai/zen/v1/responses";
const MAX_TOOL_NAME_LEN = 128;
const OPENCODE_UA = "opencode/1.18.31";
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
// Free-tier fingerprint (mirrors opencode executor, PR #4132): upstream 403s
// requests without the file-search quartet and without stream:true.
const OPENCODE_FINGERPRINT_TOOLS = ["bash", "glob", "grep", "read"];
function hasValidOpencodeVersion(ua) {
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
if (!m) return false;
const major = parseInt(m[1], 10);
const minor = parseInt(m[2], 10);
return major > 1 || (major === 1 && minor >= 17);
}
function unstableRandom() {
const bytes = crypto.randomBytes(14);
let randomPart = "";
for (let i = 0; i < 14; i++) {
randomPart += BASE62_CHARS[bytes[i] % 62];
}
return randomPart;
}
export function generateSessionId(timestamp = Date.now()) {
const current = BigInt(timestamp) * 0x1000n + 1n;
const value = ~current;
const time = Array.from({ length: 6 }, (_, index) =>
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
.toString(16)
.padStart(2, "0")
).join("");
return `ses_${time}${unstableRandom()}`;
}
export function generateRequestId(timestamp = Date.now()) {
const current = BigInt(timestamp) * 0x1000n + 1n;
const value = current;
const time = Array.from({ length: 6 }, (_, index) =>
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
.toString(16)
.padStart(2, "0")
).join("");
return `msg_${time}${unstableRandom()}`;
}
export function translateSessionId(sessionId, clientTool = "") {
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
return sessionId.trim();
}
const digest = crypto
.createHash("sha256")
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
.digest();
const timeHex = digest.subarray(0, 6).toString("hex");
let randomPart = "";
for (let i = 6; i < 20; i++) {
randomPart += BASE62_CHARS[digest[i] % 62];
}
return `ses_${timeHex}${randomPart}`;
}
function toolNameOf(tool) {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return "";
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const raw = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
return raw.trim();
}
function ensureChatFingerprintTools(body) {
if (!body || typeof body !== "object") return;
const present = new Set();
if (Array.isArray(body.tools)) {
for (const tool of body.tools) {
const name = toolNameOf(tool);
if (name) present.add(name);
}
} else {
body.tools = [];
}
for (const name of OPENCODE_FINGERPRINT_TOOLS) {
if (present.has(name)) continue;
body.tools.push({
type: "function",
function: {
name,
description: `OpenCode built-in ${name} tool`,
parameters: { type: "object", properties: {} },
},
});
present.add(name);
}
}
function ensureResponsesFingerprintTools(body) {
if (!body || typeof body !== "object") return;
const present = new Set();
if (Array.isArray(body.tools)) {
for (const tool of body.tools) {
const name = toolNameOf(tool);
if (name) present.add(name);
}
} else {
body.tools = [];
}
for (const name of OPENCODE_FINGERPRINT_TOOLS) {
if (present.has(name)) continue;
body.tools.push({
type: "function",
name,
description: `OpenCode built-in ${name} tool`,
parameters: { type: "object", properties: {} },
});
present.add(name);
}
}
function normalizeSession(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return normalized;
}
function nativeSession(headers) {
if (!headers || typeof headers !== "object") return null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) {
const normalized = normalizeSession(value);
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
}
}
return null;
}
function translatedSession(sessionId, clientTool) {
return translateSessionId(sessionId, clientTool);
}
// Strip the thinking suffix "model(level)" so checks hit the base id.
function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
function isResponsesModel(model) {
return isMuseSparkModel(baseModelId(model));
}
// Flatten Chat Completions tool declarations into the Responses flat shape and
// drop hosted/nameless tools the /responses endpoint rejects.
function normalizeResponsesTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
const name = rawName.trim();
if (!name) return false;
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
? tool.parameters
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
// Mirror the request translator: {type:"object"} without properties is rejected
// by strict Responses backends, so fill in the empty properties map.
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
for (const k of Object.keys(tool)) delete tool[k];
tool.type = "function";
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
if (description) tool.description = description;
tool.parameters = parameters;
validNames.add(tool.name);
return true;
});
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
if (!n || !validNames.has(n)) delete body.tool_choice;
}
}
}
// Last line of defense for native Responses clients (sourceFormat === targetFormat
// skips translation): coerce items in place so malformed tool payloads 400 here
// with a clear shape instead of upstream as InputValidationError.
function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
// Strip prior-turn reasoning items: Muse Spark contributor models route to
// an upstream Console backend where encrypted_content cannot be validated across
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
if (item.type === "reasoning") return false;
delete item.encrypted_content;
delete item.reasoning_encrypted_content;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
item.call_id = clampResponsesCallId(item.call_id);
item.arguments = coerceResponsesArguments(item.arguments);
return true;
}
if (item.type === "function_call_output") {
item.call_id = clampResponsesCallId(item.call_id);
item.output = coerceResponsesOutput(item.output);
return true;
}
return true;
});
}
export class OpenCodeZenExecutor extends DefaultExecutor {
constructor() {
super("opencode-zen");
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
return super.buildUrl(model, stream, urlIndex, credentials);
}
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
const sourceCredentials = credentials || {};
const native = nativeSession(sourceCredentials.rawHeaders);
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
headers: sourceCredentials.rawHeaders,
body,
connectionId: sourceCredentials.connectionId,
scope: "opencode-zen",
});
return {
...sourceCredentials,
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
};
}
async execute(args) {
const credentials = this.prepareRequestCredentials(args);
return super.execute({ ...args, credentials });
}
buildHeaders(credentials, stream = true, url, model) {
const headers = super.buildHeaders(credentials || {}, stream, url, model);
const raw = credentials?.rawHeaders || {};
const lower = {};
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
const downstreamUa = lower["user-agent"] || "";
// Free-tier gate: spoof the official client UA.
headers["User-Agent"] = hasValidOpencodeVersion(downstreamUa) ? downstreamUa : OPENCODE_UA;
headers["x-opencode-client"] = lower["x-opencode-client"] || "desktop";
const prepared = credentials?.[SESSION_FIELD];
if (prepared) {
headers[SESSION_HEADER] = prepared;
return headers;
}
const fallback = this.prepareRequestCredentials({ credentials });
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
return headers;
}
transformRequest(model, body, stream, credentials) {
const out = super.transformRequest(model, body);
// Free-tier gate: upstream 403s stream:false even when everything else is valid.
if (out && typeof out === "object") out.stream = true;
if (!isResponsesModel(model || body?.model)) {
ensureChatFingerprintTools(out);
return out;
}
const normalized = normalizeResponsesInput(out.input);
if (normalized) out.input = normalized;
if (!Array.isArray(out.input) || out.input.length === 0) {
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
}
// Responses names the output cap max_output_tokens, not max_tokens.
if (out.max_output_tokens === undefined) {
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
}
delete out.max_tokens;
delete out.max_completion_tokens;
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
}
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
if (!out.reasoning.summary) out.reasoning.summary = "auto";
}
delete out.reasoning_effort;
out.stream = true;
out.store = false;
ensureResponsesFingerprintTools(out);
normalizeResponsesTools(out);
sanitizeResponsesItems(out);
return out;
}
}

View File

@@ -6,6 +6,7 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import { applyFingerprintTools } from "../utils/opencodeFingerprint.js";
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
import {
normalizeResponsesInput,
@@ -24,68 +25,6 @@ export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
// OpenCode free tier requires both 'bash' and 'read' in tools payload.
// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read)
// take precedence while satisfying upstream verification.
const OPENCODE_DECOY_CHAT_TOOLS = [
{
type: "function",
function: {
name: "bash",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
},
{
type: "function",
function: {
name: "read",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
},
];
const OPENCODE_DECOY_RESPONSES_TOOLS = [
{
type: "function",
name: "bash",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
{
type: "function",
name: "read",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
];
function cloakOpencodeTools(body, isResponses) {
if (!body || typeof body !== "object") return;
if (isResponses) {
if (!Array.isArray(body.tools)) body.tools = [];
const names = new Set(body.tools.map((t) => t.name || t.function?.name));
for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) {
if (!names.has(tool.name)) body.tools.push({ ...tool });
}
if (!body.tool_choice) body.tool_choice = "auto";
} else {
const hasTools = Array.isArray(body.tools) && body.tools.length > 0;
if (!hasTools) {
body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } }));
if (!body.tool_choice) body.tool_choice = "none";
} else {
const names = new Set(body.tools.map((t) => t.function?.name || t.name));
for (const tool of OPENCODE_DECOY_CHAT_TOOLS) {
if (!names.has(tool.function.name)) {
body.tools.push({ ...tool, function: { ...tool.function } });
}
}
}
}
}
function hasValidOpencodeVersion(ua) {
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
if (!m) return false;
@@ -499,11 +438,12 @@ export class OpenCodeExecutor extends BaseExecutor {
body.store = false;
normalizeResponsesTools(body);
sanitizeResponsesItems(body);
if (!Array.isArray(body.tools) || body.tools.length === 0) {
cloakOpencodeTools(body, true);
}
// Free-tier fingerprint tools are required even when an agent client
// already supplied tools. ZCode/Claude Code requests normally have
// non-empty tool arrays; skipping cloaking here triggers 403 FreeTierError.
applyFingerprintTools(body, true);
} else if (body && typeof body === "object") {
cloakOpencodeTools(body, false);
applyFingerprintTools(body, false);
}
return injectReasoningContent({ provider: this.provider, model, body });
}

View File

@@ -29,7 +29,7 @@ import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { SSE_DONE } from "../utils/sseConstants.js";
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
import { FETCH_CONNECT_TIMEOUT_MS, HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
import {
QODER_CHAT_SIG_PATH,
@@ -208,16 +208,16 @@ function truncate(s, n) {
/**
* Map the OpenAI-style request body into the exact shape Qoder expects.
*/
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null }) {
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null, region = "intl" }) {
const qoderKey = String(model || "").replace(/^qoder\//, "");
// Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP.
// This allows support for new Qoder models (e.g., qmodel_latest) without code changes.
let modelConfig = await getQoderModelConfig(credentials, qoderKey, { log, proxyOptions, signal });
let modelConfig = await getQoderModelConfig(credentials, qoderKey, { log, proxyOptions, signal, region });
if (!modelConfig) {
// Try a forced refresh once before giving up — the cache may simply
// not be populated yet on first ever call for this credential.
const refreshed = await resolveQoderModels(credentials, { forceRefresh: true, log, proxyOptions, signal });
const refreshed = await resolveQoderModels(credentials, { forceRefresh: true, log, proxyOptions, signal, region });
const retried = refreshed?.rawConfigs.get(qoderKey);
if (!retried) {
throw new Error(
@@ -337,47 +337,65 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
/**
* Check if a qoder error message indicates a billing/quota block.
* Signatures: code 112 (quota exhausted), code 10605 (queue throttle), pricingUrl field.
* Signatures: code 110 (billing daily count exceeded), code 112 (quota
* exhausted), code 10605 (queue throttle), pricingUrl field.
*/
function isBillingBlock(inner) {
if (!inner || typeof inner !== "string") return false;
const lowerMsg = inner.toLowerCase();
// Match: {"code":"112",...}, {"code":"10605",...}, or pricingUrl field
return /\"code\"\s*:\s*\"(112|10605)\"/.test(inner) || lowerMsg.includes("pricingurl");
if (lowerMsg.includes("pricingurl")) return true;
// Parsed code preferred over regex: matches numeric or string "110"/"112"/"10605".
try {
const parsed = JSON.parse(inner);
const code = String(parsed?.code ?? "");
if (code === "110" || code === "112" || code === "10605") return true;
} catch { /* not JSON — fall through to legacy shape match */ }
// Match legacy exact shapes: {"code":"112",...}, {"code":"10605",...}.
return /"code"\s*:\s*"(112|10605)"/.test(inner);
}
/**
* Peek the first SSE frame to detect billing errors before piping.
* Returns { isBilling, statusVal, message, consumed } — `consumed` is every
* Peek the first SSE data line to detect upstream errors before piping.
* Returns { isError, isBilling, statusVal, message, consumed } — `consumed` is every
* byte read so far (including the peeked line) so the caller can re-process
* it and nothing is dropped from the stream.
*/
async function peekFirstQoderFrame(reader, decoder) {
let consumed = "";
let offset = 0;
let upstreamDone = false;
while (true) {
const { done, value } = await reader.read();
if (done) return { isBilling: false, consumed, upstreamDone: true };
let nl = consumed.indexOf("\n", offset);
if (nl === -1 && !upstreamDone) {
const { done, value } = await reader.read();
upstreamDone = done;
consumed += done ? decoder.decode() : decoder.decode(value, { stream: true });
continue;
}
if (offset >= consumed.length) return { isError: false, consumed, upstreamDone };
if (nl === -1) nl = consumed.length;
consumed += decoder.decode(value, { stream: true });
const nl = consumed.indexOf("\n");
if (nl === -1) continue; // need a full line first
const line = consumed.slice(0, nl).replace(/\r$/, "").trim();
const line = consumed.slice(offset, nl).replace(/\r$/, "").trim();
offset = nl + 1;
if (!line.startsWith("data:")) continue;
const data = line.slice(5).trimStart();
if (data === "[DONE]") return { isBilling: false, consumed };
if (data === "[DONE]") return { isError: false, consumed, upstreamDone };
let envelope;
try { envelope = JSON.parse(data); } catch { return { isBilling: false, consumed }; }
try { envelope = JSON.parse(data); } catch { return { isError: false, consumed, upstreamDone }; }
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
const inner = typeof envelope.body === "string" ? envelope.body : "";
// statusCodeValue is documented numeric, but accept numeric strings defensively.
const raw = Number(envelope?.statusCodeValue);
const statusVal = Number.isNaN(raw) ? 200 : raw;
const inner = typeof envelope?.body === "string"
? envelope.body
: envelope?.body != null ? JSON.stringify(envelope.body) : "";
if (statusVal !== 200 && isBillingBlock(inner)) {
return { isBilling: true, statusVal, message: inner || `qoder billing block (${statusVal})` };
if (statusVal !== 200) {
return { isError: true, isBilling: isBillingBlock(inner), statusVal, message: inner || `upstream status ${statusVal}` };
}
return { isBilling: false, consumed };
return { isError: false, consumed, upstreamDone };
}
}
@@ -388,8 +406,8 @@ async function peekFirstQoderFrame(reader, decoder) {
* Each upstream line looks like:
* data: {"statusCodeValue":200,"body":"{\"choices\":[{\"delta\":{...}}]}"}
* The inner body is an OpenAI streaming chunk (or "[DONE]"). We unwrap it
* and re-emit as `data: <inner>\n\n`. Errors become a synthetic OpenAI error
* chunk + [DONE].
* and re-emit as `data: <inner>\n\n`. First-frame errors become HTTP errors;
* errors after streaming starts retain the synthetic chunk + [DONE] path.
*
* Critical: Qoder's SSE often keeps the socket open after the terminal
* [DONE]/error frame (agent keepalive). Non-streaming clients drain via
@@ -401,24 +419,28 @@ async function peekFirstQoderFrame(reader, decoder) {
* usage from the finish chunk, so we coalesce those two frames (see
* createQoderSseCoalescer) before forwarding.
*
* NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl).
* If detected, return 403 response so chatCore marks connection unavailable
* and triggers combo fallback instead of leaking error text into chat.
* Peek the first frame for errors before committing to HTTP 200. Preserve
* upstream error statuses so chatCore can handle failures instead of recording
* error text as a successful completion. Billing blocks retain the existing
* 403 mapping for quota/account fallback.
*/
async function wrapQoderSSE(response, model) {
async function wrapQoderSSE(response, model, log = null) {
if (!response.ok || !response.body) return response;
const decoder = new TextDecoder();
const reader = response.body.getReader();
// Peek first frame to detect billing block
// Detect errors before returning a successful streaming response.
const peek = await peekFirstQoderFrame(reader, decoder);
if (peek?.isBilling) {
// Billing block detected — return 403 so chatCore fails this connection
if (peek.isError) {
await reader.cancel().catch(() => {});
const status = peek.isBilling
? HTTP_STATUS.FORBIDDEN
: Number.isInteger(peek.statusVal) && peek.statusVal >= HTTP_STATUS.BAD_REQUEST && peek.statusVal <= 599
? peek.statusVal : HTTP_STATUS.BAD_GATEWAY;
return new Response(
JSON.stringify({ error: { message: peek.message, code: peek.statusVal } }),
{ status: 403, headers: { "Content-Type": "application/json" } }
{ status, headers: { "Content-Type": "application/json" } }
);
}
@@ -449,11 +471,35 @@ async function wrapQoderSSE(response, model) {
let envelope;
try { envelope = JSON.parse(data); } catch { return; }
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
const statusVal = Number(envelope.statusCodeValue) || 200;
const inner = typeof envelope.body === "string"
? envelope.body
: envelope.body != null ? JSON.stringify(envelope.body) : "";
if (statusVal !== 200) {
// Always visible: error envelopes are rare and worth one stderr line at
// any log level (response bodies carry no credentials).
try {
console.error(`[QODER] error envelope status=${statusVal} statusType=${typeof envelope.statusCodeValue} bodyType=${typeof envelope.body} body=${truncate(inner, 300)}`);
} catch { /* logging must not break the stream */ }
if (isBillingBlock(inner)) {
// Billing/quota envelope at any stream position (peek only covers the
// first frame): emit a structured error chunk, not fake assistant text.
// parseSSEToOpenAIResponse understands chunk.error and turns it into a
// non-200 result so chat.js locks the model and falls back. Streaming
// clients receive a real SSE error instead of "[qoder error ...]" text.
const errObj = JSON.stringify({
error: {
message: inner || `qoder billing block (${statusVal})`,
code: "qoder_billing_block",
status: 403,
type: "quota_error",
},
});
controller.enqueue(encoder.encode(`data: ${errObj}\n\n`));
controller.enqueue(encoder.encode(SSE_DONE));
doneEmitted = true;
return;
}
const msg = inner || `upstream status ${statusVal}`;
const errChunk = JSON.stringify({
id: `qoder-error-${Date.now()}`,
@@ -552,12 +598,13 @@ async function wrapQoderSSE(response, model) {
}
export class QoderExecutor extends BaseExecutor {
constructor() {
super("qoder", PROVIDERS.qoder);
constructor(provider = "qoder") {
super(provider, PROVIDERS[provider]);
this.region = provider === "qoder-cn" ? "cn" : "intl";
}
buildUrl(credentials) {
return `${qoderInferenceBase(credentials)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
return `${qoderInferenceBase(credentials, this.region)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
}
// Override execute entirely — Qoder needs:
@@ -572,7 +619,7 @@ export class QoderExecutor extends BaseExecutor {
const rawToken = credentials?.apiKey || credentials?.accessToken;
if (isQoderPat(rawToken)) {
try {
credentials = await resolveQoderCredentials(credentials, proxyOptions, signal);
credentials = await resolveQoderCredentials(credentials, proxyOptions, signal, this.region);
} catch (err) {
log?.error?.("QODER", `PAT exchange failed: ${err.message}`);
const fakeResp = new Response(
@@ -607,7 +654,7 @@ export class QoderExecutor extends BaseExecutor {
let qoderKey;
let payload;
try {
({ qoderKey, payload } = await buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }));
({ qoderKey, payload } = await buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, region: this.region }));
} catch (err) {
const fakeResp = new Response(
JSON.stringify({ error: { message: err.message } }),
@@ -666,8 +713,15 @@ export class QoderExecutor extends BaseExecutor {
response = await proxyAwareFetch(
url,
{ method: "POST", headers, body: encodedBodyBuf, signal: mergedSignal },
proxyOptions,
// A failed proxy request may already have reached Qoder. Replaying
// the same COSY signature directly reuses its requestId and returns
// 403/code 103. Let the caller retry through execute() with fresh signing.
{ ...proxyOptions, strictProxy: true },
);
} catch (err) {
// strictProxy wraps transport errors; retain caller cancellation semantics.
if (mergedSignal.aborted) throw mergedSignal.reason;
throw err;
} finally {
clearTimeout(connectTimer);
}
@@ -677,7 +731,7 @@ export class QoderExecutor extends BaseExecutor {
return { response, url, headers, transformedBody: payload };
}
const wrapped = await wrapQoderSSE(response, `qoder/${qoderKey}`);
const wrapped = await wrapQoderSSE(response, `${this.provider}/${qoderKey}`, log);
return { response: wrapped, url, headers, transformedBody: payload };
}

View File

@@ -1,10 +1,15 @@
import { DefaultExecutor } from "./default.js";
import { getMimoAccountCookie, invalidateMimoAccountCookieCache, MIMO_API_BASE, MIMO_API_UA } from "../shared/mimoAccount.js";
import { getMimoAccountCookie, invalidateMimoAccountCookieCache, resolveMimoServerBase, MIMO_API_UA } from "../shared/mimoAccount.js";
// Desktop-exclusive Preview models. These are served by the account service's
// /api/route proxy, authorized by the Xiaomi account session (NOT the sk- key).
// See shared/mimoAccount.js for the session handshake.
const PREVIEW_MODELS = new Set(["mimo-x-pro-preview", "mimo-x-flash-preview"]);
// Dual-route v2.6 models.
// v2.6 models dynamically route to the account service when desktop session credentials
// (mimoPassToken or account cookie) are present to consume weekly quota, falling back to
// the cloud API (sk- key) otherwise.
const ACCOUNT_MODELS = new Set([
"mimo-v2.6-pro",
"mimo-v2.6-flash",
"mimo-v2.6-pro-ultraspeed",
]);
// Session cookie resolved in execute() (async) and read back by buildHeaders()
// (sync — BaseExecutor.execute does not await it). Carried on the per-request
@@ -23,15 +28,24 @@ export class XiaomiMimoExecutor extends DefaultExecutor {
super("xiaomi-mimo");
}
static isPreviewModel(model) {
return PREVIEW_MODELS.has(bareModel(model));
static isAccountRoute(model, credentials) {
const bare = bareModel(model);
if (!ACCOUNT_MODELS.has(bare)) return false;
return Boolean(
credentials?.[COOKIE_KEY] ||
credentials?.providerSpecificData?.mimoPassToken
);
}
isAccountRoute(model, credentials) {
return XiaomiMimoExecutor.isAccountRoute(model, credentials);
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Preview models live on the account-service route, which is not one of the
// Account route models live on the account-service route, which is not one of the
// declared transports — resolve it before the default runtimeTransport path.
if (XiaomiMimoExecutor.isPreviewModel(model)) {
return `${MIMO_API_BASE}/api/route/chat/completions`;
if (this.isAccountRoute(model, credentials)) {
return `${resolveMimoServerBase(credentials?.providerSpecificData)}/api/route/chat/completions`;
}
// Cloud API models keep default handling, so a Claude-format client reaches
// the /anthropic/v1/messages transport.
@@ -39,8 +53,8 @@ export class XiaomiMimoExecutor extends DefaultExecutor {
}
buildHeaders(credentials, stream = true, url, model) {
if (XiaomiMimoExecutor.isPreviewModel(model) && credentials?.[COOKIE_KEY]) {
// Preview models authenticate with the account-session cookie, not the key.
if (this.isAccountRoute(model, credentials) && credentials?.[COOKIE_KEY]) {
// Account route models authenticate with the account-session cookie, not the key.
return {
"Content-Type": "application/json",
Accept: stream ? "text/event-stream" : "application/json",
@@ -52,17 +66,22 @@ export class XiaomiMimoExecutor extends DefaultExecutor {
}
transformRequest(model, body, stream, credentials) {
// super runs stripUnsupportedParams, which flattens Preview content-part
// super runs stripUnsupportedParams, which flattens content-part
// arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js).
const out = super.transformRequest(model, body, stream, credentials);
// Preview models: thinking/params get defaults only — never override what the
// caller set explicitly. (body.model is already `xiaomi/<id>` via upstreamModelId.)
if (XiaomiMimoExecutor.isPreviewModel(model)) {
if (out.thinking == null) out.thinking = { type: "enabled" };
// Account route models: bridge reasoning_effort to official output_config.effort
// (matches MiMo Desktop app.asar behavior).
if (this.isAccountRoute(model, credentials)) {
const rawEffort = out.reasoning_effort || body?.reasoning_effort || body?.output_config?.effort;
if (rawEffort) {
delete out.reasoning_effort;
const norm = String(rawEffort).toLowerCase() === "xhigh" ? "high" : String(rawEffort).toLowerCase();
out.output_config = { ...(out.output_config || {}), effort: norm };
}
if (out.temperature == null) out.temperature = 1.0;
if (out.top_p == null) out.top_p = 0.95;
if (!out.max_tokens) out.max_tokens = 4096;
}
return out;
@@ -70,13 +89,11 @@ export class XiaomiMimoExecutor extends DefaultExecutor {
async execute(args) {
const { model, credentials, proxyOptions = null } = args;
if (!XiaomiMimoExecutor.isPreviewModel(model)) return super.execute(args);
if (!this.isAccountRoute(model, credentials)) return super.execute(args);
const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions);
if (!cookie) {
throw new Error(
"Xiaomi MiMo account session unavailable. Sign in to MiMo Desktop once so its passToken is present, then retry.",
);
return super.execute(args);
}
credentials[COOKIE_KEY] = cookie;
const result = await super.execute(args);
@@ -94,6 +111,6 @@ export class XiaomiMimoExecutor extends DefaultExecutor {
}
}
export const __test__ = { PREVIEW_MODELS, bareModel, COOKIE_KEY };
export const __test__ = { ACCOUNT_MODELS, bareModel, COOKIE_KEY };
export default XiaomiMimoExecutor;

View File

@@ -20,6 +20,7 @@ import { handleNonStreamingResponse } from "./chatCore/nonStreamingHandler.js";
import { handleStreamingResponse, buildOnStreamComplete } from "./chatCore/streamingHandler.js";
import { detectClientTool, isNativePassthrough } from "../utils/clientDetector.js";
import { dedupeTools } from "../utils/toolDeduper.js";
import { takeRenamedToolNames } from "../utils/opencodeFingerprint.js";
import { injectCaveman } from "../rtk/caveman.js";
import { injectPonytail } from "../rtk/ponytail.js";
import { compressMessages, formatRtkLog } from "../rtk/index.js";
@@ -116,6 +117,19 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
}
}
// Per-request opt-out: client can bypass all token savers via header
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
// Cursor's translator rewrites tool_result into user text, so RTK must run on
// the source body before translation. Every other pair translates the tool
// shapes 1:1 — keep the post-translate pass there so those providers are
// untouched (and a retry never re-compresses an already-compressed body).
const preTranslateRtk = provider === "cursor"
? compressMessages(body, tokenSaverEnabled && rtkEnabled)
: null;
const preTranslateRtkLine = formatRtkLog(preTranslateRtk);
if (preTranslateRtkLine) console.log(preTranslateRtkLine);
const clientRequestedStreaming = body.stream === true || sourceFormat === FORMATS.ANTIGRAVITY || sourceFormat === FORMATS.GEMINI || sourceFormat === FORMATS.GEMINI_CLI;
const providerRequiresStreaming = PROVIDERS[provider]?.forceStream === true;
let stream = providerRequiresStreaming ? true : (body.stream !== false);
@@ -254,11 +268,8 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
}
// Per-request opt-out: client can bypass all token savers via header
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
// RTK: compress tool_result content
const rtkStats = compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled);
// RTK: compress tool_result content. Skipped when already done pre-translate.
const rtkStats = preTranslateRtk || compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled);
const rtkLine = formatRtkLog(rtkStats);
if (rtkLine) log?.info?.("RTK", rtkLine.replace(/^\[RTK\] /, ""));
@@ -277,6 +288,8 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Token-saver flags accumulator for the single "⚙" log line below.
const xf = [];
if (rtkStats?.hits?.length) xf.push(`RTK:${rtkStats.hits.length}`);
// Caveman: inject terse-style system prompt
if (tokenSaverEnabled && cavemanEnabled && cavemanLevel) {
injectCaveman(translatedBody, finalFormat, cavemanLevel);
@@ -378,6 +391,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
providerHeaders = result.headers;
finalBody = result.transformedBody;
providerResponseFormat = result.responseFormat || targetFormat;
const renamedToolNames = takeRenamedToolNames(translatedBody);
if (renamedToolNames?.size) {
toolNameMap = new Map([...(toolNameMap || []), ...renamedToolNames]);
}
reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody);
} catch (error) {
trackPendingRequest(model, provider, connectionId, false, true);
@@ -508,7 +525,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Provider forced streaming but client wants JSON
if (!clientRequestedStreaming && providerRequiresStreaming) {
const result = await handleForcedSSEToJson({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, customToolNames, trackDone, appendLog });
const result = await handleForcedSSEToJson({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, customToolNames, toolNameMap, trackDone, appendLog });
if (result) { streamController.handleComplete(); return result; }
}

View File

@@ -11,6 +11,7 @@ import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, sav
import { saveRequestDetail } from "@/lib/usageDb.js";
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
import { decloakToolNames } from "../../utils/claudeCloaking.js";
import { restoreToolNames } from "../../utils/opencodeFingerprint.js";
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
function parseToolArguments(value) {
@@ -415,7 +416,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
return {
success: true,
response: new Response(JSON.stringify(translatedResponse), {
response: new Response(JSON.stringify(restoreToolNames(translatedResponse, toolNameMap)), {
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" }
})
};

View File

@@ -1,5 +1,6 @@
import { convertResponsesStreamToJson } from "../../transformer/streamToJsonConverter.js";
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
import { restoreToolNames } from "../../utils/opencodeFingerprint.js";
import { createErrorResult } from "../../utils/error.js";
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
import { FORMATS } from "../../translator/formats.js";
@@ -215,17 +216,13 @@ export async function handleForcedSSEToJson({
clientRawRequest,
onRequestSuccess,
customToolNames,
toolNameMap,
trackDone,
appendLog,
reqTag,
log,
streamErrorPatterns,
}) {
const contentType = providerResponse.headers.get("content-type") || "";
const isSSE =
contentType.includes("text/event-stream") ||
(contentType === "" && isResponsesProvider(provider));
if (!isSSE) return null; // not handled here
trackDone();
@@ -306,7 +303,7 @@ export async function handleForcedSSEToJson({
if (sourceFormat === FORMATS.OPENAI_RESPONSES) {
return {
success: true,
response: new Response(JSON.stringify(jsonResponse), {
response: new Response(JSON.stringify(restoreToolNames(jsonResponse, toolNameMap)), {
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
@@ -406,7 +403,7 @@ export async function handleForcedSSEToJson({
return {
success: true,
response: new Response(JSON.stringify(finalResp), {
response: new Response(JSON.stringify(restoreToolNames(finalResp, toolNameMap)), {
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
@@ -432,8 +429,16 @@ export async function handleForcedSSEToJson({
"Invalid SSE response for non-streaming request",
);
if (parsed.error) {
// Structured error chunks may carry the real upstream status (e.g. the
// Qoder executor emits status 403 for billing envelopes). Preserve it so
// the account loop locks/falls back on the right status instead of a
// generic 502. Anything outside 400-599 still maps to 502.
const upstreamStatus = Number(parsed.error.status);
const status = Number.isInteger(upstreamStatus) && upstreamStatus >= 400 && upstreamStatus <= 599
? upstreamStatus
: HTTP_STATUS.BAD_GATEWAY;
return createErrorResult(
HTTP_STATUS.BAD_GATEWAY,
status,
parsed.error.message || "Upstream SSE stream failed",
);
}
@@ -508,7 +513,7 @@ export async function handleForcedSSEToJson({
return {
success: true,
response: new Response(JSON.stringify(finalBody), {
response: new Response(JSON.stringify(restoreToolNames(finalBody, toolNameMap)), {
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",

View File

@@ -1,18 +1,93 @@
// HuggingFace Inference API — returns binary image
import { nowSec } from "./_base.js";
// HuggingFace Inference Providers router — returns binary image
//
// The router is a switchboard in front of many inference providers and is
// addressed as `<baseUrl>/<provider>/<providerModelId>`. `providerModelId` is
// the id the *provider* uses, which is not the Hub model id, so it is resolved
// through `imageConfig.modelMap` (built from the Hub API's
// inferenceProviderMapping and limited to providers the router forwards to).
//
// The legacy `api-inference.huggingface.co` host is gone (DNS ENOTFOUND) and is
// deliberately not referenced anywhere here.
import { nowSec, urlToBase64 } from "./_base.js";
import { PROVIDER_MEDIA } from "../../providers/index.js";
const BASE_URL = PROVIDER_MEDIA["huggingface"]?.imageConfig?.baseUrl;
const imageConfig = () => PROVIDER_MEDIA["huggingface"]?.imageConfig || {};
const BASE_URL = imageConfig().baseUrl;
const MODEL_MAP = imageConfig().modelMap || {};
// A plain-object lookup returns inherited truthy values for keys like "toString" or
// "constructor", which would build nonsense URLs. Resolve own keys only.
const lookup = (model) => (Object.hasOwn(MODEL_MAP, model) ? MODEL_MAP[model] : undefined);
// modelMap values are either a bare path (text-to-image) or { path, task }.
const mappingPath = (entry) => (typeof entry === "string" ? entry : entry.path);
const mappingTask = (entry) => (typeof entry === "string" ? "text-to-image" : entry.task || "text-to-image");
// A connection may point at its own endpoint (self-hosted Text Generation
// Inference / TGI container). That endpoint already knows its own model ids, so
// the router mapping does not apply and the Hub id is passed through verbatim.
function customBaseUrl(creds) {
const url = creds?.providerSpecificData?.baseUrl;
return typeof url === "string" && url.trim() ? url.trim().replace(/\/+$/, "") : null;
}
// The router's image-to-image payload wants raw base64 — not a data URL, not a URL.
// Accept every shape our own callers use (data URL, bare base64, remote URL, array).
async function sourceImage(body) {
const raw = body?.image || (Array.isArray(body?.images) ? body.images[0] : null);
if (typeof raw !== "string" || !raw.trim()) return null;
const value = raw.trim();
if (/^https?:\/\//i.test(value)) return await urlToBase64(value);
const match = /^data:image\/[^;]+;base64,(.+)$/i.exec(value);
return match ? match[1] : value;
}
export default {
buildUrl: (model) => `${BASE_URL}/${model}`,
buildUrl: (model, creds) => {
const override = customBaseUrl(creds);
if (override) {
// The model id is client-controlled; on a custom endpoint it lands in a URL
// path verbatim, so reject traversal/query injection (mirrors sttCore's guard).
if (model.includes("..") || model.includes("//") || /[?#]/.test(model)) {
throw new Error(`HuggingFace: invalid model ID "${model}"`);
}
return `${override}/${model}`;
}
const entry = lookup(model);
if (!entry) {
throw new Error(
`HuggingFace: no HuggingFace router mapping for model "${model}". ` +
`Add it to imageConfig.modelMap in open-sse/providers/registry/huggingface.js, ` +
`or set a custom base URL on the connection.`
);
}
return `${BASE_URL}/${mappingPath(entry)}`;
},
buildHeaders: (creds) => {
const headers = { "Content-Type": "application/json" };
const key = creds?.apiKey || creds?.accessToken;
if (key) headers["Authorization"] = `Bearer ${key}`;
return headers;
},
buildBody: (_model, body) => ({ inputs: body.prompt }),
buildBody: async (model, body) => {
const entry = lookup(model);
const task = mappingTask(entry || "");
if (task === "image-to-image") {
const image = await sourceImage(body);
if (!image) {
throw new Error(
`HuggingFace: model "${model}" requires a source image. ` +
`Send it as "image" (or "images") in the request body.`
);
}
// inputs carries the source image; the prompt moves under parameters.
return { inputs: image, parameters: { prompt: body.prompt } };
}
return { inputs: body.prompt };
},
// HF returns raw image bytes — convert to b64_json
async parseResponse(response) {
const buf = await response.arrayBuffer();

View File

@@ -0,0 +1,95 @@
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
import { HTTP_STATUS, FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
import { PROVIDER_MEDIA } from "../providers/index.js";
import { generateSessionId } from "../executors/opencode-zen.js";
/**
* Core System One (Jev) handler — native decision payload pass-through.
* URL/headers come from the registry's systemoneConfig; body and JSON response
* are forwarded untouched (decision models have no chat translation layer).
*
* @returns {Promise<{ success: boolean, response: Response, usage?: object, status?: number, error?: string }>}
*/
export async function handleSystemoneCore({
body,
modelInfo,
credentials,
log,
onRequestSuccess,
}) {
const { provider, model } = modelInfo;
const cfg = PROVIDER_MEDIA[provider]?.systemoneConfig;
if (!cfg?.baseUrl) {
return createErrorResult(
HTTP_STATUS.BAD_REQUEST,
`Provider '${provider}' does not support System One.`
);
}
// Validate input at the trust boundary; question-level shape is upstream's job.
if (body.state === undefined || body.state === null) {
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: state");
}
if (!body.questions || typeof body.questions !== "object" || Array.isArray(body.questions)) {
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: questions");
}
// noAuth free lanes carry accessToken "public" from the credential stub.
const token = credentials?.apiKey || credentials?.accessToken;
const headers = {
"Content-Type": "application/json",
...(token ? { Authorization: `Bearer ${token}` } : {}),
...(cfg.headers || {}),
// Zen lanes expect the official client session header on every request.
"x-opencode-session": generateSessionId(),
};
const requestBody = { ...body, model };
log?.debug?.("SYSTEMONE", `${provider.toUpperCase()} | ${model}`);
let providerResponse;
try {
providerResponse = await fetch(cfg.baseUrl, {
method: "POST",
headers,
body: JSON.stringify(requestBody),
...(typeof AbortSignal?.timeout === "function"
? { signal: AbortSignal.timeout(FETCH_CONNECT_TIMEOUT_MS) }
: {}),
});
} catch (error) {
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
log?.debug?.("SYSTEMONE", `Fetch error: ${errMsg}`);
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, errMsg);
}
if (!providerResponse.ok) {
const { statusCode, message } = await parseUpstreamError(providerResponse);
const errMsg = formatProviderError(new Error(message), provider, model, statusCode);
log?.debug?.("SYSTEMONE", `Provider error: ${errMsg}`);
return createErrorResult(statusCode, errMsg);
}
let responseBody;
try {
responseBody = await providerResponse.json();
} catch {
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Invalid JSON response from ${provider}`);
}
if (onRequestSuccess) await onRequestSuccess();
const usage = responseBody?.usage;
return {
success: true,
usage: usage
? { prompt_tokens: usage.input_tokens || 0, completion_tokens: usage.output_tokens || 0 }
: null,
response: new Response(JSON.stringify(responseBody), {
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
},
}),
};
}

View File

@@ -112,6 +112,8 @@ export const MODEL_CAPABILITIES = {
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "zai", contextWindow: 1000000, maxOutput: 131072 },
"glm-4.6v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000, maxOutput: 32768 },
"glm-4.5v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 64000, maxOutput: 16384 },
// GLM-5.2 has 1M context — pattern *glm-5* only gives 200k, so override here
"glm-5.2": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 131072 },
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
@@ -165,6 +167,12 @@ export const PROVIDER_CAPABILITIES = {
"deepseek-ai/deepseek-v4-pro": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
},
// glm-5.3-flash on OpenCode Go is served by a backend that rejects the z.ai
// `thinking` object (400: unknown field "thinking") and wants reasoning_effort.
// Overrides the global entry, whose z.ai shape is correct for z.ai itself.
"opencode-go": {
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 131072 },
},
"codex": {
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
@@ -205,6 +213,10 @@ export const PROVIDER_CAPABILITIES = {
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
// Per-model values mirror the server's product-config payload (the plugin
// fetches it from copilot.tencent.com; the `models[]` entries carry
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
@@ -226,45 +238,6 @@ export const PROVIDER_CAPABILITIES = {
// contract). maxOutput 128000 per the server's product-config payload.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors
// the codebuddy-cn entry (the openai-style reasoning_effort format matters:
// the generic *deepseek-v4* pattern would otherwise pick the vendor-native
// "deepseek" thinking shape, which the CodeBuddy gateway does not accept).
"codebuddy-intl": {
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
// (200K) without this map. contextWindow follows the real model family's
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
// max_output_tokens arrives as 0 for every model, so outputs are
// best-guess from the real model family. Vision tags below follow the
// upstream is_vl flag. The executor uploads inlined images to
// /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null
// (same as qodercli). reasoning:true on all of them — every model can
// reason; the upstream is_reasoning flag only drives model_config selection.
// thinkingFormat keeps the true-model family for documentation/UI, but
// thinkingCanDisable:false everywhere: the executor only forwards
// messages/tools/max_tokens, and thinking is fixed upstream via
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
// must never be offered as an option.
"qoder": {
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
},
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
"poolside": {
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
@@ -282,6 +255,10 @@ export const PROVIDER_CAPABILITIES = {
},
};
// Qoder CN serves the identical model catalog from the CN gateway, so it shares
// the intl Qoder capability table verbatim (vision/reasoning/contextWindow).
PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"];
/**
* Pattern fallback — glob (* = wildcard), matched case-insensitively and
* anchored (^...$) so a pattern must match the full model id. ORDER MATTERS:
@@ -351,7 +328,7 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*qwen*vl*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
{ pattern: "*qwen*omni*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144, maxOutput: 65536 } },
{ pattern: "*qwen*coder*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 } },
{ pattern: "*qwen*max*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen*max*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen3.5*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen3.6*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen3.7*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
@@ -390,14 +367,16 @@ export const PATTERN_CAPABILITIES = [
// ── MiniMax (M3 = adaptive; M2.x cannot disable) ─────────────────
{ pattern: "*minimax*image*", caps: { imageOutput: true } },
{ pattern: "*minimax-m3*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", contextWindow: 1048576, maxOutput: 512000 } },
{ pattern: "*minimax-m2.7*", caps: { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 204800, maxOutput: 131072 } },
{ pattern: "*minimax-m3*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", contextWindow: 1000000, maxOutput: 131072 } },
{ pattern: "*minimax-m2.7*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 204800, maxOutput: 131072 } },
{ pattern: "*minimax-m2.5*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 204800, maxOutput: 131072 } },
{ pattern: "*minimax*", caps: { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 131072 } },
// ── Xiaomi MiMo (vision, 1M / 262K ctx) ──────────────────────────
{ pattern: "*mimo*v2.5*", caps: { vision: true, audioInput: true, videoInput: true, contextWindow: 1048576, maxOutput: 131072 } },
{ pattern: "*mimo*omni*", caps: { vision: true, audioInput: true, contextWindow: 262144, maxOutput: 131072 } },
{ pattern: "*mimo*", caps: { vision: true, contextWindow: 262144, maxOutput: 131072 } },
// ── Xiaomi MiMo (vision + <think>-tag reasoning, always-on, can't disable) ──
{ pattern: "*mimo*v2.6*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 131072 } },
{ pattern: "*mimo*v2.5*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 131072 } },
{ pattern: "*mimo*omni*", caps: { vision: true, audioInput: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 131072 } },
{ pattern: "*mimo*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 131072 } },
// ── Llama (4 = vision/1M; 3.x = text-only/128K) ──────────────────
{ pattern: "*llama-4*", caps: { vision: true, contextWindow: 1000000 } },
@@ -440,6 +419,52 @@ export const PATTERN_CAPABILITIES = [
// unknown models on these providers, trust vision instead of stripping images.
const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
/**
* Aggregate capabilities for a combo from its constituent model IDs.
* Each entry in comboModels is a fully-qualified "provider/model" string.
*
* Union: vision, pdf, audioInput, videoInput, imageOutput, audioOutput, search
* Intersection: tools
* Primary: reasoning fields from the first (primary) model
* Conservative: contextWindow = min; maxOutput = max
*
* @param {string[]} comboModels
* @param {Object|null} [comboLookup] optional map of combo name → models array for nested resolution
* @param {number} [_depth] internal recursion depth guard
* @returns {object|null} full capabilities object, or null for empty input
*/
export function aggregateComboCapabilities(comboModels, comboLookup = null, _depth = 0) {
if (!comboModels?.length || _depth > 6) return null;
const allCaps = comboModels.map((fullId) => {
// Nested combo: bare name (no slash) that exists in the lookup — recurse
if (!fullId.includes("/") && comboLookup?.[fullId]) {
return aggregateComboCapabilities(comboLookup[fullId], comboLookup, _depth + 1)
?? getCapabilitiesForModel(null, fullId);
}
const slash = fullId.indexOf("/");
const provider = slash === -1 ? null : fullId.slice(0, slash);
const model = slash === -1 ? fullId : fullId.slice(slash + 1);
return getCapabilitiesForModel(provider, model);
});
const first = allCaps[0];
return {
vision: allCaps.some((c) => c.vision),
pdf: allCaps.some((c) => c.pdf),
audioInput: allCaps.some((c) => c.audioInput),
videoInput: allCaps.some((c) => c.videoInput),
imageOutput: allCaps.some((c) => c.imageOutput),
audioOutput: allCaps.some((c) => c.audioOutput),
search: allCaps.some((c) => c.search),
tools: allCaps.every((c) => c.tools),
reasoning: first.reasoning,
thinkingFormat: first.thinkingFormat,
thinkingCanDisable: first.thinkingCanDisable,
thinkingRange: first.thinkingRange,
contextWindow: Math.min(...allCaps.map((c) => c.contextWindow)),
maxOutput: Math.max(...allCaps.map((c) => c.maxOutput)),
};
}
/**
* Resolve capabilities for a model using the 4-step fallback chain,
* merged over DEFAULT_CAPABILITIES so the result is always complete.

View File

@@ -23,7 +23,7 @@ function buildTransport(transport, oauth) {
const MEDIA_KEYS = new Set([
"serviceKinds", "ttsConfig", "sttConfig", "embeddingConfig",
"imageConfig", "imageToTextConfig", "videoConfig", "musicConfig",
"searchViaChat", "searchConfig", "fetchConfig",
"searchViaChat", "searchConfig", "fetchConfig", "systemoneConfig",
"modelsFetcher", "mediaPriority", "hiddenKinds",
]);

View File

@@ -57,6 +57,7 @@ export default {
},
},
models: [
{ id: "claude-opus-5-5", name: "Claude Opus 5.5" },
{ id: "claude-opus-5", name: "Claude Opus 5" },
{ id: "claude-fable-5-1", name: "Claude Fable 5.1" },
{ id: "claude-fable-5", name: "Claude Fable 5" },

View File

@@ -46,7 +46,7 @@ export default {
{ id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "moonshotai/Kimi-K2.7-Code", name: "Kimi K2.7 Code" },
{ id: "moonshotai/Kimi-K2.7-Code-Highspeed", name: "Kimi K2.7 Code Highspeed" },
{ id: "moonshotai/Kimi-K2.7-Code-Highspeed", name: "Kimi K2.7 Code HighSpeed" },
{ id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6" },
{ id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5" },
{ id: "zai-org/GLM-5.2", name: "GLM 5.2" },
@@ -58,14 +58,14 @@ export default {
{ id: "MiniMaxAI/MiniMax-M2.5", name: "MiniMax M2.5" },
{ id: "xiaomi/mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
{ id: "xiaomi/mimo-v2.5", name: "MiMo V2.5" },
{ id: "Qwen/Qwen3.7-Max", name: "Qwen 3.7 Max" },
{ id: "Qwen/Qwen3.7-Plus", name: "Qwen 3.7 Plus" },
{ id: "Qwen/Qwen3.6-Max-Preview", name: "Qwen 3.6 Max Preview" },
{ id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" },
{ id: "Qwen/Qwen3.7-Max", name: "Qwen 3.7 Max" },
{ id: "Qwen/Qwen3.7-Plus", name: "Qwen 3.7 Plus" },
{ id: "stepfun/Step-3.7-Flash", name: "Step 3.7 Flash" },
{ id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" },
{ id: "tencent/Hy3", name: "Tencent Hy3" },
{ id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B A55B" },
{ id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra" },
{ id: "thinkingmachines/inkling", name: "Inkling" },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6" },

View File

@@ -15,6 +15,7 @@ export default {
website: "https://huggingface.co",
notice: {
apiKeyUrl: "https://huggingface.co/settings/tokens",
text: "Runs through the Inference Providers router. Image and speech models are billed by the provider selected per model.",
},
},
category: "apikey",
@@ -25,10 +26,79 @@ export default {
transport: null,
models: [
{ id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell", params: [], kind: "image" },
{ id: "black-forest-labs/FLUX.1-dev", name: "FLUX.1 Dev", params: [], kind: "image" },
{ id: "black-forest-labs/FLUX.1-Krea-dev", name: "FLUX.1 Krea", params: [], kind: "image" },
{ id: "black-forest-labs/FLUX.1-Kontext-dev", name: "FLUX.1 Kontext", params: [], kind: "image", capabilities: ["edit"] },
{ id: "black-forest-labs/FLUX.2-dev", name: "FLUX.2 Dev", params: [], kind: "image", capabilities: ["edit"] },
{ id: "black-forest-labs/FLUX.2-klein-9B", name: "FLUX.2 Klein 9B", params: [], kind: "image", capabilities: ["edit"] },
{ id: "black-forest-labs/FLUX.2-klein-4B", name: "FLUX.2 Klein 4B", params: [], kind: "image", capabilities: ["edit"] },
{ id: "black-forest-labs/FLUX.2-klein-base-9B", name: "FLUX.2 Klein Base 9B", params: [], kind: "image", capabilities: ["edit"] },
{ id: "black-forest-labs/FLUX.2-klein-base-4B", name: "FLUX.2 Klein Base 4B", params: [], kind: "image", capabilities: ["edit"] },
{ id: "stabilityai/stable-diffusion-xl-base-1.0", name: "SDXL Base 1.0", params: [], kind: "image" },
{ id: "openai/whisper-large-v3", name: "Whisper Large v3 (HF)", params: ["language"], kind: "stt" },
{ id: "openai/whisper-small", name: "Whisper Small (HF)", params: ["language"], kind: "stt" },
{ id: "stabilityai/stable-diffusion-3.5-large", name: "Stable Diffusion 3.5 Large", params: [], kind: "image" },
{ id: "stabilityai/stable-diffusion-3.5-large-turbo", name: "Stable Diffusion 3.5 Large Turbo", params: [], kind: "image" },
{ id: "Qwen/Qwen-Image", name: "Qwen Image", params: [], kind: "image" },
{ id: "Qwen/Qwen-Image-2512", name: "Qwen Image 2512", params: [], kind: "image" },
{ id: "Qwen/Qwen-Image-Edit", name: "Qwen Image Edit", params: [], kind: "image", capabilities: ["edit"] },
{ id: "Qwen/Qwen-Image-Edit-2509", name: "Qwen Image Edit 2509", params: [], kind: "image", capabilities: ["edit"] },
{ id: "Qwen/Qwen-Image-Edit-2511", name: "Qwen Image Edit 2511", params: [], kind: "image", capabilities: ["edit"] },
{ id: "ideogram-ai/ideogram-4-fp8", name: "Ideogram 4", params: [], kind: "image" },
{ id: "tencent/HunyuanImage-3.0", name: "HunyuanImage 3.0", params: [], kind: "image" },
{ id: "Tongyi-MAI/Z-Image-Turbo", name: "Z-Image Turbo", params: [], kind: "image" },
{ id: "krea/Krea-2-Turbo", name: "Krea 2 Turbo", params: [], kind: "image" },
{ id: "HiDream-ai/HiDream-I1-Fast", name: "HiDream I1 Fast", params: [], kind: "image" },
{ id: "playgroundai/playground-v2.5-1024px-aesthetic", name: "Playground v2.5", params: [], kind: "image" },
{ id: "openai/whisper-large-v3", name: "Whisper Large v3 (HF)", params: [], kind: "stt" },
{ id: "openai/whisper-large-v3-turbo", name: "Whisper Large v3 Turbo (HF)", params: [], kind: "stt" },
],
serviceKinds: ["image", "stt"],
imageConfig: { baseUrl: "https://api-inference.huggingface.co/models" },
// Inference Providers router. The router is addressed as
// `<baseUrl>/<provider>/<providerModelId>` — see open-sse/handlers/imageProviders/huggingface.js.
// `modelMap` resolves a Hub model id to the provider-resolved id the router expects.
// A plain string value is the provider path. Image-to-image models use
// `{ path, task: "image-to-image" }`: their payload differs — the source image goes in
// `inputs` and the prompt under `parameters.prompt`. See
// https://huggingface.co/docs/inference-providers/tasks/image-to-image
// Only providers the router actually forwards to are listed: replicate, wavespeed and
// deepinfra appear in the Hub's inferenceProviderMapping but reject router traffic with
// "Model not supported by provider <name>".
imageConfig: {
baseUrl: "https://router.huggingface.co",
modelMap: {
"black-forest-labs/FLUX.1-schnell": "fal-ai/fal-ai/flux/schnell",
"black-forest-labs/FLUX.1-dev": "fal-ai/fal-ai/flux/dev",
"black-forest-labs/FLUX.1-Krea-dev": "fal-ai/fal-ai/flux/krea",
"black-forest-labs/FLUX.1-Kontext-dev": { path: "fal-ai/fal-ai/flux-kontext/dev", task: "image-to-image" },
"black-forest-labs/FLUX.2-dev": { path: "fal-ai/fal-ai/flux-2/edit", task: "image-to-image" },
"black-forest-labs/FLUX.2-klein-9B": { path: "fal-ai/fal-ai/flux-2/klein/9b/edit", task: "image-to-image" },
"black-forest-labs/FLUX.2-klein-4B": { path: "fal-ai/fal-ai/flux-2/klein/4b/distilled/edit", task: "image-to-image" },
"black-forest-labs/FLUX.2-klein-base-9B": { path: "fal-ai/fal-ai/flux-2/klein/9b/base/edit", task: "image-to-image" },
"black-forest-labs/FLUX.2-klein-base-4B": { path: "fal-ai/fal-ai/flux-2/klein/4b/base/edit", task: "image-to-image" },
"stabilityai/stable-diffusion-xl-base-1.0": "fal-ai/fal-ai/fast-sdxl",
"stabilityai/stable-diffusion-3.5-large": "fal-ai/fal-ai/stable-diffusion-v35-large",
"stabilityai/stable-diffusion-3.5-large-turbo": "fal-ai/fal-ai/stable-diffusion-v35-large/turbo",
"Qwen/Qwen-Image": "fal-ai/fal-ai/qwen-image",
"Qwen/Qwen-Image-2512": "fal-ai/fal-ai/qwen-image-2512",
"Qwen/Qwen-Image-Edit": { path: "fal-ai/fal-ai/qwen-image-edit", task: "image-to-image" },
"Qwen/Qwen-Image-Edit-2509": { path: "fal-ai/fal-ai/qwen-image-edit-2509", task: "image-to-image" },
"Qwen/Qwen-Image-Edit-2511": { path: "fal-ai/fal-ai/qwen-image-edit-plus", task: "image-to-image" },
"ideogram-ai/ideogram-4-fp8": "fal-ai/ideogram/v4",
"tencent/HunyuanImage-3.0": "fal-ai/fal-ai/hunyuan-image/v3/text-to-image",
"Tongyi-MAI/Z-Image-Turbo": "fal-ai/fal-ai/z-image/turbo",
"krea/Krea-2-Turbo": "fal-ai/fal-ai/krea-2/turbo",
"HiDream-ai/HiDream-I1-Fast": "fal-ai/fal-ai/hidream-i1-fast",
"playgroundai/playground-v2.5-1024px-aesthetic": "fal-ai/fal-ai/playground-v25",
},
},
// Speech-to-text goes through the hf-inference provider, which keeps the Hub
// model id as its provider-resolved id (`/hf-inference/models/<hubId>`).
// No `params` are declared: the router's ASR payload carries only `inputs` and
// `parameters.return_timestamps` / `parameters.generation_parameters` — it has no
// language field, so a UI-declared "language" would be silently dropped.
sttConfig: {
baseUrl: "https://router.huggingface.co/hf-inference/models",
authType: "apikey",
authHeader: "bearer",
format: "huggingface-asr",
},
};

View File

@@ -69,6 +69,7 @@ import p66 from "./ollama.js";
import p123 from "./ollama-search.js";
import p67 from "./openai.js";
import p68 from "./opencode-go.js";
import p68z from "./opencode-zen.js";
import p69 from "./opencode.js";
import p70 from "./openrouter.js";
import p71 from "./perplexity-web.js";
@@ -76,6 +77,7 @@ import p72 from "./perplexity.js";
import p73 from "./perplexity-agent.js";
import p74 from "./playht.js";
import p75 from "./qoder.js";
import p124 from "./qoder-cn.js";
import p77 from "./recraft.js";
import p78 from "./runwayml.js";
import p79 from "./sdwebui.js";
@@ -192,8 +194,10 @@ export default [
p65,
p66,
p123,
p124,
p67,
p68,
p68z,
p69,
p70,
p71,

View File

@@ -28,6 +28,7 @@ export default {
forceStream: true,
},
models: [
{ id: "gpt-5.5", name: "GPT-5.5" },
{ id: "gpt-5.4", name: "GPT-5.4" },
{ id: "gpt-5.4-mini", name: "GPT-5.4 Mini" },
{ id: "gpt-5.4-nano", name: "GPT-5.4 Nano" },

View File

@@ -0,0 +1,135 @@
export default {
id: "opencode-zen",
priority: 205,
alias: "ocz",
aliases: [
"opencode-zen",
],
uiAlias: "ocz",
display: {
name: "OpenCode Zen",
icon: "terminal",
color: "#E87040",
textIcon: "OZ",
website: "https://opencode.ai/auth",
notice: {
text: "OpenCode Zen PAYG: pay-as-you-go, key from https://opencode.ai/auth. Same models as Zen: paid + free tiers on the fast lane.",
apiKeyUrl: "https://opencode.ai/auth",
},
},
category: "apikey",
transport: {
baseUrl: "https://opencode.ai/zen/v1/chat/completions",
headers: {},
usage: {
url: "https://opencode.ai/zen/v1/usage",
},
},
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
// translation. Mirrors opencode-go, pointed at /zen/v1 (see https://opencode.ai/docs/zen/).
transports: [
{ format: "openai", baseUrl: "https://opencode.ai/zen/v1/chat/completions", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
{ format: "claude", baseUrl: "https://opencode.ai/zen/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
],
// supportedFormats follow the endpoint table in https://opencode.ai/docs/zen/
// (live /zen/v1/models, 2026-09-18: 71 ids).
models: [
// Claude (messages)
{ id: "claude-fable-5", name: "Claude Fable 5", supportedFormats: ["claude"] },
{ id: "claude-fable-5-1", name: "Claude Fable 5.1", supportedFormats: ["claude"] },
{ id: "claude-opus-5", name: "Claude Opus 5", supportedFormats: ["claude"] },
{ id: "claude-opus-4-8", name: "Claude Opus 4.8", supportedFormats: ["claude"] },
{ id: "claude-opus-4-7", name: "Claude Opus 4.7", supportedFormats: ["claude"] },
{ id: "claude-opus-4-6", name: "Claude Opus 4.6", supportedFormats: ["claude"] },
{ id: "claude-opus-4-5", name: "Claude Opus 4.5", supportedFormats: ["claude"] },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5", supportedFormats: ["claude"] },
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6", supportedFormats: ["claude"] },
{ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", supportedFormats: ["claude"] },
{ id: "claude-sonnet-4", name: "Claude Sonnet 4", supportedFormats: ["claude"] },
{ id: "claude-haiku-4-5", name: "Claude Haiku 4.5", supportedFormats: ["claude"] },
// Gemini (own path, via chat completions transport)
{ id: "gemini-3.6-flash", name: "Gemini 3.6 Flash", supportedFormats: ["openai"] },
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash", supportedFormats: ["openai"] },
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash", supportedFormats: ["openai"] },
{ id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite", supportedFormats: ["openai"] },
{ id: "gemini-3.5-flash", name: "Gemini 3.5 Flash", supportedFormats: ["openai"] },
{ id: "gemini-3.1-pro", name: "Gemini 3.1 Pro", supportedFormats: ["openai"] },
{ id: "gemini-3-flash", name: "Gemini 3 Flash", supportedFormats: ["openai"] },
// GPT / Grok / Muse Spark paid (responses)
{ id: "gpt-6-astra", name: "GPT 6 Astra", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.5", name: "GPT 5.5", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.5-pro", name: "GPT 5.5 Pro", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.4", name: "GPT 5.4", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.4-pro", name: "GPT 5.4 Pro", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.4-mini", name: "GPT 5.4 Mini", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.4-nano", name: "GPT 5.4 Nano", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.3-codex", name: "GPT 5.3 Codex", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.2", name: "GPT 5.2", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.2-codex", name: "GPT 5.2 Codex", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.1", name: "GPT 5.1", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.1-codex-max", name: "GPT 5.1 Codex Max", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.1-codex", name: "GPT 5.1 Codex", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.1-codex-mini", name: "GPT 5.1 Codex Mini", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5", name: "GPT 5", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5-codex", name: "GPT 5 Codex", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5-nano", name: "GPT 5 Nano", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "grok-build-0.1", name: "Grok Build 0.1", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "grok-4.5", name: "Grok 4.5", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3", name: "Muse Spark 1.3", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2", name: "Muse Spark 1.2", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
// Qwen paid (messages)
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["claude"] },
{ id: "qwen3.5-plus", name: "Qwen 3.5 Plus", supportedFormats: ["claude"] },
// DeepSeek / GLM / MiniMax / Kimi / Big Pickle (chat completions)
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision Exp", supportedFormats: ["openai"] },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "glm-5", name: "GLM 5", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai"] },
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai"] },
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai"] },
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "kimi-k2.5", name: "Kimi K2.5", supportedFormats: ["openai"] },
{ id: "big-pickle", name: "Big Pickle", supportedFormats: ["openai"] },
{ id: "union-alpha", name: "Union Alpha", supportedFormats: ["claude"] },
// Free tier on the keyed lane (chat completions)
{ id: "deepseek-v4-flash-free", name: "DeepSeek V4 Flash Free", supportedFormats: ["openai"] },
{ id: "mimo-v2.6-flash-free", name: "MiMo V2.6 Flash Free", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-free", name: "MiMo V2.5 Free", supportedFormats: ["openai"] },
{ id: "ling-3.0-flash-fin-free", name: "Ling 3.0 Flash Fin Free", supportedFormats: ["openai"] },
{ id: "nemotron-3-ultra-free", name: "Nemotron 3 Ultra Free", supportedFormats: ["openai"] },
{ id: "nemotron-3.5-lightning-free", name: "Nemotron 3.5 Lightning Free", supportedFormats: ["openai"] },
// Free tier on the keyed lane (responses)
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
// System One (Jev) decision models on the native /systemone endpoint
{ id: "jev-1.13", name: "Jev 1.13", kind: "systemone" },
{ id: "jev-1.13-free", name: "Jev 1.13 Free", kind: "systemone" },
],
serviceKinds: ["llm", "systemone"],
systemoneConfig: {
baseUrl: "https://opencode.ai/zen/v1/systemone",
headers: {
"x-opencode-client": "desktop",
"User-Agent": "opencode/1.18.31",
},
},
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true,
features: {
usage: true,
usageApikey: true,
},
};

View File

@@ -28,7 +28,16 @@ export default {
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
{ id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" },
{ id: "jev-1.13-free", name: "Jev 1.13 Free", kind: "systemone" },
],
serviceKinds: ["llm", "systemone"],
systemoneConfig: {
baseUrl: "https://opencode.ai/zen/v1/systemone",
headers: {
"x-opencode-client": "desktop",
"User-Agent": "opencode/1.18.31",
},
},
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true,
};

View File

@@ -43,8 +43,14 @@ export default {
{ id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "typesafe/jev-1.13", name: "Jev 1.13", kind: "systemone" },
],
serviceKinds: ["llm","embedding","tts","imageToText","video"],
serviceKinds: ["llm","embedding","tts","imageToText","video","systemone"],
// System One decision API (TypeSafe-compatible): https://openrouter.ai/docs/guides/community/typesafe-sdk
systemoneConfig: {
baseUrl: "https://openrouter.ai/api/v1/systemone",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
ttsConfig: {
baseUrl: "https://openrouter.ai/api/v1/chat/completions",
defaultModel: "openai/gpt-4o-mini-tts",

View File

@@ -0,0 +1,61 @@
export default {
id: "qoder-cn",
priority: 30,
alias: "qdcn",
uiAlias: "qdcn",
display: {
name: "Qoder CN",
icon: "water_drop",
color: "#EC4899",
website: "https://qoder.com.cn",
notice: {
signupUrl: "https://qoder.com.cn",
},
},
category: "oauth",
authModes: ["oauth", "apikey"],
hasOAuth: true,
authHint: "Personal Access Token (pt-...) from https://qoder.com.cn/account/integrations",
transport: {
baseUrl: "https://gateway.qoder.com.cn/algo/api/v2/service/pro/sse/agent_chat_generation",
headers: {},
timeoutMs: 120000,
stallTimeoutMs: 120000,
usage: {
url: "https://openapi.qoder.com.cn/api/v2/quota/usage",
},
},
models: [
{ id: "ultimate", name: "Ultimate" },
{ id: "auto", name: "Auto" },
{ id: "performance", name: "Performance" },
{ id: "efficient", name: "Efficient" },
{ id: "lite", name: "Lite" },
{ id: "qmodel_38max", name: "Qwen3.8-Max" },
{ id: "qmodel_latest", name: "Qwen3.7-Max" },
{ id: "qmodel", name: "Qwen3.7-Plus" },
{ id: "qfmodel", name: "Qwen3.8-Flash" },
{ id: "kmodel_latest", name: "Kimi-K3" },
{ id: "kmodel", name: "Kimi-K2.7-Code" },
{ id: "gmodel", name: "GLM-5.3" },
{ id: "gfmodel", name: "GLM-5.3-Flash" },
{ id: "dmodel", name: "DeepSeek-V4-Pro" },
{ id: "dfmodel", name: "DeepSeek-V4-Flash" },
{ id: "mmodel", name: "MiniMax-M3" },
],
oauth: {
openApiBaseUrl: "https://openapi.qoder.com.cn",
centerBaseUrl: "https://gateway.qoder.com.cn",
chatBaseUrl: "https://gateway.qoder.com.cn",
deviceTokenUrl: "https://openapi.qoder.com.cn/api/v1/deviceToken/poll",
refreshUrl: "https://gateway.qoder.com.cn/algo/api/v3/user/refresh_token",
userInfoUrl: "https://openapi.qoder.com.cn/api/v1/userinfo",
quotaUsageUrl: "https://openapi.qoder.com.cn/api/v2/quota/usage",
loginUrl: "https://qoder.com.cn/device/selectAccounts",
},
features: {
usage: true,
// PAT (apikey) connections also carry quota usage (via job-token exchange).
usageApikey: true,
},
};

View File

@@ -2,9 +2,9 @@ import { CLAUDE_API_HEADERS } from "../shared.js";
// Dual auth (same pattern as kimi):
// - API key (sk-...) → cloud API on api.xiaomimimo.com
// - Desktop account/OAuth → same cloud host, plus the Desktop-exclusive Preview
// models served by the account-service route on mimo-server-cn.xiaomimimo.com
// (authorized by a Xiaomi account session cookie, not the key).
// - Desktop account/OAuth → same cloud host, plus the dual-route v2.6 models
// served by the account-service route (mimo-server-<cluster>.xiaomimimo.com),
// authorized by a Xiaomi account session cookie, not the key.
// Endpoint is picked per model in the executor, same as opencode-go's /responses split.
export default {
id: "xiaomi-mimo",
@@ -30,6 +30,16 @@ export default {
category: "oauth",
authModes: ["oauth", "apikey"],
hasOAuth: true,
// Keys are cluster-specific. MiMo Desktop declares five regions
// (CN/SGP/AMS/RU/IN) — host + sid follow mimo-server-<code> / mimo<code>.
regions: [
{ id: "cn", label: "China (中国大陆)" },
{ id: "sgp", label: "Singapore (新加坡)" },
{ id: "ams", label: "Europe · Amsterdam (欧洲)" },
{ id: "ru", label: "Russia (俄罗斯)" },
{ id: "in", label: "India (印度)" },
],
defaultRegion: "sgp",
serviceKinds: ["llm", "tts"],
transport: {
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
@@ -50,10 +60,10 @@ export default {
},
],
models: [
// Desktop-exclusive — served by the account-service route, which only accepts
// OpenAI format, so supportedFormats pins them to the openai transport.
{ id: "mimo-x-pro-preview", name: "MiMo-X-Pro-Preview", upstreamModelId: "xiaomi/mimo-x-pro-preview", supportedFormats: ["openai"] },
{ id: "mimo-x-flash-preview", name: "MiMo-X-Flash-Preview", upstreamModelId: "xiaomi/mimo-x-flash-preview", supportedFormats: ["openai"] },
// Cloud API & Desktop dual-route models (prefers the desktop account quota when available)
{ id: "mimo-v2.6-pro", name: "MiMo V2.6 Pro", upstreamModelId: "xiaomi/mimo-v2.6-pro", supportedFormats: ["openai"] },
{ id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", upstreamModelId: "xiaomi/mimo-v2.6-flash", supportedFormats: ["openai"] },
{ id: "mimo-v2.6-pro-ultraspeed", name: "MiMo V2.6 Pro UltraSpeed", upstreamModelId: "xiaomi/mimo-v2.6-pro-ultraspeed", supportedFormats: ["openai"] },
// Cloud API models (api.xiaomimimo.com/v1)
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
{ id: "mimo-v2.5", name: "MiMo V2.5" },

View File

@@ -36,6 +36,13 @@ import { DEFAULT_RETRY_CONFIG, FETCH_CONNECT_TIMEOUT_MS } from "../config/runtim
* MediaConfig: { serviceKinds:[...], ttsConfig, sttConfig, embeddingConfig, imageConfig,
* searchViaChat:{defaultModel,pricingUrl}, hiddenKinds } — each *Config: {baseUrl,authType,authHeader,
* format,defaultModel,models:[{id,name,dimensions?}]}.
*
* imageConfig.modelMap (optional): maps a client-facing model id to a provider-resolved id when
* those differ — e.g. the HuggingFace Inference Providers router, where a Hub id like
* `black-forest-labs/FLUX.1-schnell` is addressed as `fal-ai/fal-ai/flux/schnell`. A value is
* either the provider path, or `{path, task}` when the request shape differs per task
* (HuggingFace uses task:"image-to-image" to move the prompt under `parameters.prompt`).
* Ignored by providers whose model ids are sent verbatim.
*/
// Shared transport defaults — provider only overrides fields that differ.

View File

@@ -22,7 +22,7 @@ export function mapStainlessArch() {
// Anthropic API version (single source — reused across claude-format providers/executors)
export const ANTHROPIC_API_VERSION = "2023-06-01";
export const CLAUDE_CLI_VERSION = "2.1.258";
export const CLAUDE_CLI_VERSION = "2.1.280";
// Shared Claude-compatible API headers (reused across claude-format providers)
export const CLAUDE_API_HEADERS = {

View File

@@ -41,6 +41,7 @@ const PATTERN_THINKING = [
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
{ pattern: "*mimo*v2.6*", levels: ["none", "low", "medium", "high", "xhigh"] },
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
// (disable thinking instead). none kept for the picker = disable.

View File

@@ -1,5 +1,5 @@
// RTK port: compress tool_result content in LLM request bodies
// Injected at the top of translateRequest (before any format translation)
// Applied in chatCore on the source-format body, before translateRequest.
import { RAW_CAP, MIN_COMPRESS_SIZE } from "./constants.js";
import { autoDetectFilter } from "./autodetect.js";
import { safeApply } from "./applyFilter.js";

View File

@@ -12,19 +12,20 @@ import { getCapabilitiesForModel } from "../providers/capabilities.js";
const CAPABILITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
const HARD_CAPS = new Set(CAPABILITY_KEYS);
const DEFAULT_FALLBACK_MODEL = "oc/mimo-v2.5-free";
const DEFAULT_FALLBACK_MODEL = "oc/mimo-v2.6-flash-free";
const upgradeLegacyModel = (m) => (m === "oc/mimo-v2.5-free" ? DEFAULT_FALLBACK_MODEL : m);
// Normalize a capability entry to { enabled, roundRobin, models }. Backward-compat:
// accept the legacy array form [{model, enabled}] (treated as enabled, fallback).
function normalizeCapEntry(entry) {
if (Array.isArray(entry)) {
return { enabled: true, roundRobin: false, models: entry.map((e) => e?.model || e).filter(Boolean) };
return { enabled: true, roundRobin: false, models: entry.map((e) => upgradeLegacyModel(e?.model || e)).filter(Boolean) };
}
if (entry && typeof entry === "object") {
return {
enabled: entry.enabled !== false,
roundRobin: !!entry.roundRobin,
models: Array.isArray(entry.models) ? entry.models.filter(Boolean) : [],
models: Array.isArray(entry.models) ? entry.models.map(upgradeLegacyModel).filter(Boolean) : [],
};
}
return { enabled: false, roundRobin: false, models: [] };

View File

@@ -13,9 +13,13 @@
*
* PAT (Personal Access Token, pt-...) connections: a PAT cannot sign COSY
* requests directly, so we exchange it for a short-lived job token (jt-...)
* via openapi.qoder.sh/api/v1/jobToken/exchange (plain JSON POST), then use
* that job token for signing. Job-token traffic must hit api2.qoder.sh —
* api3 rejects jt- with "Login expired" (403).
* via the region's jobToken/exchange endpoint (plain JSON POST), then use
* that job token for signing. On intl, job-token traffic must hit api2.qoder.sh —
* api3 rejects jt- with "Login expired" (403); CN serves it from the same
* gateway host.
*
* The region (intl/cn) is derived from credentials.provider (or an explicit
* options.region override) so the same catalog logic works for both sites.
*/
import { createHash } from "crypto";
@@ -23,12 +27,12 @@ import { createHash } from "crypto";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { buildCosyHeaders } from "../shared/qoder/cosy.js";
import {
QODER_MODEL_LIST_URL,
QODER_CHAT_BASE_ALT,
QODER_JOB_TOKEN_EXCHANGE_URL,
QODER_USERINFO_URL,
QODER_IDE_VERSION,
QODER_CLIENT_TYPE,
qoderRegionOf,
qoderJobTokenExchangeUrl,
qoderUserInfoUrl,
qoderInferenceBase,
} from "../shared/qoder/constants.js";
const FETCH_TIMEOUT_MS = 15_000;
@@ -63,9 +67,9 @@ const inflight = new Map();
* Exchange a Qoder PAT (pt-...) for a short-lived job token (jt-...).
* This endpoint is plain JSON POST — NOT COSY-signed.
*/
async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
async function exchangeJobToken(pat, proxyOptions = null, signal = null, region = "intl") {
const res = await proxyAwareFetch(
QODER_JOB_TOKEN_EXCHANGE_URL,
qoderJobTokenExchangeUrl(region),
{
method: "POST",
headers: {
@@ -101,10 +105,10 @@ async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
* Resolve the Qoder userId for a job token (needed for COSY signing).
* Returns "" on any failure — callers fall back to the stored userId.
*/
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null) {
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null, region = "intl") {
try {
const res = await proxyAwareFetch(
QODER_USERINFO_URL,
qoderUserInfoUrl(region),
{
method: "GET",
headers: {
@@ -125,16 +129,17 @@ async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = nu
}
/**
* Resolve a PAT to a job-token credential, cached per-PAT.
* Resolve a PAT to a job-token credential, cached per-PAT-per-region.
*/
async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
const cached = patJobCache.get(pat);
async function resolvePatCredential(pat, proxyOptions = null, signal = null, region = "intl") {
const cacheKey = `${region}:${pat}`;
const cached = patJobCache.get(cacheKey);
if (cached && cached.expiresAt - Date.now() > PAT_REFRESH_BUFFER_MS) return cached;
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal);
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal);
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal, region);
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal, region);
const resolved = { accessToken: jobToken, userId, expiresAt };
patJobCache.set(pat, resolved);
patJobCache.set(cacheKey, resolved);
return resolved;
}
@@ -142,11 +147,14 @@ async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
* Resolve connection credentials to COSY-signable form:
* - PAT (pt-...) connections → exchanged to a job token (jt-...) + userId
* - everything else → passed through unchanged
*
* Region defaults to the one implied by credentials.provider (qoder-cn → cn).
*/
export async function resolveQoderCredentials(credentials, proxyOptions = null, signal = null) {
export async function resolveQoderCredentials(credentials, proxyOptions = null, signal = null, region) {
const raw = credentials?.apiKey || credentials?.accessToken;
if (isQoderPat(raw)) {
const resolved = await resolvePatCredential(raw, proxyOptions, signal);
const effRegion = region || qoderRegionOf(credentials?.provider);
const resolved = await resolvePatCredential(raw, proxyOptions, signal, effRegion);
return {
...credentials,
accessToken: resolved.accessToken,
@@ -163,13 +171,14 @@ export async function resolveQoderCredentials(credentials, proxyOptions = null,
}
/**
* Stable cache key per credential (so different login sessions for the same
* account share an entry).
* Stable cache key per credential+region (so different login sessions for the
* same account share an entry, and the same PAT on both sites stays apart).
*/
function cacheKey(credentials) {
const psd = credentials?.providerSpecificData || {};
const seed = psd.userId || credentials?.refreshToken || credentials?.accessToken || "anonymous";
return createHash("sha256").update(`qoder:${seed}`).digest("hex");
const region = qoderRegionOf(credentials?.provider);
return createHash("sha256").update(`qoder:${region}:${seed}`).digest("hex");
}
/**
@@ -192,15 +201,13 @@ function cosyCredsFromConnection(credentials) {
* rawConfigs: Map<modelKey, modelConfigObject> }
* or `null` on any error.
*/
async function fetchQoderCatalogRaw(credentials, signal, proxyOptions = null) {
async function fetchQoderCatalogRaw(credentials, signal, proxyOptions = null, region = "intl") {
const creds = cosyCredsFromConnection(credentials);
if (!creds.userId || !creds.authToken) return null;
// Job-token traffic is rejected by api3 ("Login expired" 403) — the
// official qodercli serves it from api2 instead.
const modelListUrl = String(creds.authToken).startsWith("jt-")
? `${QODER_CHAT_BASE_ALT}/algo/api/v2/model/list`
: QODER_MODEL_LIST_URL;
// Intl job-token traffic is rejected by api3 ("Login expired" 403) — the
// official qodercli serves it from api2 instead; CN uses the single gateway.
const modelListUrl = `${qoderInferenceBase(credentials, region)}/algo/api/v2/model/list`;
const headers = {
Accept: "application/json",
@@ -293,14 +300,20 @@ export async function getQoderModelConfig(credentials, modelKey, options = {}) {
* one upstream request per credential.
*/
export async function resolveQoderModels(credentials, options = {}) {
const region = options.region || qoderRegionOf(credentials?.provider);
let resolved;
try {
resolved = await resolveQoderCredentials(credentials, options.proxyOptions, options.signal);
resolved = await resolveQoderCredentials(credentials, options.proxyOptions, options.signal, region);
} catch (error) {
options.log?.warn?.("QODER", `PAT exchange failed: ${error.message}`);
return null;
}
if (!resolved?.accessToken || !(resolved.providerSpecificData || {}).userId) return null;
// Stamp the provider so cacheKey/catalog derive the region even when the
// caller's credentials object didn't carry a provider id (e.g. /v1/models).
if (resolved && !resolved.provider) {
resolved.provider = region === "cn" ? "qoder-cn" : "qoder";
}
const key = cacheKey(resolved);
const now = Date.now();
@@ -319,7 +332,7 @@ export async function resolveQoderModels(credentials, options = {}) {
}
const fetchPromise = (async () => {
const fetched = await fetchQoderCatalogRaw(resolved, options.signal, options.proxyOptions);
const fetched = await fetchQoderCatalogRaw(resolved, options.signal, options.proxyOptions, region);
if (!fetched) return null;
const entry = {
expiresAt: Date.now() + CACHE_TTL_MS,

View File

@@ -17,6 +17,7 @@ import { getKimiUsage } from "./usage/kimi.js";
import { getDeepseekUsage } from "./usage/deepseek.js";
import { getCommandCodeUsage } from "./usage/commandcode.js";
import { getOpenCodeGoUsage } from "./usage/opencode-go.js";
import { getOpenCodeZenUsage } from "./usage/opencode-zen.js";
import { getGroqUsage } from "./usage/groq.js";
import { getZedUsage } from "./usage/zed.js";
import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js";
@@ -43,12 +44,8 @@ const USAGE_HANDLERS = {
claude: (c) => getClaudeUsage(c.accessToken, c.proxyOptions, { force: c.force }),
codex: (c) => getCodexUsage(c.accessToken, c.proxyOptions),
kiro: (c) => getKiroUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
qoder: async (c) => {
// PAT (pt-...) connections must be exchanged to a job token before the
// quota endpoint accepts them.
const resolved = await resolveQoderCredentials(c, c.proxyOptions).catch(() => null);
return getQoderUsage(resolved?.accessToken || c.accessToken, c.proxyOptions);
},
qoder: (c) => getQoderUsageFor(c),
"qoder-cn": (c) => getQoderUsageFor(c),
iflow: (c) => getIflowUsage(c.accessToken),
ollama: (c) => getOllamaUsage(c.apiKey, c.providerSpecificData, c.proxyOptions),
glm: (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
@@ -62,6 +59,7 @@ const USAGE_HANDLERS = {
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData),
"opencode-go": (c) => getOpenCodeGoUsage(c.apiKey, c.proxyOptions),
"opencode-zen": (c) => getOpenCodeZenUsage(c.apiKey, c.proxyOptions),
deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions),
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
@@ -70,6 +68,14 @@ const USAGE_HANDLERS = {
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
};
// Qoder intl/CN share one usage path: PATs must be exchanged to a job token
// before the quota endpoint accepts them, and the quota URL comes from the
// provider's own registry usage block (region-correct via c.provider).
async function getQoderUsageFor(c) {
const resolved = await resolveQoderCredentials(c, c.proxyOptions).catch(() => null);
return getQoderUsage(resolved?.accessToken || c.accessToken, c.proxyOptions, c.provider || "qoder");
}
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {
const { provider, accessToken, apiKey, providerSpecificData, projectId } = connection;
const providerDataWithProjectId = {

View File

@@ -25,10 +25,18 @@ export function _clearWeeklyCache() {
weeklyCache.clear();
}
// — Group-name to stable key mapping ——————————————————————
const GROUP_MATCHERS = [
{ pattern: /gemini/i, key: "gemini_weekly", displayName: "Gemini (Weekly)" },
{ pattern: /claude|gpt/i, key: "claude_gpt_weekly", displayName: "Claude & GPT (Weekly)" },
// — Group-name and window to stable key mapping ——————————————————————
const GROUP_CONFIGS = [
{
pattern: /gemini/i,
weekly: { key: "gemini_weekly", displayName: "Gemini (Weekly)" },
session: { key: "gemini_session", displayName: "Gemini (5h)" },
},
{
pattern: /claude|gpt/i,
weekly: { key: "claude_gpt_weekly", displayName: "Claude & GPT (Weekly)" },
session: { key: "claude_gpt_session", displayName: "Claude & GPT (5h)" },
},
];
/**
@@ -60,32 +68,40 @@ export function parseWeeklyQuotaSummary(data) {
for (const bucket of buckets) {
if (!bucket || typeof bucket !== "object") continue;
// Identify weekly buckets by checking bucketId + displayName for "weekly"
const windowType = String(bucket.window || "").toLowerCase();
const bucketText = `${bucket.bucketId || ""} ${bucket.displayName || ""}`.toLowerCase();
if (!bucketText.includes("weekly")) continue;
const isWeekly = windowType === "weekly" || bucketText.includes("weekly");
const isSession = windowType === "5h" || bucketText.includes("five hour") || bucketText.includes("5h") || bucketText.includes("daily") || windowType === "daily";
// Skip disabled buckets
if (bucket.disabled === true) continue;
if (!isWeekly && !isSession) continue;
const remainingFraction = Number(bucket.remainingFraction);
// If a session (5h) bucket is marked disabled by upstream (because weekly was hit),
// keep it so the UI shows the 5h row, but with remainingFraction: 0.
// Disabled weekly buckets are truly disabled and skipped.
if (bucket.disabled === true && isWeekly) continue;
const remainingFraction = bucket.disabled === true ? 0 : Number(bucket.remainingFraction);
if (!Number.isFinite(remainingFraction)) continue;
// Match group to a known family
for (const matcher of GROUP_MATCHERS) {
if (matcher.pattern.test(displayName)) {
for (const config of GROUP_CONFIGS) {
if (config.pattern.test(displayName)) {
const target = isWeekly ? config.weekly : config.session;
if (result[target.key]) break; // first matching bucket per type wins
const total = 1000;
const remaining = Math.round(total * remainingFraction);
const used = Math.max(0, total - remaining);
result[matcher.key] = {
result[target.key] = {
used,
total,
resetAt: parseResetTime(bucket.resetTime),
remainingPercentage: remainingFraction * 100,
unlimited: false,
displayName: matcher.displayName,
displayName: target.displayName,
};
break; // first matching bucket per family wins
break;
}
}
}

View File

@@ -228,39 +228,37 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
proxyOptions
);
// Reconcile weekly quota against model family status:
// Reconcile short-window session quota if models are exhausted:
// If every model in a family is locked/exhausted (remainingPercentage === 0)
// until a future reset time, the weekly limit cannot be 100% available.
// On Google's Free Starter tier, retrieveUserQuotaSummary buggily reports
// remainingFraction: 1 even after the starter quota is depleted and all models 429.
// until a future reset time, update the 5h session row (not the weekly row).
const entries = Object.entries(quotas);
const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image"));
const claudeModels = entries.filter(([k]) => k.startsWith("claude-"));
if (weeklyQuotas.gemini_weekly && geminiModels.length > 0) {
if (weeklyQuotas.gemini_session && geminiModels.length > 0) {
const allGeminiExhausted = geminiModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0);
if (allGeminiExhausted && weeklyQuotas.gemini_weekly.remainingPercentage > 0) {
if (allGeminiExhausted && weeklyQuotas.gemini_session.remainingPercentage > 0) {
const maxResetAt = geminiModels.reduce((max, [, q]) =>
!max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null
);
weeklyQuotas.gemini_weekly.used = weeklyQuotas.gemini_weekly.total;
weeklyQuotas.gemini_weekly.remainingPercentage = 0;
weeklyQuotas.gemini_session.used = weeklyQuotas.gemini_session.total;
weeklyQuotas.gemini_session.remainingPercentage = 0;
if (maxResetAt) {
weeklyQuotas.gemini_weekly.resetAt = maxResetAt;
weeklyQuotas.gemini_session.resetAt = maxResetAt;
}
}
}
if (weeklyQuotas.claude_gpt_weekly && claudeModels.length > 0) {
if (weeklyQuotas.claude_gpt_session && claudeModels.length > 0) {
const allClaudeExhausted = claudeModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0);
if (allClaudeExhausted && weeklyQuotas.claude_gpt_weekly.remainingPercentage > 0) {
if (allClaudeExhausted && weeklyQuotas.claude_gpt_session.remainingPercentage > 0) {
const maxResetAt = claudeModels.reduce((max, [, q]) =>
!max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null
);
weeklyQuotas.claude_gpt_weekly.used = weeklyQuotas.claude_gpt_weekly.total;
weeklyQuotas.claude_gpt_weekly.remainingPercentage = 0;
weeklyQuotas.claude_gpt_session.used = weeklyQuotas.claude_gpt_session.total;
weeklyQuotas.claude_gpt_session.remainingPercentage = 0;
if (maxResetAt) {
weeklyQuotas.claude_gpt_weekly.resetAt = maxResetAt;
weeklyQuotas.claude_gpt_session.resetAt = maxResetAt;
}
}
}

View File

@@ -24,11 +24,43 @@ export async function getIflowUsage(accessToken) {
}
}
const OLLAMA_LIMIT_WINDOWS = {
session: "Session (5h)",
weekly: "Weekly (7d)",
monthly: "Monthly",
};
function addUtcMonths(date, months) {
const total = date.getUTCMonth() + months;
const year = date.getUTCFullYear() + Math.floor(total / 12);
const month = ((total % 12) + 12) % 12;
const lastDay = new Date(Date.UTC(year, month + 1, 0)).getUTCDate();
return new Date(Date.UTC(
year, month, Math.min(date.getUTCDate(), lastDay),
date.getUTCHours(), date.getUTCMinutes(), date.getUTCSeconds(),
));
}
// Free plan: "usage resets monthly from the date you signed up" (ollama.com/pricing).
function nextMonthlyResetFromSignup(createdAt, now = new Date()) {
const anchor = new Date(createdAt);
if (Number.isNaN(anchor.getTime())) return null;
const elapsedMonths = (now.getUTCFullYear() - anchor.getUTCFullYear()) * 12
+ (now.getUTCMonth() - anchor.getUTCMonth());
for (let i = Math.max(0, elapsedMonths); i <= elapsedMonths + 1; i++) {
const candidate = addUtcMonths(anchor, i);
if (candidate > now) return candidate.toISOString();
}
return null;
}
/**
* Ollama Cloud Usage
* GET https://ollama.com/api/usage — session (5h) + weekly (7d) `usage` is a 0..1
* ratio (1.0 = limit reached, e.g. weekly 100% used). No reset timestamp exposed.
* POST https://ollama.com/api/me — plan label (fail-open).
* GET https://ollama.com/api/usage — `limits.<window>.usage` is a 0..1 ratio
* (1.0 = limit reached). Paid plans report session (5h) + weekly (7d); the
* free plan reports a single monthly window. No reset timestamp exposed;
* the free monthly reset is derived from the account's signup date.
* POST https://ollama.com/api/me — plan label + CreatedAt (fail-open).
* Auth: Authorization: Bearer <apiKey>
*/
export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions = null) {
@@ -84,14 +116,20 @@ export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions
return { used: usedPct, total: 100, remainingPercentage: 100 - usedPct, resetAt, unlimited: false };
}
const sessionRaw = limits.session?.usage;
const weeklyRaw = limits.weekly?.usage;
const sessionNum = Number(sessionRaw);
const weeklyNum = Number(weeklyRaw);
const hasSession = sessionRaw !== undefined && sessionRaw !== null && !Number.isNaN(sessionNum);
const hasWeekly = weeklyRaw !== undefined && weeklyRaw !== null && !Number.isNaN(weeklyNum);
const monthlyResetAt = planRaw.toLowerCase() === "free" && me?.CreatedAt
? nextMonthlyResetFromSignup(me.CreatedAt)
: null;
if (!hasSession && !hasWeekly) {
const quotas = {};
for (const [key, label] of Object.entries(OLLAMA_LIMIT_WINDOWS)) {
const raw = limits[key]?.usage;
if (raw === undefined || raw === null) continue;
const ratio = Number(raw);
if (Number.isNaN(ratio)) continue;
quotas[label] = ratioQuota(ratio, key === "monthly" ? monthlyResetAt : null);
}
if (Object.keys(quotas).length === 0) {
return {
plan,
message: "Ollama Cloud connected. No usage limits reported.",
@@ -99,10 +137,6 @@ export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions
};
}
const quotas = {};
if (hasSession) quotas["Session (5h)"] = ratioQuota(sessionNum);
if (hasWeekly) quotas["Weekly (7d)"] = ratioQuota(weeklyNum);
return { plan, quotas };
} catch (error) {
return { message: `Ollama Cloud error: ${error.message}` };
@@ -193,13 +227,13 @@ export async function getVercelAiGatewayUsage(apiKey, proxyOptions = null) {
}
}
export async function getQoderUsage(accessToken, proxyOptions = null) {
export async function getQoderUsage(accessToken, proxyOptions = null, providerId = "qoder") {
if (!accessToken) {
return { message: "Qoder usage unavailable: no access token" };
}
try {
const response = await proxyAwareFetch(
U("qoder").url,
U(providerId).url,
{
method: "GET",
headers: {

View File

@@ -0,0 +1,107 @@
/**
* OpenCode Zen usage — GET https://opencode.ai/zen/v1/usage
* Auth: Bearer <apiKey>
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { parseResetTime, toFiniteNumber, U } from "./shared.js";
const USAGE_URL = U("opencode-zen").url;
const QUOTA_NAMES = {
rolling: "Rolling",
weekly: "Weekly",
monthly: "Monthly",
};
function parsePercent(value) {
if (typeof value === "number" && Number.isFinite(value)) return value;
if (typeof value === "string" && value.trim()) {
const parsed = Number(value);
if (Number.isFinite(parsed)) return parsed;
}
return null;
}
export async function getOpenCodeZenUsage(apiKey = null, proxyOptions = null) {
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
return {
message: "OpenCode Zen API key not available. Add a key to view usage.",
};
}
try {
const response = await proxyAwareFetch(
USAGE_URL,
{
method: "GET",
headers: {
Authorization: `Bearer ${apiKey.trim()}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (response.status === 401) {
return {
plan: "OpenCode Zen",
message: "OpenCode Zen authentication failed. Check the API key.",
};
}
if (response.status === 403) {
const error = await response.json().catch(() => null);
const subscriptionRequired = error?.error?.type === "EntitlementError";
return {
plan: "OpenCode Zen",
message: subscriptionRequired
? "OpenCode Zen billing required for this API key."
: "OpenCode Zen access forbidden for this API key.",
};
}
if (!response.ok) {
return {
plan: "OpenCode Zen",
message: `OpenCode Zen usage API error (${response.status}).`,
};
}
const data = await response.json().catch(() => null);
if (!data?.usage || typeof data.usage !== "object") {
return {
plan: "OpenCode Zen",
message: "OpenCode Zen usage response did not contain quota data.",
};
}
const quotas = {};
for (const [period, name] of Object.entries(QUOTA_NAMES)) {
const quota = data.usage[period];
if (!quota || typeof quota !== "object") continue;
const percent = parsePercent(quota.percent);
if (percent === null) continue;
const used = Math.max(0, Math.min(100, toFiniteNumber(percent, 0)));
quotas[name] = {
used,
total: 100,
remaining: 100 - used,
remainingPercentage: 100 - used,
resetAt: parseResetTime(quota.resetsAt),
unlimited: false,
};
}
if (Object.keys(quotas).length === 0) {
return {
plan: "OpenCode Zen",
message: "OpenCode Zen usage response did not contain valid quota data.",
};
}
return { plan: "OpenCode Zen", quotas };
} catch (error) {
return { message: `OpenCode Zen error: ${error.message}` };
}
}

View File

@@ -8,20 +8,42 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
* Xiaomi MiMo account-session helpers (used for weekly quota).
*
* The weekly quota endpoint lives on the account service domain and is authorized
* by an account session cookie, NOT the sk- API key. Acquiring that cookie mirrors
* MiMo Desktop: a passToken (persisted in Desktop's cookie store) is exchanged via
* the passportapi SSO, then authorized for the `mimopc` service, and finally stamped
* by the mimo-server /api/sts callback into a `serviceToken` cookie.
* by an account session cookie, NOT the sk- API key. Acquiring that cookie is a
* 1:1 port of MiMo Desktop's ServiceTokenManager (app.asar) — the GOLD STANDARD:
*
* Flow (verified against MiMo Desktop traffic):
* 1. GET {api}/api/user/xiaomi/me -> 302 to account SSO (sid=mimopc)
* 2. GET account /pass/serviceLogin?sid=passportapi&_json=true -> nonce/ssecurity
* 3. GET {location}&clientSign=... -> account-level serviceToken
* 4. GET account /pass/serviceLogin?sid=mimopc&callback=<sts>&_json=true
* 5. GET {api}/api/sts?...&ticket... -> Set-Cookie: serviceToken (mimopc scope)
* getServiceToken(sid) / refreshServiceToken(sid):
* PHASE 1: GET https://account.xiaomi.com/pass/serviceLogin
* ?_locale=zh_CN&_snsNone=true&sid=<clusterSid>&_json=true
* Cookie: {userId, passToken, cUserId}
* -> {code, location, ssecurity, nonce, bSecondValidation, notificationUrl}
* -> code !== 0 is an error (never silent)
* PHASE 2: GET {location}&clientSign=sha1(nonce & ssecurity), follow the
* redirect chain absorbing Set-Cookie -> serviceToken
*
* sid is per-cluster (SID_BY_REGION): CN = mimopc, SGP = mimosgp.
*/
const API_BASE = "https://mimo-server-cn.xiaomimimo.com";
// Account-service cluster hosts. MiMo Desktop declares five regions
// (rn = {CN, SGP, RU, IN, EU}); the EU cluster is deployed in Amsterdam.
// Host + sid naming is unified: mimo-server-<code> / sid = mimo<code>
// (ams is the only non-country code). Verified live via /api/user/xiaomi/me.
const API_BASE_BY_REGION = {
cn: "https://mimo-server-cn.xiaomimimo.com",
sgp: "https://mimo-server-sgp.xiaomimimo.com",
ams: "https://mimo-server-ams.xiaomimimo.com",
ru: "https://mimo-server-ru.xiaomimimo.com",
in: "https://mimo-server-in.xiaomimimo.com",
};
const DEFAULT_API_BASE = API_BASE_BY_REGION.sgp;
// Cluster service sid — 1:1 with the host code: mimo<code>.
// Unknown/absent region falls back to SGP (the international/open cluster).
const SID_BY_REGION = { cn: "mimopc", sgp: "mimosgp", ams: "mimoams", ru: "mimoru", in: "mimoin" };
function sidForRegion(region) {
const r = String(region || "").toLowerCase();
return SID_BY_REGION[r] || SID_BY_REGION.sgp;
}
const API_BASE = DEFAULT_API_BASE;
const ACCOUNT_HOST = "account.xiaomi.com";
const API_UA =
"miNative PC/Normal Windows_NT/10.0.19045 SDKV/1.0.0 DEVT/PC DEVS/Windows APP/miaccount_desktop APPV/0.1.0";
@@ -112,65 +134,106 @@ function cookieHeader(jar) {
.join("; ");
}
/**
* Resolve the account-service base URL for a connection.
* @param {object|null} providerSpecificData - may carry `region` ("cn"|"sgp"|"ams"|"ru"|"in")
*/
export function resolveMimoServerBase(providerSpecificData = null) {
const region = String(providerSpecificData?.region || "").toLowerCase();
return API_BASE_BY_REGION[region] || DEFAULT_API_BASE;
}
/**
* Exchange a passToken for a mimo-server service session cookie.
* Primary path mirrors the Desktop ServiceTokenManager (app.asar):
* PHASE 1: GET /pass/serviceLogin?_locale=zh_CN&_snsNone=true&sid=<clusterSid>&_json=true
* Cookie {userId,passToken,cUserId} -> {code,location,ssecurity,nonce}
* PHASE 2: GET {location}&clientSign=sha1(nonce&ssecurity), follow the chain
* (manual, absorbing Set-Cookie) -> serviceToken
* sid is per-cluster (SID_BY_REGION): cn=mimopc, sgp=mimosgp, ams=mimoams, ru=mimoru, in=mimoin.
* @returns {Promise<string|null>} Cookie header value, or null on failure.
*/
async function acquireServiceCookie(passJar, proxyOptions) {
async function acquireServiceCookie(passJar, proxyOptions, apiBase = DEFAULT_API_BASE, region = "sgp") {
const r = String(region || "").toLowerCase();
// Hard constraint: CN is ALWAYS direct (ignores proxy even if set)
const effectiveProxy = r === "cn" ? null : proxyOptions;
const sid = sidForRegion(r);
const viaDesktop = await acquireViaDesktopPhases(passJar, effectiveProxy, apiBase, sid);
if (viaDesktop) console.log(`[mimoAccount] desktop 2-phase OK (sid=${sid})`);
return viaDesktop;
}
async function acquireViaDesktopPhases(passJar, proxyOptions, apiBase, sid) {
const failLog = (reason) => console.log(`[mimoAccount] desktopPhase fail: ${reason}`);
const jar = { ...passJar };
const ck = () => cookieHeader(jar);
// 1. Unauthenticated API call -> 302 carrying the sts callback (sid=mimopc)
const r1 = await proxyAwareFetch(
`${API_BASE}/api/user/xiaomi/me`,
{ redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } },
// PHASE 1 — single serviceLogin call with the TARGET sid (no passportapi
// prelude; ssecurity/nonce come straight from this response).
// Desktop only sends: userId, passToken, cUserId (no extra cookies)
const p1Jar = {};
if (jar.userId) p1Jar.userId = jar.userId;
if (jar.passToken) p1Jar.passToken = jar.passToken;
if (jar.cUserId) p1Jar.cUserId = jar.cUserId;
const p1Url = `https://${ACCOUNT_HOST}/pass/serviceLogin?_locale=zh_CN&_snsNone=true&sid=${encodeURIComponent(sid)}&_json=true`;
const p1 = await proxyAwareFetch(
p1Url,
{ headers: { Cookie: cookieHeader(p1Jar), "User-Agent": SSO_UA, Accept: "application/json" } },
proxyOptions,
);
const redirect = r1.headers.get("location");
if (!redirect) return null;
const stsCallback = new URL(redirect).searchParams.get("callback");
if (!stsCallback) return null;
const raw = await p1.text();
const clean = raw.replace(/^&&&START&&&/, "");
// Nonce > 2^53 loses precision in JSON.parse — extract raw literal for signing
const rawNonce = clean.match(/"nonce"\s*:\s*(\d+)/)?.[1];
let j = null;
try { j = JSON.parse(clean); } catch { /* handled below */ }
if (rawNonce && j) j.nonce = rawNonce;
// 2. passportapi SSO phase 1 -> nonce + ssecurity
const sso1 = await proxyAwareFetch(
`https://${ACCOUNT_HOST}/pass/serviceLogin?sid=passportapi&_json=true`,
{ headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } },
proxyOptions,
);
const j1 = JSON.parse((await sso1.text()).replace(/^&&&START&&&/, ""));
const nonce = j1.nonce || (j1.location ? new URL(j1.location).searchParams.get("nonce") : null);
if (!nonce || !j1.location) return null;
if (!j || typeof j.code !== "number" || j.code !== 0 || !j.location || !j.nonce || !j.ssecurity) {
failLog(
`phase1 sid=${sid} http=${p1.status} code=${j?.code ?? "?"} hasLoc=${!!j?.location}`
+ ` secondValidation=${j?.bSecondValidation ?? "?"} notificationUrl=${j?.notificationUrl ? "present" : "no"}`
+ ` body=${JSON.stringify(raw.slice(0, 200))}`,
);
return null;
}
absorbSetCookie(jar, p1);
// 3. passportapi SSO phase 2 -> account-level serviceToken
const sso2 = await proxyAwareFetch(
`${j1.location}&clientSign=${signatureClientSign(nonce, j1.ssecurity)}`,
{ redirect: "manual", headers: { Cookie: ck(), "User-Agent": SSO_UA } },
proxyOptions,
);
absorbSetCookie(jar, sso2);
// PHASE 2 — clientSign the redirect, follow the redirect chain server-side.
// ⚠️ CRITICAL DESKTOP SPEC (app.asar / SSO_curl.cpp line 728: cookies.clear()):
// Phase 2 MUST NOT send ANY Cookie header! The server returns 200 OK with Set-Cookie: serviceToken!
const sep = j.location.includes("?") ? "&" : "?";
let current = `${j.location}${sep}clientSign=${signatureClientSign(rawNonce || j.nonce, j.ssecurity)}`;
// 4. mimopc SSO -> sts callback carrying a ticket
const sso3 = await proxyAwareFetch(
`https://${ACCOUNT_HOST}/pass/serviceLogin?sid=mimopc&callback=${encodeURIComponent(stsCallback)}&_json=true`,
{ headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } },
proxyOptions,
);
const j3 = JSON.parse((await sso3.text()).replace(/^&&&START&&&/, ""));
absorbSetCookie(jar, sso3);
if (!j3?.location || !/\/api\/sts/.test(j3.location)) return null;
for (let hop = 0; hop < 8; hop++) {
const res = await proxyAwareFetch(
current,
{ redirect: "manual", headers: { "User-Agent": SSO_UA } },
proxyOptions,
);
absorbSetCookie(jar, res);
const loc = res.headers.get("location");
if (res.status >= 300 && res.status < 400 && loc) {
current = new URL(loc, current).toString();
continue;
}
break;
}
// 5. sts callback -> Set-Cookie: serviceToken (mimopc scope)
const sts = await proxyAwareFetch(
j3.location,
{ redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } },
proxyOptions,
);
absorbSetCookie(jar, sts);
const sidKey = `${sid}_serviceToken`;
if (!jar.serviceToken && jar[sidKey]) {
jar.serviceToken = jar[sidKey];
}
const needed = ["serviceToken", "mimopc_ph", "mimopc_slh", "userId"];
if (!jar.serviceToken) return null;
if (!jar.serviceToken) {
failLog(`phase2 no serviceToken sid=${sid} jar=[${Object.keys(jar).join(",")}]`);
return null;
}
const out = {};
for (const k of needed) if (jar[k]) out[k] = jar[k];
for (const [k, v] of Object.entries(jar)) {
if (!v) continue;
if (k === "serviceToken" || k === "userId" || /_(ph|slh)$/.test(k)) out[k] = v;
}
return cookieHeader(out);
}
@@ -179,13 +242,15 @@ async function acquireServiceCookie(passJar, proxyOptions) {
* @param {object|null} providerSpecificData - may carry `mimoPassToken` override
*/
async function getServiceCookie(providerSpecificData, proxyOptions) {
const apiBase = resolveMimoServerBase(providerSpecificData);
const passJar = providerSpecificData?.mimoPassToken
? { passToken: providerSpecificData.mimoPassToken, userId: providerSpecificData.mimoUserId, cUserId: providerSpecificData.mimoCUserId }
: await readDesktopAccountCookies();
if (!passJar) return { cookie: null, reason: "no-pass-token" };
// One cached session per passToken — accounts/connections rotate independently.
const key = crypto.createHash("sha256").update(passJar.passToken).digest("hex");
// One cached session per passToken+cluster — accounts/connections rotate
// independently, and the same passToken maps to different sessions per region.
const key = crypto.createHash("sha256").update(`${apiBase}|${passJar.passToken}`).digest("hex");
const cached = _cache.get(key);
if (cached && Date.now() - cached.at < COOKIE_TTL_MS) {
@@ -202,8 +267,9 @@ async function getServiceCookie(providerSpecificData, proxyOptions) {
const promise = (async () => {
try {
return await acquireServiceCookie(passJar, proxyOptions);
} catch {
return await acquireServiceCookie(passJar, proxyOptions, apiBase, providerSpecificData?.region);
} catch (e) {
console.log(`[mimoAccount] acquire threw: ${e?.message || e} | ${String(e?.stack || "").split("\n").slice(1, 4).join(" <- ")}`);
return null; // network/parse failure — callers degrade, never throw
} finally {
_inflight.delete(key);
@@ -234,7 +300,8 @@ export async function getMimoAccountCookie(providerSpecificData = null, proxyOpt
try {
const { cookie } = await getServiceCookie(providerSpecificData, proxyOptions);
return cookie;
} catch {
} catch (e) {
console.log(`[mimoAccount] getMimoAccountCookie threw: ${e?.message || e} | ${String(e?.stack || "").split("\n").slice(1, 4).join(" <- ")}`);
return null;
}
}
@@ -250,7 +317,7 @@ export async function getMimoAccountUsage(providerSpecificData = null, proxyOpti
}
try {
const res = await proxyAwareFetch(
`${API_BASE}/api/user/usage`,
`${resolveMimoServerBase(providerSpecificData)}/api/user/usage`,
{ headers: { "User-Agent": API_UA, Cookie: cookie, Accept: "application/json" }, signal: AbortSignal.timeout(10000) },
proxyOptions,
);

View File

@@ -1,38 +1,106 @@
/**
* Qoder API constants ported from CLIProxyAPIPlus qoder-provider branch.
*
* Endpoint set:
* openapi.qoder.sh - device flow + userinfo + quota usage
* center.qoder.sh - token refresh (best-effort, currently 403 for device tokens)
* api3.qoder.sh - inference (chat) + model list, requires COSY signing
* qoder.com/device - browser landing page for device authorization
* Qoder runs two regional sites with parallel endpoint shapes:
* intl (qoder) CN (qoder-cn)
* openapi.qoder.sh openapi.qoder.com.cn - device flow + userinfo + quota usage
* center.qoder.sh gateway.qoder.com.cn - token refresh (best-effort, 403 for device tokens)
* api3.qoder.sh gateway.qoder.com.cn - inference (chat) + model list, requires COSY signing
* qoder.com/device qoder.com.cn/device - browser landing page for device authorization
*
* All path suffixes are identical between regions — only the hosts differ.
* Region-aware consumers call qoder*Url(region) / qoderInferenceBase(creds, region)
* and derive the region from the provider id via qoderRegionOf(). The named
* QODER_* constants below keep the intl defaults for backward compatibility.
*/
export const QODER_OPENAPI_BASE = "https://openapi.qoder.sh";
export const QODER_CENTER_BASE = "https://center.qoder.sh";
export const QODER_CHAT_BASE = "https://api3.qoder.sh";
export const QODER_REGION_INTL = "intl";
export const QODER_REGION_CN = "cn";
// Per-region base URLs. CN serves job tokens (jt-...) from the same gateway
// host — there is no api2-style split like intl's api2.qoder.sh.
const QODER_REGION_BASES = {
[QODER_REGION_INTL]: {
chat: "https://api3.qoder.sh",
chatAlt: "https://api2.qoder.sh",
openApi: "https://openapi.qoder.sh",
center: "https://center.qoder.sh",
login: "https://qoder.com/device/selectAccounts",
website: "https://qoder.com",
},
[QODER_REGION_CN]: {
chat: "https://gateway.qoder.com.cn",
chatAlt: "https://gateway.qoder.com.cn",
openApi: "https://openapi.qoder.com.cn",
center: "https://gateway.qoder.com.cn",
login: "https://qoder.com.cn/device/selectAccounts",
website: "https://qoder.com.cn",
},
};
/** Base URL set for a region; unknown regions fall back to intl. */
export function qoderRegionBases(region) {
return QODER_REGION_BASES[region] || QODER_REGION_BASES[QODER_REGION_INTL];
}
/** Region for a provider id — "cn" for qoder-cn, "intl" otherwise. */
export function qoderRegionOf(providerId) {
return providerId === "qoder-cn" ? QODER_REGION_CN : QODER_REGION_INTL;
}
export const QODER_OPENAPI_BASE = QODER_REGION_BASES[QODER_REGION_INTL].openApi;
export const QODER_CENTER_BASE = QODER_REGION_BASES[QODER_REGION_INTL].center;
export const QODER_CHAT_BASE = QODER_REGION_BASES[QODER_REGION_INTL].chat;
// Job-token (jt-...) traffic is rejected by api3 with "Login expired" (403);
// the official qodercli serves it from api2 instead.
export const QODER_CHAT_BASE_ALT = "https://api2.qoder.sh";
// the official qodercli serves it from api2 instead (intl only).
export const QODER_CHAT_BASE_ALT = QODER_REGION_BASES[QODER_REGION_INTL].chatAlt;
export const QODER_LOGIN_URL = "https://qoder.com/device/selectAccounts";
export const QODER_LOGIN_URL = QODER_REGION_BASES[QODER_REGION_INTL].login;
// Device flow endpoints
export const QODER_DEVICE_TOKEN_URL = `${QODER_OPENAPI_BASE}/api/v1/deviceToken/poll`;
export const QODER_USERINFO_URL = `${QODER_OPENAPI_BASE}/api/v1/userinfo`;
export const QODER_QUOTA_USAGE_URL = `${QODER_OPENAPI_BASE}/api/v2/quota/usage`;
export const QODER_REFRESH_TOKEN_URL = `${QODER_CENTER_BASE}/algo/api/v3/user/refresh_token`;
// Device flow endpoints (region-aware variants; these are the intl defaults)
export function qoderOpenApiBase(region) {
return qoderRegionBases(region).openApi;
}
export function qoderDeviceTokenUrl(region) {
return `${qoderOpenApiBase(region)}/api/v1/deviceToken/poll`;
}
export function qoderUserInfoUrl(region) {
return `${qoderOpenApiBase(region)}/api/v1/userinfo`;
}
export function qoderQuotaUsageUrl(region) {
return `${qoderOpenApiBase(region)}/api/v2/quota/usage`;
}
export function qoderRefreshTokenUrl(region) {
return `${qoderRegionBases(region).center}/algo/api/v3/user/refresh_token`;
}
export function qoderLoginUrl(region) {
return qoderRegionBases(region).login;
}
export function qoderWebsiteUrl(region) {
return qoderRegionBases(region).website;
}
export const QODER_DEVICE_TOKEN_URL = qoderDeviceTokenUrl(QODER_REGION_INTL);
export const QODER_USERINFO_URL = qoderUserInfoUrl(QODER_REGION_INTL);
export const QODER_QUOTA_USAGE_URL = qoderQuotaUsageUrl(QODER_REGION_INTL);
export const QODER_REFRESH_TOKEN_URL = qoderRefreshTokenUrl(QODER_REGION_INTL);
// PAT (Personal Access Token, pt-...) → short-lived job token (jt-...) exchange.
// PATs cannot sign COSY requests directly — they must be exchanged first.
// This endpoint is NOT COSY-signed (plain JSON POST).
export const QODER_JOB_TOKEN_EXCHANGE_URL = `${QODER_OPENAPI_BASE}/api/v1/jobToken/exchange`;
export function qoderJobTokenExchangeUrl(region) {
return `${qoderOpenApiBase(region)}/api/v1/jobToken/exchange`;
}
export const QODER_JOB_TOKEN_EXCHANGE_URL = qoderJobTokenExchangeUrl(QODER_REGION_INTL);
// Inference endpoints (under /algo on api3.qoder.sh, all COSY-signed)
// Inference endpoints (under /algo on the chat host, all COSY-signed)
export const QODER_CHAT_SIG_PATH = "/api/v2/service/pro/sse/agent_chat_generation";
export const QODER_CHAT_URL = `${QODER_CHAT_BASE}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common`;
export const QODER_CHAT_URL_ENCODED = `${QODER_CHAT_URL}&Encode=1`;
export const QODER_MODEL_LIST_URL = `${QODER_CHAT_BASE}/algo/api/v2/model/list`;
export function qoderModelListUrl(region) {
return `${qoderRegionBases(region).chat}/algo/api/v2/model/list`;
}
export const QODER_MODEL_LIST_URL = qoderModelListUrl(QODER_REGION_INTL);
// Official qodercli uploads images here (COSY-signed PUT multipart, field "file")
// instead of inlining base64 into agent_chat_generation.
export const QODER_IMAGE_UPLOAD_SIG_PATH = "/api/v2/image/upload";
@@ -55,7 +123,9 @@ export const QODER_CONTEXT_TIER_MODES = Object.freeze({ AUTO: "auto", MAX: "max"
* "Login expired" (403). Device tokens (dt-...) stay on api3. PATs (pt-...)
* are exchanged for jt- before this is consulted.
*/
export function qoderInferenceBase(credentials) {
export function qoderInferenceBase(credentials, region = QODER_REGION_INTL) {
// CN serves every token kind from the single gateway host.
if (region === QODER_REGION_CN) return QODER_REGION_BASES[QODER_REGION_CN].chat;
const raw = credentials?.apiKey || credentials?.accessToken;
if (
typeof raw === "string" &&

View File

@@ -11,6 +11,10 @@ export function toOpenAIFinish(reason, format) {
case CLAUDE_STOP.MAX_TOKENS: return OPENAI_FINISH.LENGTH;
case CLAUDE_STOP.TOOL_USE: return OPENAI_FINISH.TOOL_CALLS;
case CLAUDE_STOP.STOP_SEQUENCE: return OPENAI_FINISH.STOP;
// A refusal is a blocked turn, not a clean stop: with the default mapping an
// OpenAI client saw finish_reason "stop" and an empty message (9Router logged
// "succeeded", OUT 0) and could not tell it from a real answer.
case CLAUDE_STOP.REFUSAL: return OPENAI_FINISH.CONTENT_FILTER;
default: return OPENAI_FINISH.STOP;
}
case "commandcode":
@@ -55,6 +59,7 @@ export function fromOpenAIFinish(reason, format) {
case OPENAI_FINISH.STOP: return CLAUDE_STOP.END_TURN;
case OPENAI_FINISH.LENGTH: return CLAUDE_STOP.MAX_TOKENS;
case OPENAI_FINISH.TOOL_CALLS: return CLAUDE_STOP.TOOL_USE;
case OPENAI_FINISH.CONTENT_FILTER: return CLAUDE_STOP.REFUSAL;
default: return CLAUDE_STOP.END_TURN;
}
default:

View File

@@ -14,9 +14,6 @@ const STRIP_RULES = [
{ provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] },
// Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926)
{ provider: "cloudflare-ai", flattenContent: true },
// MiMo Desktop Preview models (account-service route): content must be plain string,
// rejects OpenAI content-part array. Cloud models keep their parts (mimo-v2-omni is multi-modal).
{ provider: "xiaomi-mimo", match: /preview/i, flattenContent: true },
{ provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true },
// VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's
// advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144),
@@ -24,6 +21,15 @@ const STRIP_RULES = [
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
// min() with the model ceiling still applies if a variant's own limit is lower.
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
// Strict OpenAI-compatible validators reject unknown assistant-message fields.
// Clients that talk to reasoning models (e.g. Hermes) echo the prior turn's
// reasoning back on every assistant message; Groq answers 400 and Mistral 422
// ("extra_forbidden") on it, which knocks these providers out of every
// multi-turn combo. Providers that *require* the field (DeepSeek, Kimi) are
// handled by reasoningContentInjector and are not listed here.
{ provider: "groq", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
{ provider: "mistral", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
{ provider: "cerebras", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
];
// Test a rule's match (regex or predicate) against the model id.
@@ -47,6 +53,15 @@ export function stripUnsupportedParams(provider, model, body) {
for (const key of rule.drop || []) {
if (body[key] !== undefined) delete body[key];
}
// Per-message field drop (assistant turns only — that is where clients replay reasoning).
if (Array.isArray(rule.dropMessageFields) && Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (!msg || msg.role !== "assistant") continue;
for (const key of rule.dropMessageFields) {
if (msg[key] !== undefined) delete msg[key];
}
}
}
// CF Workers AI oneOf root schema only accepts content as plain string (#1926)
if (rule.flattenContent && Array.isArray(body.messages)) {
for (const msg of body.messages) {

View File

@@ -34,6 +34,8 @@ export function effortToThinkingLevel(effort) {
// Numeric budget → nearest discrete level (reverse map via thresholds).
// Returns null when budget <= 0 (no reasoning).
// Thresholds are midpoints between LEVEL_TO_BUDGET values: max (128000) is
// reachable, with the xhigh/max boundary at the 32768/128000 midpoint (80384).
export function budgetToLevel(budget) {
const b = Number(budget);
if (!b || b <= 0) return null;
@@ -41,7 +43,8 @@ export function budgetToLevel(budget) {
if (b <= 4096) return "low";
if (b <= 16384) return "medium";
if (b <= 28672) return "high";
return "xhigh";
if (b <= 80384) return "xhigh";
return "max";
}
// Gemini thinkingBudget (numeric) → OpenAI reasoning_effort (antigravity reverse map).

View File

@@ -2,6 +2,7 @@ import { FORMATS } from "./formats.js";
import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js";
import { prepareClaudeRequest } from "./formats/claude.js";
import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js";
import { restoreToolNames } from "../utils/opencodeFingerprint.js";
import { filterToOpenAIFormat } from "./formats/openai.js";
import { normalizeThinkingConfig } from "../services/provider.js";
import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js";
@@ -166,7 +167,7 @@ export function translateResponse(targetFormat, sourceFormat, chunk, state) {
// even when no format conversion is needed, so streamed tool_use blocks must
// be decloaked here or the client sees an unknown ("_ide"-suffixed) tool.
if (sourceFormat === targetFormat) {
return [decloakStreamChunk(chunk, state?.toolNameMap)];
return [restoreToolNames(decloakStreamChunk(chunk, state?.toolNameMap), state?.toolNameMap)];
}
let results = [chunk];
@@ -179,7 +180,8 @@ export function translateResponse(targetFormat, sourceFormat, chunk, state) {
const directFn = responseRegistry.get(`${targetFormat}:${sourceFormat}`);
if (directFn) {
const converted = directFn(chunk, state);
return converted ? (Array.isArray(converted) ? converted : [converted]) : [];
const directResults = converted ? (Array.isArray(converted) ? converted : [converted]) : [];
return restoreToolNames(directResults, state?.toolNameMap);
}
// Step 1: target -> openai (if target is not openai)
@@ -210,6 +212,8 @@ export function translateResponse(targetFormat, sourceFormat, chunk, state) {
}
}
results = restoreToolNames(results, state?.toolNameMap);
// Attach OpenAI intermediate results for logging
if (openaiResults && sourceFormat !== FORMATS.OPENAI && targetFormat !== FORMATS.OPENAI) {
results._openaiIntermediate = openaiResults;

View File

@@ -279,10 +279,11 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra
}
};
// Antigravity specific fields
if (isAntigravity) {
envelope.requestType = "agent";
} else {
// Antigravity specific fields.
// NOTE: the official Antigravity client omits `requestType` entirely on the
// agent (chat) path. Sending `requestType: "agent"` triggers a detail-free
// 429 RESOURCE_EXHAUSTED even with quota available.
if (!isAntigravity) {
// Keep safetySettings for Gemini CLI
envelope.request.safetySettings = geminiCLI.safetySettings;
}
@@ -305,7 +306,8 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
model: model,
userAgent: "antigravity",
requestId: `agent-${generateUUID()}`,
requestType: "agent",
// NOTE: official Antigravity client omits `requestType` on the agent (chat)
// path — see the note in wrapInCloudCodeEnvelope() above.
request: {
sessionId: toNumericSessionId(credentials?._clientSessionId) || deriveSessionId(credentials?.email || credentials?.connectionId),
contents: [],

View File

@@ -149,6 +149,13 @@ export function claudeToOpenAIResponse(chunk, state) {
if (chunk.delta?.stop_reason) {
state.finishReason = convertStopReason(chunk.delta.stop_reason);
// A refusal produces no content blocks at all. Surface Anthropic's own
// explanation as the message text so the client shows *why* the turn is
// empty instead of a blank reply.
const refusalNote = chunk.delta.stop_reason === "refusal" && chunk.delta.stop_details?.explanation;
if (refusalNote) {
results.push(createChunk(state, { content: refusalNote }));
}
const finalChunk = createChunk(state, {}, state.finishReason);
if (state.usage) {

View File

@@ -14,13 +14,47 @@ import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } fro
* Translate OpenAI chunk to Responses API events
* @returns {Array} Array of events with { event, data } structure
*/
// Upstream Chat Completions usage -> Responses API usage shape.
// Without this, /v1/responses never reports usage: Responses clients (Codex CLI)
// keep their "context used" gauge pinned at 0 and never auto-compact, so a long
// session grows until the upstream context limit rejects it (9router issue #3432).
//
// Note this is stored under state.responsesUsage, NOT state.usage: state.usage is
// owned by the stream layer, which fills it with normalizeUsage()-shaped counts
// (prompt_tokens/prompt_tokens_details) and hands it to finalizeStream() for
// logging and cost accounting. Overwriting it with this shape silently drops
// cached/reasoning tokens from those stats.
function toResponsesUsage(usage) {
if (!usage || typeof usage !== "object") return null;
const inputTokens = [usage.input_tokens, usage.prompt_tokens].find(Number.isFinite) ?? 0;
const outputTokens = [usage.output_tokens, usage.completion_tokens].find(Number.isFinite) ?? 0;
const responseUsage = {
input_tokens: inputTokens,
output_tokens: outputTokens,
total_tokens: Number.isFinite(usage.total_tokens) ? usage.total_tokens : inputTokens + outputTokens
};
const cachedTokens = [usage.input_tokens_details?.cached_tokens, usage.prompt_tokens_details?.cached_tokens].find(Number.isFinite);
const reasoningTokens = [usage.output_tokens_details?.reasoning_tokens, usage.completion_tokens_details?.reasoning_tokens].find(Number.isFinite);
if (Number.isFinite(cachedTokens)) responseUsage.input_tokens_details = { cached_tokens: cachedTokens };
if (Number.isFinite(reasoningTokens)) responseUsage.output_tokens_details = { reasoning_tokens: reasoningTokens };
return responseUsage;
}
export function openaiToOpenAIResponsesResponse(chunk, state) {
if (!chunk) {
return flushEvents(state);
}
// Capture upstream usage BEFORE the choices guard below: the last OpenAI chunk
// may carry usage together with an empty choices array, and it must not be dropped.
if (chunk.usage) {
state.responsesUsage = toResponsesUsage(chunk.usage);
}
if (!chunk.choices?.length) return [];
const events = [];
const nextSeq = () => ++state.seq;
@@ -112,7 +146,19 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
for (const i in state.msgItemAdded) closeMessage(state, emit, i);
closeReasoning(state, emit);
for (const i in state.funcCallIds) closeToolCall(state, emit, i);
sendCompleted(state, emit);
// Upstreams report usage either on the finish chunk itself or on a trailing chunk
// whose `choices` array is empty (OpenAI does the latter). Emitting
// response.completed here would freeze the payload before that trailing chunk is
// parsed, so when usage is not known yet we leave completion to flushEvents(),
// which runs once the upstream stream ends and by then has seen every chunk.
//
// That only holds on the direct openai:openai-responses route. When this converter
// runs as the second hop of a pivot (Claude/Gemini/Kiro upstream), translateResponse()
// drops the terminal null chunk before reaching us — the first hop returns null for
// it, leaving nothing to iterate — so flushEvents() is never called and deferring
// would swallow the terminal event entirely. Keep the old behaviour there.
const flushReachesUs = state.targetFormat === FORMATS.OPENAI;
if (state.responsesUsage || !flushReachesUs) sendCompleted(state, emit);
}
return events;
@@ -376,7 +422,8 @@ function sendCompleted(state, emit) {
created_at: state.created,
status: "completed",
background: false,
error: null
error: null,
...(state.responsesUsage ? { usage: state.responsesUsage } : {})
}
});
}

View File

@@ -14,6 +14,9 @@ export const CLAUDE_STOP = {
MAX_TOKENS: "max_tokens",
TOOL_USE: "tool_use",
STOP_SEQUENCE: "stop_sequence",
// Anthropic's API-level refusal (streaming classifier / ToS). Arrives in
// message_delta with zero output tokens; stop_details carries the reason.
REFUSAL: "refusal",
};
// Gemini finishReason values.

View File

@@ -218,6 +218,12 @@ export function encodeField(fieldNum, wireType, value) {
return concatArrays(tagBytes, lengthBytes, dataBytes);
}
if (wireType === WIRE_TYPE.FIXED64) {
const buf = Buffer.alloc(8);
buf.writeDoubleLE(Number(value));
return concatArrays(tagBytes, buf);
}
return new Uint8Array(0);
}
@@ -887,6 +893,211 @@ export function extractTextFromResponse(payload) {
}
}
// ==================== AGENT SERVICE (google.protobuf.Value + MCP) ====================
const PB_VALUE = { NULL: 1, NUMBER: 2, STRING: 3, BOOL: 4, STRUCT: 5, LIST: 6 };
const PB_STRUCT_FIELDS = 1;
const PB_MAP_KEY = 1;
const PB_MAP_VALUE = 2;
const PB_LIST_VALUES = 1;
const MTD_NAME = 1;
const MTD_DESCRIPTION = 2;
const MTD_INPUT_SCHEMA = 3;
const MTD_PROVIDER = 4;
const MTD_TOOL_NAME = 5;
const MCP_TOOLS_TOOL = 1;
const MCP_ARGS_NAME = 1;
const MCP_ARGS_ENTRY = 2;
const MCP_ARGS_CALL_ID = 3;
const MCP_ARGS_TOOL_NAME = 5;
const MCR_SUCCESS = 1;
const MCR_ERROR = 2;
const MCR_TOOL_NOT_FOUND = 5;
const MCS_CONTENT = 1;
const MCS_IS_ERROR = 2;
const MCC_TEXT = 1;
const MCC_IMAGE = 2;
const MTC_TEXT = 1;
const MIC_DATA = 1;
const MIC_MIME = 2;
const MER_MESSAGE = 1;
const TNF_NAME = 1;
function asBytes(value) {
if (!value) return Buffer.alloc(0);
return Buffer.isBuffer(value) ? value : Buffer.from(value);
}
/**
* Encode a JS value as google.protobuf.Value (oneof body, no outer tag).
*/
export function encodeAgentValue(value) {
if (value === null || value === undefined) {
return encodeField(PB_VALUE.NULL, WIRE_TYPE.VARINT, 0);
}
if (typeof value === "boolean") {
return encodeField(PB_VALUE.BOOL, WIRE_TYPE.VARINT, value ? 1 : 0);
}
if (typeof value === "number") {
return encodeField(PB_VALUE.NUMBER, WIRE_TYPE.FIXED64, value);
}
if (typeof value === "string") {
return encodeField(PB_VALUE.STRING, WIRE_TYPE.LEN, value);
}
if (Array.isArray(value)) {
const items = value.map((item) => encodeField(PB_LIST_VALUES, WIRE_TYPE.LEN, encodeAgentValue(item)));
return encodeField(PB_VALUE.LIST, WIRE_TYPE.LEN, concatArrays(...items));
}
if (typeof value === "object") {
const entries = Object.entries(value).map(([key, val]) => encodeField(
PB_STRUCT_FIELDS,
WIRE_TYPE.LEN,
concatArrays(
encodeField(PB_MAP_KEY, WIRE_TYPE.LEN, key),
encodeField(PB_MAP_VALUE, WIRE_TYPE.LEN, encodeAgentValue(val)),
),
));
return encodeField(PB_VALUE.STRUCT, WIRE_TYPE.LEN, concatArrays(...entries));
}
return encodeField(PB_VALUE.STRING, WIRE_TYPE.LEN, String(value));
}
/**
* Decode google.protobuf.Value bytes back to a JS value.
*/
export function decodeAgentValue(bytes) {
const fields = decodeMessage(asBytes(bytes));
if (fields.has(PB_VALUE.NULL)) return null;
if (fields.has(PB_VALUE.BOOL)) return fields.get(PB_VALUE.BOOL)[0].value !== 0;
if (fields.has(PB_VALUE.NUMBER)) {
return asBytes(fields.get(PB_VALUE.NUMBER)[0].value).readDoubleLE(0);
}
if (fields.has(PB_VALUE.STRING)) {
return asBytes(fields.get(PB_VALUE.STRING)[0].value).toString("utf8");
}
if (fields.has(PB_VALUE.STRUCT)) {
const result = {};
for (const entry of decodeMessage(asBytes(fields.get(PB_VALUE.STRUCT)[0].value)).get(PB_STRUCT_FIELDS) || []) {
const pair = decodeMessage(asBytes(entry.value));
const key = asBytes(pair.get(PB_MAP_KEY)?.[0]?.value).toString("utf8");
if (key) result[key] = decodeAgentValue(pair.get(PB_MAP_VALUE)?.[0]?.value);
}
return result;
}
if (fields.has(PB_VALUE.LIST)) {
return (decodeMessage(asBytes(fields.get(PB_VALUE.LIST)[0].value)).get(PB_LIST_VALUES) || [])
.map((item) => decodeAgentValue(item.value));
}
return null;
}
function toolNameAndSchema(tool) {
const fn = tool?.function || tool || {};
return {
name: fn.name || tool?.name || "",
description: fn.description || tool?.description || "",
schema: fn.parameters || tool?.parameters || tool?.inputSchema || tool?.input_schema || {},
};
}
/**
* Encode agent.v1.McpToolDefinition body (name, description, Value schema, provider, tool_name).
*/
export function encodeMcpToolDefinition(tool) {
const { name, description, schema } = toolNameAndSchema(tool);
return concatArrays(
encodeField(MTD_NAME, WIRE_TYPE.LEN, name),
encodeField(MTD_DESCRIPTION, WIRE_TYPE.LEN, description),
encodeField(MTD_INPUT_SCHEMA, WIRE_TYPE.LEN, encodeAgentValue(schema)),
encodeField(MTD_PROVIDER, WIRE_TYPE.LEN, "9router"),
encodeField(MTD_TOOL_NAME, WIRE_TYPE.LEN, name),
);
}
/**
* Encode AgentRunRequest.mcp_tools: repeated McpToolDefinition under field 1.
*/
export function encodeMcpTools(tools = []) {
if (!tools?.length) return new Uint8Array();
return concatArrays(
...tools.map((tool) => encodeField(MCP_TOOLS_TOOL, WIRE_TYPE.LEN, encodeMcpToolDefinition(tool))),
);
}
/**
* Decode agent.v1.McpArgs (name, typed args map, toolCallId, toolName).
*/
export function decodeMcpArgs(bytes) {
const msg = decodeMessage(asBytes(bytes));
const args = {};
for (const entry of msg.get(MCP_ARGS_ENTRY) || []) {
const pair = decodeMessage(asBytes(entry.value));
const key = asBytes(pair.get(PB_MAP_KEY)?.[0]?.value).toString("utf8");
if (key) args[key] = decodeAgentValue(pair.get(PB_MAP_VALUE)?.[0]?.value);
}
const read = (field) => asBytes(msg.get(field)?.[0]?.value).toString("utf8");
return {
name: read(MCP_ARGS_NAME),
toolCallId: read(MCP_ARGS_CALL_ID),
toolName: read(MCP_ARGS_TOOL_NAME),
args,
};
}
function encodeMcpTextItem(text) {
return encodeField(
MCS_CONTENT,
WIRE_TYPE.LEN,
encodeField(MCC_TEXT, WIRE_TYPE.LEN, encodeField(MTC_TEXT, WIRE_TYPE.LEN, text)),
);
}
function encodeMcpImageItem(image) {
const data = image?.data || image || new Uint8Array();
const mimeType = image?.mimeType || "application/octet-stream";
return encodeField(
MCS_CONTENT,
WIRE_TYPE.LEN,
encodeField(
MCC_IMAGE,
WIRE_TYPE.LEN,
concatArrays(
encodeField(MIC_DATA, WIRE_TYPE.LEN, data),
encodeField(MIC_MIME, WIRE_TYPE.LEN, mimeType),
),
),
);
}
export function encodeMcpResultSuccess({ textItems = [], imageItems = [], isError = false } = {}) {
const success = concatArrays(
...textItems.map(encodeMcpTextItem),
...imageItems.map(encodeMcpImageItem),
encodeField(MCS_IS_ERROR, WIRE_TYPE.VARINT, isError ? 1 : 0),
);
return encodeField(MCR_SUCCESS, WIRE_TYPE.LEN, success);
}
export function encodeMcpResultError(message) {
return encodeField(
MCR_ERROR,
WIRE_TYPE.LEN,
encodeField(MER_MESSAGE, WIRE_TYPE.LEN, String(message || "")),
);
}
export function encodeMcpResultToolNotFound(name) {
return encodeField(
MCR_TOOL_NOT_FOUND,
WIRE_TYPE.LEN,
encodeField(TNF_NAME, WIRE_TYPE.LEN, String(name || "")),
);
}
// ==================== EXPORTS ====================
export default {
@@ -900,5 +1111,13 @@ export default {
decodeField,
decodeMessage,
parseConnectRPCFrame,
extractTextFromResponse
extractTextFromResponse,
encodeAgentValue,
decodeAgentValue,
encodeMcpToolDefinition,
encodeMcpTools,
decodeMcpArgs,
encodeMcpResultSuccess,
encodeMcpResultError,
encodeMcpResultToolNotFound,
};

View File

@@ -0,0 +1,232 @@
/**
* Helpers for the OpenCode Zen free-tier client fingerprint.
*
* Live upstream probes show that free-tier requests must include the lowercase
* file-search quartet (bash/glob/grep/read). Agent clients such as Claude Code
* may declare the same tools with different casing, so those case variants must
* be renamed instead of duplicated. The response side restores the caller's
* original spelling so downstream clients still recognise their own tool calls.
*/
/** Canonical names required by the upstream free-tier gate. */
export const OPENCODE_FINGERPRINT_TOOLS = ["bash", "glob", "grep", "read"];
// Request body -> names renamed for that request. transformRequest() mutates the
// same body object that chatCore passed into the executor, so a WeakMap keeps the
// mapping request-local without putting transport metadata on the wire.
const renamedToolNames = new WeakMap();
/** Canonical lowercase name when `name` is a quartet member; "" otherwise. */
export function fingerprintToolKey(name) {
const lower = String(name ?? "").trim().toLowerCase();
return OPENCODE_FINGERPRINT_TOOLS.includes(lower) ? lower : "";
}
/** Read a tool name from either flat ({name}) or chat ({function:{name}}) shape. */
function toolNameOf(tool) {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return "";
if (typeof tool.name === "string" && tool.name.trim()) return tool.name.trim();
const fn = tool.function;
if (fn && typeof fn === "object" && !Array.isArray(fn) && typeof fn.name === "string") {
return fn.name.trim();
}
return "";
}
/**
* Canonicalise only the fingerprint quartet and remove duplicate quartet
* variants. Non-fingerprint tools are preserved verbatim, including tools whose
* names differ only by case; they are outside OpenCode's fingerprint contract.
*
* @param {Array} tools
* @returns {{ tools: Array, map: Map<string,string> }} map: sent name -> original name
*/
export function concealFingerprintToolNames(tools) {
const map = new Map();
if (!Array.isArray(tools) || tools.length === 0) return { tools, map };
const seenQuartet = new Set();
const out = [];
for (const tool of tools) {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) {
out.push(tool);
continue;
}
const current = toolNameOf(tool);
const key = fingerprintToolKey(current);
if (!key) {
out.push(tool);
continue;
}
// `Bash` + `bash` is rejected upstream as a duplicate. Keep exactly one
// declaration for each quartet member.
if (seenQuartet.has(key)) continue;
seenQuartet.add(key);
if (current !== key) {
map.set(key, current);
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function)
? tool.function
: null;
out.push(fn ? { ...tool, function: { ...fn, name: key } } : { ...tool, name: key });
} else {
out.push(tool);
}
}
return { tools: out, map };
}
/** Append only genuinely missing quartet declarations. */
export function appendMissingFingerprintTools(tools, flat) {
const list = Array.isArray(tools) ? tools : [];
for (const name of OPENCODE_FINGERPRINT_TOOLS) {
if (list.some((tool) => fingerprintToolKey(toolNameOf(tool)) === name)) continue;
list.push(flat ? {
type: "function",
name,
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
} : {
type: "function",
function: {
name,
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
});
}
return list;
}
/** Point an explicit tool_choice at a quartet member after canonicalisation. */
export function retargetToolChoice(body, map) {
if (!body || typeof body !== "object" || !map?.size) return;
const choice = body.tool_choice;
if (!choice || typeof choice !== "object" || Array.isArray(choice)) return;
if (typeof choice.name === "string") {
const key = fingerprintToolKey(choice.name);
if (key && map.has(key)) body.tool_choice = { ...choice, name: key };
return;
}
const fn = choice.function;
if (fn && typeof fn === "object" && !Array.isArray(fn) && typeof fn.name === "string") {
const key = fingerprintToolKey(fn.name);
if (key && map.has(key)) {
body.tool_choice = { ...choice, function: { ...fn, name: key } };
}
}
}
/**
* Full request-side pass: canonicalise quartet case variants, remove duplicate
* quartet declarations, append missing members and preserve the legacy
* tool_choice defaults used by the OpenCode executor.
*
* @param {object} body
* @param {boolean} flat - true for Responses tools ({name}), false for chat tools
* @returns {Map<string,string>} map: sent name -> original name
*/
export function applyFingerprintTools(body, flat) {
if (!body || typeof body !== "object") return new Map();
const hadClientTools = Array.isArray(body.tools) && body.tools.length > 0;
const { tools, map } = concealFingerprintToolNames(body.tools);
body.tools = appendMissingFingerprintTools(tools, flat);
retargetToolChoice(body, map);
// Preserve the existing executor semantics. Responses uses auto when the
// fingerprint helper supplies tools; chat requests with no caller tools use
// none so the injected decoys cannot be selected.
if (!body.tool_choice) {
if (flat) body.tool_choice = "auto";
else if (!hadClientTools) body.tool_choice = "none";
}
recordRenamedToolNames(body, map);
return map;
}
/** Store the rename map for `body`. */
export function recordRenamedToolNames(body, map) {
if (!body || typeof body !== "object" || !map?.size) return;
renamedToolNames.set(body, map);
}
/** Retrieve the rename map for `body`. */
export function takeRenamedToolNames(body) {
if (!body || typeof body !== "object") return null;
return renamedToolNames.get(body) || null;
}
// Response side -------------------------------------------------------------
/** Restore caller tool spellings in supported response/event shapes. */
export function restoreToolNames(payload, map) {
if (!map?.size || !payload) return payload;
if (Array.isArray(payload)) return payload.map((item) => restoreToolNames(item, map));
if (typeof payload !== "object") return payload;
let out = payload;
const put = (key, value) => {
if (out === payload) out = { ...payload };
out[key] = value;
};
// Claude streaming content_block_start event.
if (payload.type === "content_block_start") {
const block = payload.content_block;
if (block?.type === "tool_use" && typeof block.name === "string" && map.has(block.name)) {
put("content_block", { ...block, name: map.get(block.name) });
}
}
// Claude non-streaming message body.
if (Array.isArray(payload.content)) {
put("content", payload.content.map((block) =>
block?.type === "tool_use" && typeof block.name === "string" && map.has(block.name)
? { ...block, name: map.get(block.name) }
: block));
}
// OpenAI Chat Completions, both streaming delta and JSON message shapes.
if (Array.isArray(payload.choices)) {
put("choices", payload.choices.map((choice) => {
let changed = false;
const next = { ...choice };
for (const holder of ["delta", "message"]) {
const value = choice?.[holder];
if (!value || !Array.isArray(value.tool_calls) || value.tool_calls.length === 0) continue;
const calls = value.tool_calls.map((call) => {
const name = call?.function?.name;
if (typeof name === "string" && map.has(name)) {
changed = true;
return { ...call, function: { ...call.function, name: map.get(name) } };
}
return call;
});
next[holder] = { ...value, tool_calls: calls };
}
return changed ? next : choice;
}));
}
// OpenAI Responses final JSON body.
if (Array.isArray(payload.output)) {
put("output", payload.output.map((item) =>
item?.type === "function_call" && typeof item.name === "string" && map.has(item.name)
? { ...item, name: map.get(item.name) }
: item));
}
// OpenAI Responses SSE events such as response.output_item.added/done.
const item = payload.item;
if (item?.type === "function_call" && typeof item.name === "string" && map.has(item.name)) {
put("item", { ...item, name: map.get(item.name) });
}
return out;
}

View File

@@ -215,8 +215,11 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) {
const vercelRelayUrl = normalizeString(proxyOptions?.vercelRelayUrl);
if (vercelRelayUrl) {
const parsed = new URL(targetUrl);
const baseHeaders = options.headers instanceof Headers
? Object.fromEntries(options.headers.entries())
: { ...(options.headers || {}) };
const relayHeaders = {
...options.headers,
...baseHeaders,
"x-relay-target": `${parsed.protocol}//${parsed.host}`,
"x-relay-path": `${parsed.pathname}${parsed.search}`,
};

View File

@@ -60,7 +60,13 @@ export function createSSEStream(options = {}) {
const decoder = new TextDecoder("utf-8", { fatal: false });
const state = mode === STREAM_MODE.TRANSLATE
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null }
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null,
// Which upstream format this stream came from. A response translator can be
// reached either directly (target === its registered source) or as the second
// hop of a pivot, and on the terminal null chunk the pivot drops it — so a
// translator that defers closing events until flush needs to know which case
// it is in. Absent/undefined means "unknown", i.e. do not defer.
targetFormat }
: null;
let totalContentLength = 0;