Merge origin/master (v0.5.81) into gitea/new_feature

Resolve conflicts:
- streamingHandler.js: merge buildStreamErrorBytes onAbortTerminal + local shouldPersistRequestDetail & streamStatusForContent
- capabilities.js: preserve server-injected user-asserted caps and models.dev catalog lookup; wire CommandCode /alpha/generate caps inside resolve()
- commandcode.js (services/usage): adopt upstream whoami + billing credits/subscriptions with 5h/weekly rate windows and plan caps
- openai-to-commandcode.js: merge toNativeImageBlock (data-URI & http(s) support) and assistant reasoning_content preservation
- commandcode-to-openai.js: adopt upstream mid-stream error throw for clean retry and abortion
- tests: sync commandcode test suite and exclude .next from vitest config
This commit is contained in:
2026-09-22 10:08:58 +07:00
95 changed files with 4059 additions and 1016 deletions

View File

@@ -178,6 +178,11 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C
// makes the backend flag the request and answer 429 Quota Exhausted.
export const ANTIGRAVITY_PROMPT_REWRITES = [
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
{ from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." },
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.
{ from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" },
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
];

View File

@@ -27,11 +27,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]);
// ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only.
function findModel(models, modelId, aliasOrId) {
if (!models) return undefined;
const found = models.find(m => m.id === modelId);
const baseModelId = typeof modelId === "string"
? modelId.replace(/\([^()]+\)\s*$/, "").trim()
: modelId;
const found = models.find(m => m.id === modelId || m.id === baseModelId);
if (found) return found;
if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined;
const normalized = normalizeModelId(modelId);
if (normalized === modelId) return undefined;
const normalized = normalizeModelId(baseModelId);
if (normalized === baseModelId) return undefined;
return models.find(m => m.id === normalized);
}

View File

@@ -212,7 +212,7 @@ export class AntigravityExecutor extends BaseExecutor {
const modifiedParts = parts?.map(p => {
if (!p.functionCall) return p;
const callId = p.functionCall.id;
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null;
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null;
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
firstFunctionCallSeen = true;
if (callSig) {

View File

@@ -128,7 +128,7 @@ export class BaseExecutor {
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
const url = this.buildUrl(model, stream, urlIndex, credentials);
const transformedBody = this.transformRequest(model, body, stream, credentials);
const headers = this.buildHeaders(credentials, stream, url, model);
const headers = this.buildHeaders(credentials, stream, url, model, transformedBody);
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;

View File

@@ -47,10 +47,24 @@ export class CommandCodeExecutor extends BaseExecutor {
}
async execute(opts) {
const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result;
result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
return result;
const maxRetries = 2;
for (let attempt = 0; attempt <= maxRetries; attempt++) {
const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result;
const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
if (!wrappedResponse.ok && attempt < maxRetries) {
const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504;
if (isRetryableStatus) {
opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`);
await new Promise(r => setTimeout(r, 1000 * (attempt + 1)));
continue;
}
}
result.response = wrappedResponse;
return result;
}
}
parseError(response, bodyText) {

View File

@@ -146,7 +146,7 @@ export class DefaultExecutor extends BaseExecutor {
return BEARER;
}
buildHeaders(credentials, stream = true, url, model) {
buildHeaders(credentials, stream = true, url, model, body = null) {
const rt = credentials?.runtimeTransport;
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
@@ -166,7 +166,7 @@ export class DefaultExecutor extends BaseExecutor {
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
if (model && (this.provider === "claude"
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
headers["Anthropic-Beta"] = selectAnthropicBeta(model, body);
}
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams

View File

@@ -1,7 +1,8 @@
import crypto from "node:crypto";
import { DefaultExecutor } from "./default.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import { modelTargetFormat } from "../providers/models/schema.js";
import { getProviderModels } from "../config/providerModels.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
@@ -45,8 +46,11 @@ function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …).
// Reading the registry keeps this in sync with config — never hardcode model ids here.
function isResponsesModel(model) {
return isMuseSparkModel(baseModelId(model));
const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model));
return modelTargetFormat(entry) === "openai-responses";
}
// Flatten Chat Completions tool declarations into the Responses flat shape and
@@ -90,6 +94,12 @@ function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
// Strip prior-turn reasoning items: Muse Spark contributor models route to
// an upstream Console backend where encrypted_content cannot be validated across
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
if (item.type === "reasoning") return false;
delete item.encrypted_content;
delete item.reasoning_encrypted_content;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);

View File

@@ -1,24 +1,311 @@
import crypto from "crypto";
import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js";
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../translator/formats/responsesApi.js";
const OPENCODE_UA = "opencode";
const OPENCODE_UA = "opencode/1.18.31";
const MAX_SESSION_LENGTH = 256;
const MAX_TOOL_NAME_LEN = 128;
const SESSION_HEADER = "x-opencode-session";
const SESSION_FIELD = "_opencodeSession";
const REQ_FIELD = "_opencodeRequest";
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
// OpenCode free tier requires both 'bash' and 'read' in tools payload.
// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read)
// take precedence while satisfying upstream verification.
const OPENCODE_DECOY_CHAT_TOOLS = [
{
type: "function",
function: {
name: "bash",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
},
{
type: "function",
function: {
name: "read",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
},
];
const OPENCODE_DECOY_RESPONSES_TOOLS = [
{
type: "function",
name: "bash",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
{
type: "function",
name: "read",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
];
function cloakOpencodeTools(body, isResponses) {
if (!body || typeof body !== "object") return;
if (isResponses) {
if (!Array.isArray(body.tools)) body.tools = [];
const names = new Set(body.tools.map((t) => t.name || t.function?.name));
for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) {
if (!names.has(tool.name)) body.tools.push({ ...tool });
}
if (!body.tool_choice) body.tool_choice = "auto";
} else {
const hasTools = Array.isArray(body.tools) && body.tools.length > 0;
if (!hasTools) {
body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } }));
if (!body.tool_choice) body.tool_choice = "none";
} else {
const names = new Set(body.tools.map((t) => t.function?.name || t.name));
for (const tool of OPENCODE_DECOY_CHAT_TOOLS) {
if (!names.has(tool.function.name)) {
body.tools.push({ ...tool, function: { ...tool.function } });
}
}
}
}
}
function hasValidOpencodeVersion(ua) {
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
if (!m) return false;
const major = parseInt(m[1], 10);
const minor = parseInt(m[2], 10);
return major > 1 || (major === 1 && minor >= 17);
}
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
const RESPONSES_MODELS = new Set([
"muse-spark-1.2-contributor-free",
"muse-spark-1.3-contributor-free",
]);
const MESSAGES_MODELS = new Set(["union-alpha"]);
function generateRequestId() {
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
let lastTimestamp = 0;
let counter = 0;
function unstableRandom() {
const bytes = crypto.randomBytes(14);
let randomPart = "";
for (let i = 0; i < 14; i++) {
randomPart += BASE62_CHARS[bytes[i] % 62];
}
return randomPart;
}
function generateSessionId() {
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
export function generateSessionId(timestamp = Date.now()) {
if (timestamp !== lastTimestamp) {
lastTimestamp = timestamp;
counter = 0;
}
counter++;
const current = BigInt(timestamp) * 0x1000n + BigInt(counter);
const value = ~current;
const time = Array.from({ length: 6 }, (_, index) =>
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
.toString(16)
.padStart(2, "0")
).join("");
return `ses_${time}${unstableRandom()}`;
}
export function generateRequestId(timestamp = Date.now()) {
const current = BigInt(timestamp) * 0x1000n + 1n;
const value = current;
const time = Array.from({ length: 6 }, (_, index) =>
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
.toString(16)
.padStart(2, "0")
).join("");
return `msg_${time}${unstableRandom()}`;
}
export function translateSessionId(sessionId, clientTool = "") {
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
return sessionId.trim();
}
const digest = crypto
.createHash("sha256")
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
.digest();
const timeHex = digest.subarray(0, 6).toString("hex");
let randomPart = "";
for (let i = 6; i < 20; i++) {
randomPart += BASE62_CHARS[digest[i] % 62];
}
return `ses_${timeHex}${randomPart}`;
}
function normalizeSession(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return normalized;
}
function nativeSession(headers) {
if (!headers || typeof headers !== "object") return null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) {
const normalized = normalizeSession(value);
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
}
}
return null;
}
// Upstream free-tier quota is accounted per session. Minting a fresh
// x-opencode-session on every request burns through it and surfaces as
// 429 FreeUsageLimitError with growing reset-after delays, while the real
// CLI reuses one long-lived canonical session per conversation. Mirror
// that: one stable canonical session per downstream identity, evicted
// after MEMORY_CONFIG.sessionTtlMs like the other session stores.
const stableOpencodeSessions = new Map();
const MAX_STABLE_SESSIONS = 1000;
const stableSessionCleanup = setInterval(() => {
const now = Date.now();
for (const [key, entry] of stableOpencodeSessions) {
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) {
stableOpencodeSessions.delete(key);
}
}
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
if (stableSessionCleanup.unref) stableSessionCleanup.unref();
function identityKey(credentials) {
const connectionId = credentials?.connectionId || credentials?.id;
if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`;
const raw = credentials?.rawHeaders || {};
const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || "";
if (auth) {
const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32);
return `opencode:auth:${digest}`;
}
return "opencode:default";
}
export function stableSessionId(credentials) {
const key = identityKey(credentials);
const existing = stableOpencodeSessions.get(key);
if (existing) {
existing.lastUsed = Date.now();
stableOpencodeSessions.delete(key);
stableOpencodeSessions.set(key, existing);
return existing.sessionId;
}
const sessionId = generateSessionId();
if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) {
stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value);
}
stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() });
return sessionId;
}
function lastUserText(body) {
try {
if (!body || typeof body !== "object") return "";
const arr = Array.isArray(body.messages)
? body.messages
: Array.isArray(body.input)
? body.input
: null;
if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : "";
for (let i = arr.length - 1; i >= 0; i--) {
const msg = arr[i];
if (!msg) continue;
if (msg.role && msg.role !== "user") continue;
const content = msg.content;
if (typeof content === "string" && content.trim()) return content.trim().slice(-600);
if (Array.isArray(content)) {
const text = content
.map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || ""))
.join(" ")
.trim();
if (text) return text.slice(-600);
}
}
} catch {
return "";
}
return "";
}
// The real CLI sends the current user message id (stable per turn, same on
// retries) as x-opencode-request. Derive it deterministically from the
// session plus the last user message so retries share the id.
export function deriveRequestId(sessionId, body) {
const text = lastUserText(body);
if (!text) return generateRequestId();
const digest = crypto
.createHash("sha256")
.update(`opencode-req\0${sessionId || ""}\0${text}`)
.digest();
const timeHex = digest.subarray(0, 6).toString("hex");
let randomPart = "";
for (let i = 6; i < 20; i++) {
randomPart += BASE62_CHARS[digest[i] % 62];
}
const id = `msg_${timeHex}${randomPart}`;
return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId();
}
function normalizeRequestId(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null;
}
function bodyHasSessionHints(body) {
try {
if (!body || typeof body !== "object") return false;
if (typeof body.session_id === "string" && body.session_id.trim()) return true;
if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true;
if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true;
if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true;
if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true;
const arr = Array.isArray(body.messages)
? body.messages
: Array.isArray(body.input)
? body.input
: null;
if (arr) {
let assistantText = "";
for (const msg of arr) {
if (msg?.role === "assistant") {
const content = msg.content;
if (typeof content === "string") assistantText += content;
else if (Array.isArray(content)) {
for (const part of content) assistantText += part?.text || part?.output || "";
}
if (assistantText.length >= 50) return true;
}
}
}
return false;
} catch {
return false;
}
}
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
@@ -31,14 +318,116 @@ function isResponsesModel(model) {
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
}
function resolveOpencodeSession(body, credentials) {
function isMessagesModel(model) {
return MESSAGES_MODELS.has(baseModelId(model));
}
function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) {
const headers = credentials?.rawHeaders || {};
return resolveSessionId({
headers,
body,
connectionId: credentials?.connectionId,
scope: "opencode",
generate: generateSessionId,
const native = nativeSession(headers);
if (native) return native;
let incoming = null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) {
incoming = normalizeSession(value);
break;
}
}
const hinted = incoming || normalizeSession(providerSessionId);
if (hinted) return translateSessionId(hinted, clientTool);
if (credentials?.connectionId || bodyHasSessionHints(body)) {
let viaManager = null;
try {
viaManager = resolveSessionId({
headers,
body,
connectionId: credentials?.connectionId,
scope: "opencode",
});
} catch {
viaManager = null;
}
if (viaManager) return translateSessionId(viaManager, clientTool);
}
return stableSessionId(credentials);
}
function resolveOpencodeRequestId(body, credentials, sessionId) {
const raw = credentials?.rawHeaders || {};
for (const [key, value] of Object.entries(raw)) {
if (key.toLowerCase() === "x-opencode-request") {
const normalized = normalizeRequestId(value);
if (normalized) return normalized;
break;
}
}
return deriveRequestId(sessionId, body);
}
function normalizeResponsesTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
const name = rawName.trim();
if (!name) return false;
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
? tool.parameters
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
for (const k of Object.keys(tool)) delete tool[k];
tool.type = "function";
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
if (description) tool.description = description;
tool.parameters = parameters;
validNames.add(tool.name);
return true;
});
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
if (!n || !validNames.has(n)) delete body.tool_choice;
}
}
}
function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
// Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials
// (`Bearer public`) routing to an upstream OpenAI/Console account pool.
// OpenAI Responses API strictly enforces that reasoning `encrypted_content`
// can only be decrypted by the exact caller/account that issued it; sending it
// across different accounts or rotating proxy relays triggers:
// [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400).
// Furthermore, under stateless mode (store=false), omitting encrypted_content
// causes OpenAI to reject the referenced reasoning item as "not found or was deleted".
// Dropping prior reasoning items allows multi-turn conversations and tool-calling
// loops to succeed cleanly.
if (item.type === "reasoning") return false;
delete item.encrypted_content;
delete item.reasoning_encrypted_content;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
item.call_id = clampResponsesCallId(item.call_id);
item.arguments = coerceResponsesArguments(item.arguments);
return true;
}
if (item.type === "function_call_output") {
item.call_id = clampResponsesCallId(item.call_id);
item.output = coerceResponsesOutput(item.output);
return true;
}
return true;
});
}
@@ -68,12 +457,35 @@ function normalizeOpencodeReasoning(model, body) {
export class OpenCodeExecutor extends BaseExecutor {
constructor() {
super("opencode", PROVIDERS.opencode);
this._currentSessionId = null;
}
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
const sourceCredentials = credentials || {};
const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool);
return {
...sourceCredentials,
[SESSION_FIELD]: session,
[REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session),
};
}
transformRequest(model, body, stream, credentials) {
this._currentSessionId = resolveOpencodeSession(body, credentials);
if (isResponsesModel(model)) {
if (body && typeof body === "object" && model && !body.model) body.model = model;
// Zen rejects non-streaming requests on free models with 403 FreeTierError;
// always stream upstream and let the handler layer aggregate for non-stream clients.
if (body && typeof body === "object") body.stream = true;
if (isResponsesModel(model || body?.model) && body && typeof body === "object") {
// ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng.
if ("tool_choice" in body && body.tool_choice !== "auto"
&& this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) {
body.tool_choice = "auto";
}
const normalized = normalizeResponsesInput(body.input);
if (normalized) body.input = normalized;
if (!Array.isArray(body.input) || body.input.length === 0) {
body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
}
// Responses API names the output cap max_output_tokens and takes thinking
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
if (body.max_output_tokens === undefined) {
@@ -83,34 +495,53 @@ export class OpenCodeExecutor extends BaseExecutor {
delete body.max_tokens;
delete body.max_completion_tokens;
normalizeOpencodeReasoning(model, body);
body.stream = true;
body.store = false;
normalizeResponsesTools(body);
sanitizeResponsesItems(body);
if (!Array.isArray(body.tools) || body.tools.length === 0) {
cloakOpencodeTools(body, true);
}
} else if (body && typeof body === "object") {
cloakOpencodeTools(body, false);
}
return injectReasoningContent({ provider: this.provider, model, body });
}
buildUrl(model) {
const base = this.config.baseUrl;
return isResponsesModel(model)
? `${base}/zen/v1/responses`
: `${base}/zen/v1/chat/completions`;
async execute(args) {
return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) });
}
buildHeaders(credentials, stream = true) {
buildUrl(model) {
const base = this.config.baseUrl;
if (isResponsesModel(model)) return `${base}/zen/v1/responses`;
if (isMessagesModel(model)) return `${base}/zen/v1/messages`;
return `${base}/zen/v1/chat/completions`;
}
buildHeaders(credentials, stream = true, url = "") {
const raw = credentials?.rawHeaders || {};
const lower = {};
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
const downstreamUa = lower["user-agent"] || "";
const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode");
const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa);
return {
const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD];
const downstreamReq = normalizeRequestId(lower["x-opencode-request"]);
const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId();
const headers = {
"Content-Type": "application/json",
"Authorization": "Bearer public",
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
"x-opencode-client": lower["x-opencode-client"] || "desktop",
"x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(),
"x-opencode-request": lower["x-opencode-request"] || generateRequestId(),
"x-opencode-session": session,
"x-opencode-request": requestId,
"x-opencode-project": lower["x-opencode-project"] || "global",
"Accept": stream ? "text/event-stream" : "*/*",
};
if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION;
return headers;
}
}

View File

@@ -29,11 +29,18 @@ import {
zedLlmFetch,
} from "../shared/zedAuth.js";
// Wire values for the `provider` field of POST /completions. These are NOT
// display names: cloud.zed.dev matches them exactly, and an unrecognized value
// fails the whole request with `500 {"message":"An internal server error
// occurred."}` before the model is ever looked at. Spellings come from Zed's
// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore),
// `x_ai` follows the same convention — so feeding a catalog value back through
// normalizeZedProvider is identity.
const ZED_PROVIDER = {
anthropic: "Anthropic",
openai: "OpenAi",
google: "Google",
xai: "XAi",
anthropic: "anthropic",
openai: "open_ai",
google: "google",
xai: "x_ai",
};
function normalizeZedProvider(value, model) {
@@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) {
return openaiToClaudeRequest(model, body, true);
}
if (provider === ZED_PROVIDER.google) {
return openaiToGeminiRequest(model, body, true);
const geminiRequest = openaiToGeminiRequest(model, body, true);
// Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the
// public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`,
// `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google
// path so Zed applies its own defaults — scoped here so native Gemini/
// Antigravity is untouched.
delete geminiRequest.safetySettings;
return geminiRequest;
}
if (provider === ZED_PROVIDER.openai) {
return openaiToOpenAIResponsesRequest(model, body, true, credentials);

View File

@@ -6,8 +6,9 @@ import {
} from "../../utils/stream.js";
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
import { PROVIDERS } from "../../config/providers.js";
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
import { buildStreamErrorBytes } from "../../utils/streamHelpers.js";
import {
buildRequestDetail,
extractRequestConfig,
@@ -130,13 +131,21 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
// Terminal bytes when the stream aborts after HTTP 200 was already sent, so the
// client sees a real error instead of a silently truncated stream.
// Responses passthrough keeps its own response.failed shape; every other client
// format gets the OpenAI error frame + [DONE], or `event: error` for Claude.
const isResponsesPassthrough =
sourceFormat === FORMATS.OPENAI_RESPONSES &&
targetFormat === FORMATS.OPENAI_RESPONSES;
const onAbortTerminal = isResponsesPassthrough
? buildAbortedResponsesTerminalBytes
: null;
: (message) =>
buildStreamErrorBytes(
HTTP_STATUS.GATEWAY_TIMEOUT,
message,
sourceFormat,
);
const stallTimeoutMs =
PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
const transformedBody = pipeWithDisconnect(

View File

@@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = {
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
// DeepSeek V4.1-Flash is natively multimodal — models.dev lists
// opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream
// the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the
// same image capability as the exp id above. "deepseek-flash" is the GA id on the
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
// short-circuits the pattern table, so a vision-only delta would drop them.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
"coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
@@ -131,6 +141,8 @@ export const MODEL_CAPABILITIES = {
// via OpenAI Responses input_image; reasoning supports up to xhigh.
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
// OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output
"union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 },
};
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
@@ -214,6 +226,13 @@ export const PROVIDER_CAPABILITIES = {
// contract). maxOutput 128000 per the server's product-config payload.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors
// the codebuddy-cn entry (the openai-style reasoning_effort format matters:
// the generic *deepseek-v4* pattern would otherwise pick the vendor-native
// "deepseek" thinking shape, which the CodeBuddy gateway does not accept).
"codebuddy-intl": {
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
@@ -251,6 +270,16 @@ export const PROVIDER_CAPABILITIES = {
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
"laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 },
},
// Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge
// the library page publishes for this model (text+image in, 1M context).
// ponytail: thinkingFormat stays "deepseek" to preserve today's body shape;
// Ollama's native toggle is the top-level `think` field (bool or
// low/medium/high/max), which no format in thinkingUnified.js emits yet —
// openai-to-ollama.js drops it. Wire a "think" format when thinking on
// Ollama Cloud is actually needed.
"ollama": {
"deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
},
};
/**
@@ -349,7 +378,11 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
// ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } },
// v4.1+ has real image input (probed live on Alibaba MaaS: correct color
// read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore
// them (answered "Unknown"), so vision stays scoped to v4.* dotted releases.
{ pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } },
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } },
{ pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
{ pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
{ pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } },
@@ -425,6 +458,7 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
// globalThis, which IS shared across server bundles in the same process.
// Same reason the browser bundle is safe: it never calls a setter, so the slots
// stay empty and every consumer below short-circuits.
let catalogSource = null;
const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
catalog: null, // { getModalities, getLimits } — synced models.dev catalog
userCaps: null, // (provider, model) => asserted caps — dashboard toggles
@@ -432,10 +466,20 @@ const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
/**
* Install the synced catalog reader (server only).
* @param {{ getModalities: Function, getLimits: Function } | null} source
* @param {{ getModalities: (provider: string, model: string) => object|null,
* getLimits: (provider: string, model: string) => object|null } | null} source
*/
export function setCatalogSource(source) {
catalogSource = source || null;
SOURCE_SLOTS.catalog = source || null;
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source || null;
}
function getCatalogSource() {
if (catalogSource) return catalogSource;
if (SOURCE_SLOTS.catalog) return (catalogSource = SOURCE_SLOTS.catalog);
if (typeof globalThis === "undefined") return null;
return (catalogSource = globalThis.__9rCatalogSource || null);
}
// Capabilities the user asserted per provider+model (dashboard "Add/Edit Model"
@@ -478,16 +522,17 @@ function applyUserCaps(result, provider, model) {
// flips when an outside source positively declares support.
function refine(base, provider, model) {
const result = { ...DEFAULT_CAPABILITIES, ...base };
const catalogSource = SOURCE_SLOTS.catalog;
if (catalogSource) {
const modalities = catalogSource.getModalities(model);
const source = getCatalogSource();
if (source) {
const modalities = source.getModalities(provider, model);
if (modalities) {
for (const key of MODALITY_KEYS) {
if (modalities[key] === true) result[key] = true;
}
}
const limits = catalogSource.getLimits(provider, model);
const limits = source.getLimits(provider, model);
if (limits) {
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
@@ -499,12 +544,67 @@ function refine(base, provider, model) {
return result;
}
// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models
// default to vision; only this denylist stays text-only.
const COMMANDCODE_TEXT_ONLY = new Set([
"deepseek/deepseek-v4-pro",
"deepseek/deepseek-v4-flash",
"deepseek/deepseek-v4-flash-fast",
"zai-org/glm-5.3",
"zai-org/glm-5.2",
"zai-org/glm-5.2-fast",
"zai-org/glm-5.1",
"zai-org/glm-5",
"minimaxai/minimax-m2.7",
"minimax/minimax-m2.7-free",
"minimaxai/minimax-m2.5",
"xiaomi/mimo-v2.5-pro",
"qwen/qwen3.6-max-preview",
"qwen/qwen3.7-max",
"meituan/longcat-2.0:free",
"stepfun/step-3.5-flash",
"tencent/hy4-preview",
"tencent/hy3",
"tencent/hy3-paid",
"nvidia/nemotron-3-ultra-550b-a55b",
"poolside/laguna-s-2.1-free",
"inclusionai/ling-3.0-flash-free",
"inclusionai/ling-3.0-flash-sante:free",
]);
function isCommandCodeTextOnly(model) {
const key = String(model || "").toLowerCase();
if (COMMANDCODE_TEXT_ONLY.has(key)) return true;
for (const id of COMMANDCODE_TEXT_ONLY) {
const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
if (key === base || key.endsWith("/" + base)) return true;
}
return false;
}
export function getCapabilitiesForModel(provider, model) {
if (!model) return { ...DEFAULT_CAPABILITIES };
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
const baseModel = model.includes("/") ? model.split("/").pop() : model;
const resolve = () => {
// CommandCode wire is /alpha/generate for every model. Family patterns
// (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here.
if (provider === "commandcode" || provider === "cmc") {
const providerCaps = PROVIDER_CAPABILITIES.commandcode;
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
return {
...DEFAULT_CAPABILITIES,
reasoning: true,
thinkingFormat: "commandcode",
thinkingEffortSupported: true,
vision: !isCommandCodeTextOnly(model),
contextWindow: 1000000,
maxOutput: 384000,
};
}
// 1. Provider-specific override
if (provider) {
const providerCaps = PROVIDER_CAPABILITIES[provider];

View File

@@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
// Trimmed upstream catalog, read by the add-models skill (not by the router).
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
// Schema of the file this module reads. The writer stamps it; a file carrying an
// older value predates provider-scoped modality keys, and its flat keys are not
// looked up here, so the sync rebuilds it instead of asking upstream for a 304.
export const CATALOG_VERSION = 2;
const EMPTY = { models: {}, providers: {} };
let cache = EMPTY;
let cachedMtime = -1;
@@ -45,14 +50,19 @@ function load() {
return cache;
}
// Modality is a property of the model itself — any gateway serving it inherits
// the same image/video/pdf support, so this is keyed by model id alone.
export function getCatalogModalities(model) {
return load().models[baseId(model)] || null;
// Modalities are recorded per gateway upstream, and gateways disagree about the
// same weights — some do not proxy images at all — so the key is provider +
// model, in the local provider id space, exactly like the limits below. Keying
// by model id alone made short ids collide across vendors: "auto", "free" and
// "efficient" are router modes in one catalog and model names in another, and a
// request to the router mode inherited a stranger's vision.
export function getCatalogModalities(provider, model) {
if (!provider) return null;
return load().models[`${provider}:${baseId(model)}`] || null;
}
// Context and output limits are a property of the gateway, not the model: each
// one truncates differently, so these stay keyed by provider + model.
// Context and output limits are a property of the gateway too: each one
// truncates differently, so these stay keyed by provider + model.
export function getCatalogLimits(provider, model) {
const byProvider = provider && load().providers[provider];
if (!byProvider) return null;

View File

@@ -111,6 +111,8 @@ export const MODEL_PRICING = {
"deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 },
// === GLM ===

View File

@@ -58,7 +58,9 @@ export default {
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
{ id: "hy3-preview", name: "Hy3 Preview" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
// deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same
// catalog as CN; the old endpoint still answers 200 but the list is the contract).
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
],
oauth: {

View File

@@ -65,6 +65,9 @@ export default {
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
// Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived
// from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398).
{ id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },

View File

@@ -74,4 +74,8 @@ export default {
{ id: "claude-opus-4-7", name: "Claude Opus 4.7" },
{ id: "claude-haiku-4-5", name: "Claude Haiku 4.5" },
],
features: {
usage: true,
usageApikey: true,
},
};

View File

@@ -59,6 +59,7 @@ export default {
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },

View File

@@ -30,6 +30,7 @@ export default {
{ id: "glm-4.7-flash", name: "GLM 4.7 Flash" },
{ id: "qwen3.5", name: "Qwen3.5" },
{ id: "minimax-m3", name: "MiniMax M3" },
{ id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" },
],
serviceKinds: ["llm", "webFetch"],
fetchConfig: {

View File

@@ -13,7 +13,7 @@ export default {
textIcon: "OC",
website: "https://opencode.ai/auth",
notice: {
text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
apiKeyUrl: "https://opencode.ai/auth",
},
},

View File

@@ -17,13 +17,17 @@ export default {
headers: {
"x-opencode-client": "desktop",
},
forceStream: true,
noAuth: true,
quirks: {
forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"],
},
},
models: [
// Muse Spark models are served by /zen/v1/responses; the rest stay on
// /chat/completions, so the format is declared per-model, not per-provider.
// Endpoint formats differ per model, so declare non-chat models explicitly.
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
{ id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" },
],
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true,

View File

@@ -1,10 +1,9 @@
// Zed provider — RSA keypair callback auth (NOT standard OAuth).
export default {
id: "zed",
priority: 10,
priority: 999,
alias: "zd",
uiAlias: "zd",
hidden: true,
display: {
name: "Zed",
icon: "code",

View File

@@ -62,8 +62,17 @@ const ANTHROPIC_BETA_BASE = [
const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"];
// Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them.
export function selectAnthropicBeta(model = "") {
const flags = [...ANTHROPIC_BETA_BASE];
// `redact-thinking` asks Anthropic to return signature-only thinking blocks, which
// is right for clients that never render thinking but blanks the summaries a
// client explicitly requested with `thinking.display: "summarized"`.
const ANTHROPIC_BETA_REDACT_THINKING = "redact-thinking-2026-02-12";
export function wantsThinkingSummaries(body) {
return body?.thinking?.display === "summarized";
}
export function selectAnthropicBeta(model = "", body = null) {
const flags = ANTHROPIC_BETA_BASE.filter((flag) => flag !== ANTHROPIC_BETA_REDACT_THINKING || !wantsThinkingSummaries(body));
if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT);
return flags.join(",");
}

View File

@@ -26,6 +26,7 @@ const FORMAT_LEVELS = {
qwen: L.base,
kimi: L.levelMax,
deepseek: L.hiMax,
commandcode: ["none", "low", "medium", "high", "xhigh", "max"],
minimax: L.onOff,
hunyuan: L.base,
step: L.base,
@@ -40,6 +41,10 @@ const PATTERN_THINKING = [
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
// (disable thinking instead). none kept for the picker = disable.
{ pattern: "*deepseek-v4.*", levels: ["none", "low", "medium", "high", "xhigh", "max"] },
// codebuddy-cn per-model effort sets — the server's product-config payload
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
@@ -52,6 +57,8 @@ const PATTERN_THINKING = [
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
// codebuddy-intl rides the same gateway catalog, so its deepseek levels match.
{ provider: "codebuddy-intl", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
];
// The generic level set used when a model's thinking format is unknown. Exported

View File

@@ -29,7 +29,7 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) {
// Request-scoped rule: the request body itself is at fault — no cooldown,
// no account lock. Caller must stop rotating and surface the error.
if (rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
return { shouldFallback: false, requestScoped: true, cooldownMs: 0 };
return { shouldFallback: false, cooldownMs: 0 };
}
// Text-based rule: match substring in error message
@@ -51,6 +51,20 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) {
}
}
// Request-scoped client errors that matched no rule above: a 400 caused by the
// request itself (context overflow, malformed body, unsupported parameter) says
// nothing about the credential, so cooling the account down only removes a
// healthy connection from rotation. With a single connection it is worse: every
// later request in the window fails with a copy of this very error
// ("all 1 accounts locked for <model> | lastError=[400]: ..."), which hides the
// real cause from the caller and makes unrelated sessions look like they hit the
// same limit. Hand the upstream error back for this request instead.
// Account-scoped statuses keep their rules above (401/402/403/404/429), and the
// text rules still win for rate-limit / quota / capacity wording.
if (status >= 400 && status < 500 && status !== 401 && status !== 402 && status !== 403 && status !== 429) {
return { shouldFallback: false, cooldownMs: 0 };
}
// Default: transient cooldown for any unmatched error
return { shouldFallback: true, cooldownMs: TRANSIENT_COOLDOWN_MS };
}

View File

@@ -124,6 +124,8 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) {
// Config-driven prefix → provider inference (first match wins, fallback "openai").
const MODEL_PREFIX_PROVIDERS = [
// Codex CLI sends this bare virtual model for auto-review — keep it on OAuth Codex (#1398).
[/^codex-auto-review$/, "codex"],
[/^claude-/, "anthropic"],
[/^gemini-/, "gemini"],
[/^gpt-/, "openai"],

View File

@@ -10,6 +10,24 @@ const signatureKv = makeKv(SCOPE);
const memorySignatures = new Map();
let pruneCounter = 0;
/**
* Model family that produced / will consume a signature. Antigravity serves Gemini and Claude
* models behind the same API, and each backend only accepts its own signatures: a Claude
* signature replayed to Gemini fails with 400 "Corrupted thought signature." (and vice versa).
*/
export function signatureFamily(model) {
const m = typeof model === "string" ? model.toLowerCase() : "";
if (!m) return null;
if (m.includes("claude")) return "claude";
if (m.includes("gemini")) return "gemini";
return m;
}
// Entries stored before families were recorded (no `family`) stay usable for any model.
function isCompatible(entry, family) {
return !entry.family || !family || entry.family === family;
}
function pruneMemoryExpired() {
const now = Date.now();
for (const [key, value] of memorySignatures.entries()) {
@@ -62,13 +80,15 @@ async function maybePrunePersisted() {
}
/**
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async)
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async).
* `model` is the model that produced the signature; lookups for another model family skip it.
*/
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) {
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null, model = null) {
if (typeof toolCallId !== "string" || !toolCallId) return;
if (typeof signature !== "string" || !signature) return;
const now = Date.now();
const family = signatureFamily(model);
pruneMemoryExpired();
const keys = [];
@@ -80,12 +100,14 @@ export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = n
for (const k of keys) {
memorySignatures.set(k, {
signature,
family,
expiresAt: now + MEMORY_TTL_MS,
});
// Async persist to SQLite kv table without blocking
signatureKv.set(k, {
signature,
family,
createdAt: now,
expiresAt: now + PERSISTED_TTL_MS,
}).catch(() => {});
@@ -95,23 +117,25 @@ export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = n
}
/**
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback)
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback).
* `model` is the target model; signatures produced by another model family are ignored.
*/
export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
export async function getGeminiThoughtSignature(toolCallId, sessionId = null, model = null) {
if (typeof toolCallId !== "string" || !toolCallId) return null;
const family = signatureFamily(model);
pruneMemoryExpired();
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionEntry = memorySignatures.get(sessionKey);
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) {
return sessionEntry.signature;
}
}
const entry = memorySignatures.get(toolCallId);
if (entry && entry.expiresAt > Date.now()) {
if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) {
return entry.signature;
}
@@ -119,9 +143,10 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionRow = await signatureKv.get(sessionKey);
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) {
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now()) && isCompatible(sessionRow, family)) {
memorySignatures.set(sessionKey, {
signature: sessionRow.signature,
family: sessionRow.family || null,
expiresAt: Date.now() + MEMORY_TTL_MS,
});
return sessionRow.signature;
@@ -134,8 +159,10 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
signatureKv.remove(toolCallId).catch(() => {});
return null;
}
if (!isCompatible(row, family)) return null;
memorySignatures.set(toolCallId, {
signature: row.signature,
family: row.family || null,
expiresAt: Date.now() + MEMORY_TTL_MS,
});
return row.signature;
@@ -148,22 +175,24 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
}
/**
* Synchronous get from RAM cache only (for sync translators)
* Synchronous get from RAM cache only (for sync translators).
* `model` is the target model; signatures produced by another model family are ignored.
*/
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) {
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null, model = null) {
if (typeof toolCallId !== "string" || !toolCallId) return null;
const family = signatureFamily(model);
pruneMemoryExpired();
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionEntry = memorySignatures.get(sessionKey);
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) {
return sessionEntry.signature;
}
}
const entry = memorySignatures.get(toolCallId);
if (entry && entry.expiresAt > Date.now()) {
if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) {
return entry.signature;
}
return null;

View File

@@ -22,6 +22,7 @@ import { getZedUsage } from "./usage/zed.js";
import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js";
import { resolveQoderCredentials } from "./qoderModels.js";
import { getGlmUsage } from "./usage/glm.js";
import { getCommandCodeUsage } from "./usage/commandcode.js";
import {
getIflowUsage,
getOllamaUsage,
@@ -66,6 +67,7 @@ const USAGE_HANDLERS = {
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
"xiaomi-mimo": (c) => getXiaomiMimoUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
};
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {

View File

@@ -1,207 +1,134 @@
/**
* CommandCode usage handler
*
* Mirrors the official command-code CLI /usage command: it calls the alpha API
* to surface the 5-hour + weekly usage windows, the subscription plan, and the
* credits consumed in the current billing period.
*
* GET /alpha/whoami → org.id (org-scoped billing; null for personal)
* GET /alpha/billing/credits → { credits: { monthlyCredits, purchasedCredits,
* freeCredits }, windowLimits: { fiveHour, weekly } }
* GET /alpha/billing/subscriptions → { data: { planId, currentPeriodStart, ... } }
* GET /alpha/usage/summary?since= → period token/cost totals
*
* The CLI fetches whoami first (for orgId), then credits + subscription in
* parallel, then the summary with since = currentPeriodStart. We keep the same
* order/dependencies: window limits live on credits, and the plan period start
* determines the summary window.
* Command Code usage — billing credits + 5h/weekly rate windows.
* Mirrors ~/cc-usage.mjs: whoami → credits + subscriptions.
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U, parseResetTime } from "./shared.js";
import { parseResetTime, toFiniteNumber } from "./shared.js";
const USAGE = U("commandcode");
const BASE = USAGE.baseUrl || "https://api.commandcode.ai";
const WHOAMI_URL = BASE + (USAGE.whoamiUrl || "/alpha/whoami");
const CREDITS_URL = BASE + (USAGE.creditsUrl || "/alpha/billing/credits");
const SUBSCRIPTIONS_URL =
BASE + (USAGE.subscriptionsUrl || "/alpha/billing/subscriptions");
const SUMMARY_URL = BASE + (USAGE.summaryUrl || "/alpha/usage/summary");
const BASE = (process.env.COMMAND_CODE_API_BASE_URL || "https://api.commandcode.ai").replace(/\/$/, "");
function buildHeaders(token) {
return {
Authorization: `Bearer ${token}`,
Accept: "application/json",
};
const PLAN_NAMES = {
"individual-go": "Go",
"individual-goat": "GOAT",
"individual-pro": "Pro",
"individual-pro-v1": "Pro",
"individual-provider": "Provider",
"individual-max": "Max",
"individual-ultra": "Ultra",
"teams-pro": "Teams Pro",
};
const PLAN_CAPS = {
"individual-go": 10,
"individual-goat": 70,
"individual-pro": 30,
"individual-pro-v1": 80,
"individual-provider": 15,
"individual-max": 150,
"individual-ultra": 300,
"teams-pro": 40,
};
function qs(route, params) {
const s = new URLSearchParams(
Object.entries(params || {}).filter(([, v]) => v != null),
).toString();
return s ? `${route}?${s}` : route;
}
/** Build a normalized quota row. `unit` is "$" — the API reports currency credits. */
function makeQuota({ used, total, resetAt, unlimited = false, unit = "$" }) {
const safeTotal = Math.max(0, Number(total) || 0);
const safeUsed = Math.max(0, Number(used) || 0);
if (unlimited || safeTotal === 0) {
return {
used: safeUsed,
total: 0,
remainingPercentage: unlimited ? 100 : 0,
resetAt: resetAt || null,
unit,
unlimited: true,
};
}
const remaining = Math.max(0, safeTotal - safeUsed);
const remainingPercentage = (remaining / safeTotal) * 100;
return {
used: safeUsed,
total: safeTotal,
remainingPercentage,
resetAt: resetAt || null,
unit,
unlimited: false,
};
function windowQuota(win) {
if (!win || typeof win !== "object") return null;
const used = toFiniteNumber(win.used, 0);
const total = toFiniteNumber(win.cap, 0);
if (total <= 0 && used <= 0) return null;
return {
used,
total,
remaining: Math.max(0, total - used),
unlimited: false,
resetAt: parseResetTime(win.resetAt),
};
}
/**
* @param {string} apiKey - commandcode API key (user_...)
* @param {string|null|undefined} apiKey
* @param {object|null} proxyOptions
*/
export async function getCommandCodeUsage(apiKey, proxyOptions = null) {
if (!apiKey) {
return { message: "CommandCode credential not available." };
}
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
return { message: "Command Code API key not available. Add a key to view usage." };
}
const headers = buildHeaders(apiKey);
const headers = {
Authorization: `Bearer ${apiKey.trim()}`,
Accept: "application/json",
};
try {
// whoami resolves the org id (billing is org-scoped; null for personal).
const whoamiRes = await proxyAwareFetch(
WHOAMI_URL,
{ method: "GET", headers },
proxyOptions,
);
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
return { message: "CommandCode credential invalid or expired." };
}
if (!whoamiRes.ok) {
return { message: `CommandCode whoami API error (${whoamiRes.status}).` };
}
const whoami = await whoamiRes.json().catch(() => null);
const orgId = whoami?.org?.id ?? null;
const get = async (route) => {
const response = await proxyAwareFetch(
BASE + route,
{ method: "GET", headers },
proxyOptions,
);
return response;
};
const orgQuery = orgId ? `?orgId=${encodeURIComponent(orgId)}` : "";
try {
const whoamiRes = await get(qs("/alpha/whoami", { limits: "1" }));
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." };
}
if (!whoamiRes.ok) {
return { plan: "Command Code", message: `Command Code usage API error (${whoamiRes.status})` };
}
const whoami = await whoamiRes.json().catch(() => ({}));
const orgId = whoami?.org?.id ?? null;
const [creditsRes, subsRes] = await Promise.all([
proxyAwareFetch(
CREDITS_URL + orgQuery,
{ method: "GET", headers },
proxyOptions,
),
proxyAwareFetch(
SUBSCRIPTIONS_URL + orgQuery,
{ method: "GET", headers },
proxyOptions,
),
]);
const [creditsRes, subsRes] = await Promise.all([
get(qs("/alpha/billing/credits", { orgId })),
get(qs("/alpha/billing/subscriptions", { orgId })),
]);
if (
creditsRes.status === 401 ||
creditsRes.status === 403 ||
subsRes.status === 401 ||
subsRes.status === 403
) {
return { message: "CommandCode credential invalid or expired." };
}
if (!creditsRes.ok) {
return {
message: `CommandCode credits API error (${creditsRes.status}).`,
};
}
if (creditsRes.status === 401 || creditsRes.status === 403 || subsRes.status === 401 || subsRes.status === 403) {
return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." };
}
if (!creditsRes.ok) {
return { plan: "Command Code", message: `Command Code credits API error (${creditsRes.status})` };
}
if (!subsRes.ok) {
return { plan: "Command Code", message: `Command Code subscriptions API error (${subsRes.status})` };
}
const credits = await creditsRes.json().catch(() => null);
const subs = await subsRes.json().catch(() => null);
const creditsBody = await creditsRes.json().catch(() => ({}));
const subsBody = await subsRes.json().catch(() => ({}));
const planId = subsBody?.data?.planId ?? null;
const plan = (planId && PLAN_NAMES[planId]) || planId || "Command Code";
const cap = planId ? (PLAN_CAPS[planId] || 0) : 0;
const c = creditsBody?.credits || {};
const remaining =
toFiniteNumber(c.monthlyCredits, 0) +
toFiniteNumber(c.purchasedCredits, 0) +
toFiniteNumber(c.freeCredits, 0);
const used = cap > 0 ? Math.max(0, cap - remaining) : 0;
const total = cap > 0 ? cap : remaining;
const subData = subs?.data;
const planId = subData?.planId ?? null;
const periodStart = subData?.currentPeriodStart ?? null;
const quotas = {};
quotas.Credits = {
used,
total,
remaining,
unlimited: cap <= 0,
resetAt: parseResetTime(subsBody?.data?.currentPeriodEnd),
};
// Summary needs `since`; the CLI falls back to first-of-month when the
// subscription period start is unavailable.
const since = periodStart || firstOfMonth();
const summaryRes = await proxyAwareFetch(
`${SUMMARY_URL}?since=${encodeURIComponent(since)}`,
{ method: "GET", headers },
proxyOptions,
);
const summary = summaryRes.ok
? await summaryRes.json().catch(() => null)
: null;
const fiveHour = windowQuota(creditsBody?.windowLimits?.fiveHour);
if (fiveHour) quotas["Session (5h)"] = fiveHour;
const weekly = windowQuota(creditsBody?.windowLimits?.weekly);
if (weekly) quotas.Weekly = weekly;
const quotas = {};
const windowLimits = credits?.windowLimits || {};
const fiveHour = windowLimits.fiveHour;
if (fiveHour && Number(fiveHour.cap) > 0) {
quotas["5-hour window"] = makeQuota({
used: fiveHour.used,
total: fiveHour.cap,
resetAt: parseResetTime(fiveHour.resetAt),
});
}
const weekly = windowLimits.weekly;
if (weekly && Number(weekly.cap) > 0) {
quotas["Weekly window"] = makeQuota({
used: weekly.used,
total: weekly.cap,
resetAt: parseResetTime(weekly.resetAt),
});
}
// The credits API reports remaining balances (monthly/purchased/free),
// not a total. The official CLI renders the monthly line as
// `used = summary.totalCost`, `total = totalCost + remaining` — i.e.
// the plan ceiling is the sum of what was consumed and what is left.
const monthlyUsed =
typeof summary?.totalCredits === "number"
? summary.totalCredits
: typeof summary?.totalCost === "number"
? summary.totalCost
: 0;
const creditsObj = credits?.credits || {};
const remaining =
Math.max(0, Number(creditsObj.monthlyCredits) || 0) +
Math.max(0, Number(creditsObj.purchasedCredits) || 0) +
Math.max(0, Number(creditsObj.freeCredits) || 0);
const monthlyTotal = monthlyUsed + remaining;
if (monthlyTotal > 0 || monthlyUsed > 0) {
quotas["Monthly credits"] = makeQuota({
used: monthlyUsed,
total: monthlyTotal,
resetAt: periodStart ? undefined : null,
});
}
if (Object.keys(quotas).length === 0) {
return {
plan: planId || "CommandCode",
message: "CommandCode connected, but no quota was reported.",
quotas: {},
};
}
return {
plan: planId || "CommandCode",
quotas,
periodBasis: summary?.periodBasis || "billing-period",
};
} catch (error) {
return { message: `CommandCode usage error: ${error.message}` };
}
}
function firstOfMonth() {
const now = new Date();
return new Date(now.getFullYear(), now.getMonth(), 1).toISOString();
return { plan, quotas };
} catch (error) {
return { message: `Command Code error: ${error.message}` };
}
}

View File

@@ -91,14 +91,15 @@ export async function getDeepseekUsage(apiKey = null, proxyOptions = null) {
const quotas = {};
for (const b of balances) {
const total = Math.max(0, b.totalBalance);
// Credit pot: show full remaining against current balance; never set absolute
// `remaining` — QuotaTable treats it as a 0–100 percentage.
// Credit balance: show as "Credit: $X.XX USD" not a usage quota
quotas[`Balance (${b.currency})`] = {
used: 0,
total,
remainingPercentage: total > 0 ? 100 : 0,
resetAt: null,
unlimited: total > 0,
unlimited: false,
isCreditBalance: true,
currency: b.currency,
};
}

View File

@@ -112,7 +112,14 @@ export function parseZedCallbackPayload(input) {
url = new URL(raw);
} catch {
try {
url = new URL(`http://127.0.0.1/?${raw.replace(/^\?/, "")}`);
// Accept pathname+query (what the local proxy forwards, e.g.
// "/?user_id=..&access_token=.." or "/callback?.."), a bare query,
// or a lone query string. Only the query part is parsed — a leading
// path must never become part of the first parameter name.
const query = raw.includes("?")
? raw.slice(raw.indexOf("?") + 1)
: raw.replace(/^\?/, "");
url = new URL(`http://127.0.0.1/?${query}`);
} catch {
throw new Error("Invalid Zed callback URL");
}
@@ -134,6 +141,10 @@ export function parseZedCallbackPayload(input) {
export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier) {
const privateKey = decodeZedPrivateKeyVerifier(privateKeyVerifier);
const encrypted = Buffer.from(String(encryptedAccessToken), "base64url");
const fail = (oaepError) => {
const message = oaepError instanceof Error ? oaepError.message : String(oaepError);
throw new Error(`Failed to decrypt Zed access token: ${message}`);
};
try {
return crypto
.privateDecrypt(
@@ -143,15 +154,21 @@ export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier)
.toString("utf8");
} catch (oaepError) {
try {
return crypto
const text = crypto
.privateDecrypt(
{ key: privateKey, padding: crypto.constants.RSA_PKCS1_PADDING },
encrypted,
)
.toString("utf8");
} catch {
const message = oaepError instanceof Error ? oaepError.message : String(oaepError);
throw new Error(`Failed to decrypt Zed access token: ${message}`);
// PKCS#1 v1.5 unpadding is not integrity-checked: a wrong-key decrypt
// can "succeed" with garbage bytes instead of throwing. Replacement
// characters prove the output is not the real UTF-8 token — fail loudly
// rather than storing garbage as a credential.
if (text.includes("<22>")) fail(oaepError);
return text;
} catch (err) {
if (err.message.startsWith("Failed to decrypt Zed access token")) throw err;
fail(oaepError);
}
}
}
@@ -280,6 +297,7 @@ export async function fetchZedLlmToken(credentials, options = {}) {
body: JSON.stringify({ organization_id: organizationId }),
signal: options.signal ?? undefined,
},
options.proxyOptions ?? null,
);
const token =
typeof data?.token === "string" ? data.token : data?.token?.[0] || data?.token?.value;

View File

@@ -5,6 +5,20 @@ import {
} from "../../config/kiroConstants.js";
const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/;
/**
* Kiro rejects user turns with empty `content`, so a turn that only carries
* tool results needs placeholder text. It must not read like a user
* instruction: with "continue", models answer the word itself ("Nothing in
* progress to continue") and drop the task they were in the middle of.
*/
export const KIRO_TOOL_RESULTS_PLACEHOLDER = "Tool results provided.";
export const KIRO_EMPTY_USER_PLACEHOLDER = "continue";
/** Placeholder content for a user turn with no text of its own. */
export function kiroEmptyUserContent(hasToolResults) {
return hasToolResults ? KIRO_TOOL_RESULTS_PLACEHOLDER : KIRO_EMPTY_USER_PLACEHOLDER;
}
const TOOL_NAME_PATTERN = /[^a-zA-Z0-9_-]/g;
function clone(value) {
@@ -34,7 +48,6 @@ function uniqueName(rawName, index, usedNames) {
const cleaned = String(rawName || "")
.trim()
.replace(TOOL_NAME_PATTERN, "_")
.replace(/_+/g, "_")
.replace(/^_+|_+$/g, "");
const base = trimCodePoints(cleaned || `tool_${index + 1}`, KIRO_TOOL_NAME_MAX_LENGTH);
let candidate = base;
@@ -174,7 +187,8 @@ function normalizeTurns(history, currentMessage, modelId) {
for (const turn of turns) {
if (turn.userInputMessage) {
turn.userInputMessage.content = text(turn.userInputMessage.content).trim() || "continue";
turn.userInputMessage.content = text(turn.userInputMessage.content).trim()
|| kiroEmptyUserContent(turn.userInputMessage.userInputMessageContext?.toolResults?.length > 0);
turn.userInputMessage.modelId ||= modelId;
if (turn.userInputMessage.userInputMessageContext?.tools) {
delete turn.userInputMessage.userInputMessageContext.tools;

View File

@@ -8,6 +8,7 @@ import { fetchImageAsBase64, parseDataUri } from "./image.js";
const TARGETS_NEED_BASE64 = new Set([
FORMATS.GEMINI, FORMATS.GEMINI_CLI, FORMATS.VERTEX,
FORMATS.ANTIGRAVITY, FORMATS.OLLAMA, FORMATS.KIRO,
FORMATS.COMMANDCODE,
]);
function isRemoteUrl(url) {

View File

@@ -19,6 +19,7 @@ const FORMAT_TO_NATIVE = {
vertex: "gemini-budget",
antigravity: "gemini-budget",
kiro: "kiro",
commandcode: "commandcode",
};
// Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent).
@@ -108,6 +109,7 @@ export const captureThinking = extractThinking;
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
function resolveFormat(targetFormat, model, provider) {
if (targetFormat === "commandcode") return "commandcode";
const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null;
if (providerFmt) return providerFmt;
const caps = getCapabilitiesForModel(provider, model);
@@ -223,10 +225,14 @@ function stripAll(body) {
delete body.output_config;
if (body.generationConfig) delete body.generationConfig.thinkingConfig;
if (body.request?.generationConfig) delete body.request.generationConfig.thinkingConfig;
if (body.params && typeof body.params === "object") {
delete body.params.reasoning_effort;
delete body.params.thinking;
}
}
// Apply unified thinking config to body in the resolved provider-native format.
function applyFormat(fmt, body, cfg, caps, supportedLevels) {
function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
const none = cfg.mode === "none";
const canDisable = caps.thinkingCanDisable !== false;
// Model cannot disable thinking → clamp "none" to minimal effort instead.
@@ -243,7 +249,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
// Models that can disable thinking need the explicit adaptive switch.
// Permanently adaptive models such as Fable 5.1 accept effort directly.
if (canDisable) body.thinking = { type: "adaptive" };
if (canDisable) body.thinking = { type: "adaptive", ...(display ? { display } : {}) };
else delete body.thinking;
const level = toLevel(eff);
body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level };
@@ -252,7 +258,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
case "claude-budget": {
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
const budget = toBudget(eff, caps.thinkingRange);
body.thinking = budget === -1 ? { type: "enabled" } : { type: "enabled", budget_tokens: budget || 8192 };
body.thinking = budget === -1 ? { type: "enabled", ...(display ? { display } : {}) } : { type: "enabled", budget_tokens: budget || 8192, ...(display ? { display } : {}) };
break;
}
case "gemini-level": {
@@ -336,6 +342,17 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
case "kiro":
// Kiro thinking handled via system-tag injection in openai-to-kiro.js; no body field here.
break;
case "commandcode": {
// Native CLI sends reasoning_effort inside params of the /alpha/generate envelope.
if (!body.params || typeof body.params !== "object") body.params = {};
if (none && canDisable) {
delete body.params.reasoning_effort;
break;
}
const level = toLevel(eff);
if (level) body.params.reasoning_effort = level;
break;
}
default:
break;
}
@@ -361,7 +378,10 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
const fmt = resolveFormat(targetFormat, cleanModel, provider);
const supportedLevels = getThinkingLevels(provider, cleanModel);
// Anthropic's `display` (summarized | omitted) decides whether thinking text
// comes back at all; keep what the client asked for instead of resetting it.
const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined;
stripAll(body);
applyFormat(fmt, body, cfg, caps, supportedLevels);
applyFormat(fmt, body, cfg, caps, supportedLevels, display);
return body;
}

View File

@@ -415,6 +415,28 @@ export function anchorClaudeCache(body) {
// - Add thinking block for Anthropic endpoint (provider === "claude")
// - Fix tool_use/tool_result ordering
// - Apply cloaking (billing header + fake user ID) for OAuth tokens
export function hoistToolResultImages(body) {
if (!Array.isArray(body?.messages)) return body;
let touched = false;
const messages = body.messages.map((msg) => {
if (msg?.role !== ROLE.USER || !Array.isArray(msg.content)) return msg;
const hoisted = [];
const content = msg.content.map((block) => {
if (block?.type !== CLAUDE_BLOCK.TOOL_RESULT || !Array.isArray(block.content)) return block;
const images = block.content.filter((c) => c?.type === CLAUDE_BLOCK.IMAGE);
if (!images.length) return block;
const rest = block.content.filter((c) => c?.type !== CLAUDE_BLOCK.IMAGE);
hoisted.push({ type: CLAUDE_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` }, ...images);
return { ...block, content: rest.length ? rest : [{ type: CLAUDE_BLOCK.TEXT, text: "(image attached below)" }] };
});
if (!hoisted.length) return msg;
touched = true;
// tool_result blocks must lead a user message; the hoisted image follows them.
return { ...msg, content: [...content, ...hoisted] };
});
return touched ? { ...body, messages } : body;
}
export function prepareClaudeRequest(body, provider = null, apiKey = null, connectionId = null, rawHeaders = null, sessionId = null) {
// quirk: MiniMax's Claude-compatible endpoint rejects Anthropic's output_config (400 invalid params)
if (PROVIDERS[provider]?.quirks?.dropOutputConfig) {
@@ -608,6 +630,14 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
}
}
// Anthropic itself reads images inside tool_result; other Anthropic-compatible
// endpoints (OpenCode Go, Kimi, DeepSeek, GLM, MiniMax) accept image blocks
// only as user content and silently drop them inside a tool result. Move a
// tool's screenshot out of the result and into the same user turn.
if (provider !== "claude" && !provider?.startsWith("anthropic-compatible")) {
body = hoistToolResultImages(body);
}
// Apply cloaking for OAuth tokens (billing header + fake user ID)
// session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency
if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) {

View File

@@ -34,6 +34,7 @@ import { ROLE, CLAUDE_BLOCK } from "../schema/index.js";
import {
canonicalizeKiroConversation,
normalizeKiroToolSpecs,
kiroEmptyUserContent,
} from "../concerns/kiroConversation.js";
/**
@@ -53,7 +54,8 @@ function convertClaudeMessagesToKiro(messages, model) {
const flushPending = () => {
if (currentRole === ROLE.USER) {
const content = pendingUserContent.join("\n\n").trim() || "continue";
const content = pendingUserContent.join("\n\n").trim()
|| kiroEmptyUserContent(pendingToolResults.length > 0);
const userMsg = { userInputMessage: { content, modelId: model } };
if (pendingImages.length > 0) {
@@ -97,11 +99,21 @@ function convertClaudeMessagesToKiro(messages, model) {
if (typeof block.content === "string") {
resultContent = block.content;
} else if (Array.isArray(block.content)) {
// Images a tool returned (screenshots) ride along as user images;
// Kiro tool results are text-only.
let hasImage = false;
for (const c of block.content) {
if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") {
hasImage = true;
const imageType = c.source.media_type || DEFAULT_IMAGE_MIME;
pendingImages.push({ format: imageType.split("/")[1] || imageType, source: { bytes: c.source.data } });
}
}
resultContent =
block.content
.filter((c) => c.type === CLAUDE_BLOCK.TEXT)
.map((c) => c.text)
.join("\n") || JSON.stringify(block.content);
.join("\n") || (hasImage ? "(image attached)" : JSON.stringify(block.content));
} else if (block.content) {
resultContent = JSON.stringify(block.content);
}
@@ -341,6 +353,13 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
enumerable: false,
});
// Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the
// reverse map so tool calls stream back under the client's own names.
const restoredToolNames = new Map();
for (const [original, sanitized] of nameMap) {
if (original !== sanitized) restoredToolNames.set(sanitized, original);
}
if (restoredToolNames.size) payload._toolNameMap = restoredToolNames;
return payload;
}

View File

@@ -196,25 +196,41 @@ function convertClaudeMessage(msg) {
});
break;
case CLAUDE_BLOCK.TOOL_RESULT:
case CLAUDE_BLOCK.TOOL_RESULT: {
let resultContent = "";
const resultImages = [];
if (typeof block.content === "string") {
resultContent = block.content;
} else if (Array.isArray(block.content)) {
resultContent = block.content
.filter(c => c.type === CLAUDE_BLOCK.TEXT)
.map(c => c.text)
.join("\n") || JSON.stringify(block.content);
for (const c of block.content) {
if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") {
resultImages.push({
type: OPENAI_BLOCK.IMAGE_URL,
image_url: { url: encodeDataUri(c.source.media_type, c.source.data) }
});
}
}
const textOnly = block.content.filter(c => c?.type === CLAUDE_BLOCK.TEXT);
resultContent = textOnly.map(c => c.text).join("\n")
|| (resultImages.length ? "" : JSON.stringify(block.content));
} else if (block.content) {
resultContent = JSON.stringify(block.content);
}
toolResults.push({
role: ROLE.TOOL,
tool_call_id: block.tool_use_id,
content: resultContent
});
// The OpenAI tool role is text-only, so a screenshot or any other image a
// tool returned would otherwise vanish. Hand it to the model in the user
// turn that follows the tool messages, tagged with the call it came from.
if (resultImages.length) {
parts.push({ type: OPENAI_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` });
parts.push(...resultImages);
}
break;
}
}
}

View File

@@ -5,6 +5,7 @@
* - params.system: STRING at top level (Anthropic-style; system messages NOT allowed in messages[])
* - params.messages[*].role ∈ {"user","assistant","tool"}
* - params.messages[*].content: Array of content blocks (NEVER a string)
* - image_url / image source → {type:"image", image:"data:...;base64,...", mimeType}
* - tool_use blocks (assistant): {type:"tool-call", toolCallId, toolName, input}
* - tool_result blocks (role=user): {type:"tool-result", toolCallId, toolName, output}
* - tools[*]: Anthropic plain {name, description, input_schema}
@@ -12,10 +13,9 @@
import { register } from "../index.js";
import { FORMATS } from "../formats.js";
import { randomUUID } from "crypto";
import { ROLE, OPENAI_BLOCK } from "../schema/index.js";
import { DEFAULT_IMAGE_MIME } from "../schema/index.js";
import { parseDataUri } from "../concerns/image.js";
import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js";
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
import { parseDataUri, encodeDataUri } from "../concerns/image.js";
function flattenText(content) {
if (content == null) return "";
@@ -32,6 +32,58 @@ function flattenText(content) {
return String(content);
}
function toNativeImageBlock(part) {
if (!part || typeof part !== "object") return null;
if (part.type === OPENAI_BLOCK.IMAGE_URL) {
const url = typeof part.image_url === "string" ? part.image_url : part.image_url?.url;
if (!url) return null;
const parsed = parseDataUri(url);
if (parsed) {
return {
type: OPENAI_BLOCK.IMAGE,
image: encodeDataUri(parsed.mimeType, parsed.base64),
mimeType: parsed.mimeType,
};
}
if (typeof url === "string" && (url.startsWith("http://") || url.startsWith("https://"))) {
return {
type: OPENAI_BLOCK.IMAGE,
image: url,
};
}
return null;
}
if (part.type === OPENAI_BLOCK.IMAGE || part.type === CLAUDE_BLOCK.IMAGE) {
if (typeof part.image === "string" && part.image.startsWith("data:")) {
const parsed = parseDataUri(part.image);
return {
type: OPENAI_BLOCK.IMAGE,
image: part.image,
mimeType: part.mimeType || parsed?.mimeType || "image/png",
};
}
if (typeof part.image === "string" && (part.image.startsWith("http://") || part.image.startsWith("https://"))) {
return {
type: OPENAI_BLOCK.IMAGE,
image: part.image,
};
}
const source = part.source;
if (source?.type === "base64" && typeof source.data === "string") {
const mime = source.media_type || "image/png";
return {
type: OPENAI_BLOCK.IMAGE,
image: encodeDataUri(mime, source.data),
mimeType: mime,
};
}
}
return null;
}
function toContentBlocks(content) {
if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }];
if (typeof content === "string")
@@ -44,30 +96,12 @@ function toContentBlocks(content) {
} else if (part && typeof part === "object") {
if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") {
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
} else if (
part.type === OPENAI_BLOCK.IMAGE_URL ||
part.type === OPENAI_BLOCK.IMAGE
) {
// CommandCode `/alpha/generate` accepts {type:"image", image:"<data URI | url>"} —
// same shape the official command-code CLI sends (verified from CLI source).
const src = part.source;
let raw = part.image_url?.url || src?.data || src?.url || "";
let parsed = parseDataUri(raw);
if (!parsed && src?.type === "base64" && src?.data) {
// Claude-style base64 source without a data-URI prefix → wrap it.
raw = `data:${src.media_type || DEFAULT_IMAGE_MIME};base64,${src.data}`;
parsed = parseDataUri(raw);
} else {
const image = toNativeImageBlock(part);
if (image) blocks.push(image);
else if (typeof part.text === "string") {
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
}
if (parsed) {
blocks.push({
type: "image",
image: `data:${parsed.mimeType};base64,${parsed.base64}`,
});
} else if (raw) {
blocks.push({ type: "image", image: raw });
}
} else if (typeof part.text === "string") {
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
}
}
}
@@ -119,6 +153,10 @@ function convertMessages(messages = []) {
if (role === ROLE.ASSISTANT) {
const blocks = [];
const rc = m.reasoning_content || m.thought || m.reasoning;
if (rc || (Array.isArray(m.tool_calls) && m.tool_calls.length > 0)) {
blocks.push({ type: "reasoning", text: rc || " " });
}
const text = flattenText(m.content);
if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
if (Array.isArray(m.tool_calls)) {

View File

@@ -129,7 +129,7 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
if (tc.type !== OPENAI_BLOCK.FUNCTION) continue;
const args = tryParseJSON(tc.function?.arguments || "{}");
const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null;
const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId, model) : null;
// First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig
const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined);
firstFunctionCallSeen = true;
@@ -341,7 +341,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
if (block.type === CLAUDE_BLOCK.TEXT) {
parts.push({ text: block.text });
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null;
const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId, model) : null;
const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined);
firstToolUseSeen = true;

View File

@@ -23,6 +23,7 @@ import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js";
import {
canonicalizeKiroConversation,
normalizeKiroToolSpecs,
kiroEmptyUserContent,
} from "../concerns/kiroConversation.js";
/**
@@ -51,7 +52,8 @@ function convertMessages(messages, model) {
const flushPending = () => {
if (currentRole === "user") {
const content = pendingUserContent.join("\n\n").trim() || "continue";
const content = pendingUserContent.join("\n\n").trim()
|| kiroEmptyUserContent(pendingToolResults.length > 0);
const userMsg = {
userInputMessage: {
content: content,
@@ -434,6 +436,13 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
enumerable: false
});
// Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the
// reverse map so tool calls stream back under the client's own names.
const restoredToolNames = new Map();
for (const [original, sanitized] of nameMap) {
if (original !== sanitized) restoredToolNames.set(sanitized, original);
}
if (restoredToolNames.size) payload._toolNameMap = restoredToolNames;
return payload;
}

View File

@@ -183,27 +183,12 @@ export function commandCodeToOpenAIResponse(chunk, state) {
break;
}
case "error": {
// Terminal upstream failure (AI SDK v5 error event) — NOT content. Emit an
// OpenAI-shaped error chunk (chunk.error) so downstream — parseSSEToOpenAIResponse
// for non-streaming, OpenAI SDK clients for streaming — treats the request as
// failed instead of surfacing fake success content like "[CommandCode error: ...]".
state.finishReason = OPENAI_FINISH.STOP;
const errVal = event.error ?? event.message ?? "unknown";
const errStr =
typeof errVal === "string"
? errVal
: typeof errVal?.message === "string"
? errVal.message
: JSON.stringify(errVal);
const errType =
typeof errVal === "string"
? "upstream_error"
: errVal?.type || "upstream_error";
const errChunk = makeChunk(state, {});
errChunk.error = { message: errStr, type: errType };
out.push(errChunk);
out.push(makeChunk(state, {}, OPENAI_FINISH.STOP));
break;
typeof errVal === "string" ? errVal : JSON.stringify(errVal);
// Mid-stream error: throw rather than emitting as fake content with finish_reason: "stop"
// This ensures the downstream stream handler marks the stream as errored/aborted.
throw new Error(`[CommandCode error: ${errStr}]`);
}
// Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end,
// provider-metadata, message-metadata, etc. They carry no client-visible content.

View File

@@ -22,7 +22,7 @@ function emitFunctionCall(functionCall, state, signature = null) {
const toolCallIndex = state.functionIndex++;
const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`;
if (signature) {
storeGeminiThoughtSignature(callId, signature, state.sessionId);
storeGeminiThoughtSignature(callId, signature, state.sessionId, state.model);
}
const toolCall = {
id: callId,
@@ -52,7 +52,7 @@ export function geminiToOpenAIResponse(chunk, state) {
// Initialize state
if (!state.messageId) {
state.messageId = response.responseId || `msg_${Date.now()}`;
state.model = response.modelVersion || "gemini";
state.model = response.modelVersion || state.model || "gemini";
state.functionIndex = 0;
state.geminiToolCallCount = 0;
results.push(buildChunk(chunkMeta(state), { role: ROLE.ASSISTANT }, null));

View File

@@ -46,6 +46,14 @@ function convertFinishReason(reason) {
* Convert one OpenAI-format chunk (from KiroExecutor) into Claude SSE events.
* Returns an array of Claude events, or null when the chunk yields nothing.
*/
// Kiro only accepts sanitized tool names; the request translator leaves the
// reverse map on the stream state so calls come back under the client's names.
function restoreToolName(stateOrData, name) {
const raw = name || "";
const map = stateOrData?.toolNameMap || stateOrData?._toolNameMap;
return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw;
}
export function kiroToClaudeResponse(chunk, state) {
// KiroExecutor emits chat.completion.chunk objects; tolerate string chunks
// by attempting a parse (defensive — the direct path is always objects).
@@ -161,7 +169,7 @@ export function kiroToClaudeResponse(chunk, state) {
const toolBlockIndex = state.nextBlockIndex++;
state.toolCalls.set(idx, {
id: tc.id,
name: tc.function?.name || "",
name: restoreToolName(state, tc.function?.name),
blockIndex: toolBlockIndex,
});
results.push({
@@ -170,7 +178,7 @@ export function kiroToClaudeResponse(chunk, state) {
content_block: {
type: "tool_use",
id: tc.id,
name: tc.function?.name || "",
name: restoreToolName(state, tc.function?.name),
input: {},
},
});
@@ -246,7 +254,7 @@ export function kiroToClaudeNonStreaming(data) {
content.push({
type: "tool_use",
id: tc.id || `toolu_${Date.now()}`,
name: tc.function?.name || "",
name: restoreToolName(data, tc.function?.name),
input,
});
}

View File

@@ -20,13 +20,38 @@ function chunkMeta(state) {
* Parse Kiro SSE event and convert to OpenAI format
* Kiro events: assistantResponseEvent, codeEvent, supplementaryWebLinksEvent, etc.
*/
// Kiro only accepts sanitized tool names; the request translator leaves the
// reverse map on the stream state so calls come back under the client's names.
function restoreToolName(state, name) {
const raw = name || "";
const map = state?.toolNameMap;
return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw;
}
export function kiroToOpenAIResponse(chunk, state) {
if (!chunk) return null;
// If chunk is already in OpenAI format (from executor transform), return as-is
// If chunk is already in OpenAI format (from executor transform), return it
// with the client's tool names restored.
if (chunk.object === "chat.completion.chunk" && chunk.choices) {
return chunk;
if (!state?.toolNameMap?.size) return chunk;
return {
...chunk,
choices: chunk.choices.map((choice) => {
const calls = choice?.delta?.tool_calls;
if (!Array.isArray(calls)) return choice;
return {
...choice,
delta: {
...choice.delta,
tool_calls: calls.map((tc) => tc?.function?.name
? { ...tc, function: { ...tc.function, name: restoreToolName(state, tc.function.name) } }
: tc),
},
};
}),
};
}
// Handle string chunk (raw SSE data)
@@ -109,7 +134,7 @@ export function kiroToOpenAIResponse(chunk, state) {
state.hadToolUse = true;
const toolUse = data.toolUseEvent || data;
const toolCallId = toolUse.toolUseId || fallbackToolCallId();
const toolName = toolUse.name || "";
const toolName = restoreToolName(state, toolUse.name);
const toolInput = toolUse.input || {};
const openaiChunk = buildChunk(chunkMeta(state), {

View File

@@ -95,6 +95,9 @@ export function createStreamController({ onDisconnect, onError, log, provider, m
* activity), not here — output of the transform stream may be silent
* for long periods while raw bytes still flow (e.g. Kiro EventStream
* binary frames buffering, Claude reasoning streams).
*
* @param {function} [onAbortTerminal] - Receives a human-readable abort
* message and returns terminal SSE bytes to emit downstream.
*/
export function createDisconnectAwareStream(transformStream, streamController, onAbortTerminal = null) {
const reader = transformStream.readable.getReader();
@@ -194,6 +197,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont
let chunkCount = 0;
let totalBytes = 0;
let lastChunkAt = Date.now();
let abortMessage = "upstream connection lost";
const t0 = Date.now();
const tag = "STREAM";
const clearStall = () => {
@@ -203,6 +207,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont
clearStall();
stallTimer = setTimeout(() => {
stallTimer = null;
abortMessage = "stream stall timeout";
dbg(tag, `STALL TIMEOUT ${stallTimeoutMs}ms | chunks=${chunkCount} | bytes=${totalBytes} | sinceLast=${Date.now() - lastChunkAt}ms`);
streamController.handleError?.(new Error("stream stall timeout"));
streamController.abort?.();
@@ -249,7 +254,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont
return createDisconnectAwareStream(
{ readable: transformedBody, writable: { getWriter: () => ({ abort: () => Promise.resolve() }) } },
wrappedController,
onAbortTerminal
onAbortTerminal ? () => onAbortTerminal(abortMessage) : null
);
}

View File

@@ -1,4 +1,8 @@
import { FORMATS } from "../translator/formats.js";
import { buildErrorBody } from "./error.js";
import { SSE_DONE } from "./sseConstants.js";
const sharedEncoder = new TextEncoder();
// Parse SSE data line
export function parseSSELine(line, format = null) {
@@ -120,3 +124,24 @@ export function formatSSE(data, sourceFormat) {
return `data: ${JSON.stringify(data)}\n\n`;
}
// Terminal frames for a stream that aborted after HTTP 200 was already sent, so
// the status code can no longer change. OpenAI-compatible clients (openai-python
// raises APIError on any `data:` payload carrying an `error` key, checked before
// [DONE]) need the error frame first, then [DONE]; Anthropic clients need
// `event: error`. Never fabricate a successful finish_reason instead.
//
// Returns encoded bytes: onAbortTerminal callbacks are enqueued verbatim, same
// as buildAbortedResponsesTerminalBytes.
//
// NOTE: non-SSE client formats (Ollama NDJSON) get an SSE frame here — dead in
// practice because detectFormatByEndpoint never resolves to OLLAMA.
export function buildStreamErrorBytes(statusCode, message, clientFormat) {
const { error } = buildErrorBody(statusCode, message);
const sse = clientFormat === FORMATS.CLAUDE
? formatSSE({ type: "error", error }, FORMATS.CLAUDE)
: formatSSE({ error }, clientFormat) + SSE_DONE;
return sharedEncoder.encode(sse);
}