Merge origin/master (v0.5.81) into gitea/new_feature
Resolve conflicts: - streamingHandler.js: merge buildStreamErrorBytes onAbortTerminal + local shouldPersistRequestDetail & streamStatusForContent - capabilities.js: preserve server-injected user-asserted caps and models.dev catalog lookup; wire CommandCode /alpha/generate caps inside resolve() - commandcode.js (services/usage): adopt upstream whoami + billing credits/subscriptions with 5h/weekly rate windows and plan caps - openai-to-commandcode.js: merge toNativeImageBlock (data-URI & http(s) support) and assistant reasoning_content preservation - commandcode-to-openai.js: adopt upstream mid-stream error throw for clean retry and abortion - tests: sync commandcode test suite and exclude .next from vitest config
This commit is contained in:
@@ -178,6 +178,11 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C
|
||||
// makes the backend flag the request and answer 429 Quota Exhausted.
|
||||
export const ANTIGRAVITY_PROMPT_REWRITES = [
|
||||
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
|
||||
{ from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." },
|
||||
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
|
||||
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
|
||||
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.
|
||||
{ from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" },
|
||||
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
|
||||
];
|
||||
|
||||
|
||||
@@ -27,11 +27,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]);
|
||||
// ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only.
|
||||
function findModel(models, modelId, aliasOrId) {
|
||||
if (!models) return undefined;
|
||||
const found = models.find(m => m.id === modelId);
|
||||
const baseModelId = typeof modelId === "string"
|
||||
? modelId.replace(/\([^()]+\)\s*$/, "").trim()
|
||||
: modelId;
|
||||
const found = models.find(m => m.id === modelId || m.id === baseModelId);
|
||||
if (found) return found;
|
||||
if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined;
|
||||
const normalized = normalizeModelId(modelId);
|
||||
if (normalized === modelId) return undefined;
|
||||
const normalized = normalizeModelId(baseModelId);
|
||||
if (normalized === baseModelId) return undefined;
|
||||
return models.find(m => m.id === normalized);
|
||||
}
|
||||
|
||||
|
||||
@@ -212,7 +212,7 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
const modifiedParts = parts?.map(p => {
|
||||
if (!p.functionCall) return p;
|
||||
const callId = p.functionCall.id;
|
||||
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null;
|
||||
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null;
|
||||
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
|
||||
firstFunctionCallSeen = true;
|
||||
if (callSig) {
|
||||
|
||||
@@ -128,7 +128,7 @@ export class BaseExecutor {
|
||||
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
|
||||
const url = this.buildUrl(model, stream, urlIndex, credentials);
|
||||
const transformedBody = this.transformRequest(model, body, stream, credentials);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model, transformedBody);
|
||||
|
||||
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
|
||||
|
||||
|
||||
@@ -47,10 +47,24 @@ export class CommandCodeExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
async execute(opts) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
|
||||
return result;
|
||||
const maxRetries = 2;
|
||||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
|
||||
const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
|
||||
if (!wrappedResponse.ok && attempt < maxRetries) {
|
||||
const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504;
|
||||
if (isRetryableStatus) {
|
||||
opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`);
|
||||
await new Promise(r => setTimeout(r, 1000 * (attempt + 1)));
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
result.response = wrappedResponse;
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
|
||||
@@ -146,7 +146,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
return BEARER;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
buildHeaders(credentials, stream = true, url, model, body = null) {
|
||||
const rt = credentials?.runtimeTransport;
|
||||
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
|
||||
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
|
||||
@@ -166,7 +166,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
|
||||
if (model && (this.provider === "claude"
|
||||
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model, body);
|
||||
}
|
||||
|
||||
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
import crypto from "node:crypto";
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import { modelTargetFormat } from "../providers/models/schema.js";
|
||||
import { getProviderModels } from "../config/providerModels.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
@@ -45,8 +46,11 @@ function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …).
|
||||
// Reading the registry keeps this in sync with config — never hardcode model ids here.
|
||||
function isResponsesModel(model) {
|
||||
return isMuseSparkModel(baseModelId(model));
|
||||
const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model));
|
||||
return modelTargetFormat(entry) === "openai-responses";
|
||||
}
|
||||
|
||||
// Flatten Chat Completions tool declarations into the Responses flat shape and
|
||||
@@ -90,6 +94,12 @@ function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: Muse Spark contributor models route to
|
||||
// an upstream Console backend where encrypted_content cannot be validated across
|
||||
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
|
||||
@@ -1,24 +1,311 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
const OPENCODE_UA = "opencode";
|
||||
const OPENCODE_UA = "opencode/1.18.31";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeSession";
|
||||
const REQ_FIELD = "_opencodeRequest";
|
||||
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
|
||||
|
||||
// OpenCode free tier requires both 'bash' and 'read' in tools payload.
|
||||
// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read)
|
||||
// take precedence while satisfying upstream verification.
|
||||
const OPENCODE_DECOY_CHAT_TOOLS = [
|
||||
{
|
||||
type: "function",
|
||||
function: {
|
||||
name: "bash",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "function",
|
||||
function: {
|
||||
name: "read",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
const OPENCODE_DECOY_RESPONSES_TOOLS = [
|
||||
{
|
||||
type: "function",
|
||||
name: "bash",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
{
|
||||
type: "function",
|
||||
name: "read",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
];
|
||||
|
||||
function cloakOpencodeTools(body, isResponses) {
|
||||
if (!body || typeof body !== "object") return;
|
||||
if (isResponses) {
|
||||
if (!Array.isArray(body.tools)) body.tools = [];
|
||||
const names = new Set(body.tools.map((t) => t.name || t.function?.name));
|
||||
for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) {
|
||||
if (!names.has(tool.name)) body.tools.push({ ...tool });
|
||||
}
|
||||
if (!body.tool_choice) body.tool_choice = "auto";
|
||||
} else {
|
||||
const hasTools = Array.isArray(body.tools) && body.tools.length > 0;
|
||||
if (!hasTools) {
|
||||
body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } }));
|
||||
if (!body.tool_choice) body.tool_choice = "none";
|
||||
} else {
|
||||
const names = new Set(body.tools.map((t) => t.function?.name || t.name));
|
||||
for (const tool of OPENCODE_DECOY_CHAT_TOOLS) {
|
||||
if (!names.has(tool.function.name)) {
|
||||
body.tools.push({ ...tool, function: { ...tool.function } });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function hasValidOpencodeVersion(ua) {
|
||||
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
|
||||
if (!m) return false;
|
||||
const major = parseInt(m[1], 10);
|
||||
const minor = parseInt(m[2], 10);
|
||||
return major > 1 || (major === 1 && minor >= 17);
|
||||
}
|
||||
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
|
||||
const RESPONSES_MODELS = new Set([
|
||||
"muse-spark-1.2-contributor-free",
|
||||
"muse-spark-1.3-contributor-free",
|
||||
]);
|
||||
const MESSAGES_MODELS = new Set(["union-alpha"]);
|
||||
|
||||
function generateRequestId() {
|
||||
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
let lastTimestamp = 0;
|
||||
let counter = 0;
|
||||
|
||||
function unstableRandom() {
|
||||
const bytes = crypto.randomBytes(14);
|
||||
let randomPart = "";
|
||||
for (let i = 0; i < 14; i++) {
|
||||
randomPart += BASE62_CHARS[bytes[i] % 62];
|
||||
}
|
||||
return randomPart;
|
||||
}
|
||||
|
||||
function generateSessionId() {
|
||||
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
export function generateSessionId(timestamp = Date.now()) {
|
||||
if (timestamp !== lastTimestamp) {
|
||||
lastTimestamp = timestamp;
|
||||
counter = 0;
|
||||
}
|
||||
counter++;
|
||||
|
||||
const current = BigInt(timestamp) * 0x1000n + BigInt(counter);
|
||||
const value = ~current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `ses_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function generateRequestId(timestamp = Date.now()) {
|
||||
const current = BigInt(timestamp) * 0x1000n + 1n;
|
||||
const value = current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `msg_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function translateSessionId(sessionId, clientTool = "") {
|
||||
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
|
||||
return sessionId.trim();
|
||||
}
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
return `ses_${timeHex}${randomPart}`;
|
||||
}
|
||||
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
const normalized = normalizeSession(value);
|
||||
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Upstream free-tier quota is accounted per session. Minting a fresh
|
||||
// x-opencode-session on every request burns through it and surfaces as
|
||||
// 429 FreeUsageLimitError with growing reset-after delays, while the real
|
||||
// CLI reuses one long-lived canonical session per conversation. Mirror
|
||||
// that: one stable canonical session per downstream identity, evicted
|
||||
// after MEMORY_CONFIG.sessionTtlMs like the other session stores.
|
||||
const stableOpencodeSessions = new Map();
|
||||
const MAX_STABLE_SESSIONS = 1000;
|
||||
const stableSessionCleanup = setInterval(() => {
|
||||
const now = Date.now();
|
||||
for (const [key, entry] of stableOpencodeSessions) {
|
||||
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) {
|
||||
stableOpencodeSessions.delete(key);
|
||||
}
|
||||
}
|
||||
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
|
||||
if (stableSessionCleanup.unref) stableSessionCleanup.unref();
|
||||
|
||||
function identityKey(credentials) {
|
||||
const connectionId = credentials?.connectionId || credentials?.id;
|
||||
if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`;
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || "";
|
||||
if (auth) {
|
||||
const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32);
|
||||
return `opencode:auth:${digest}`;
|
||||
}
|
||||
return "opencode:default";
|
||||
}
|
||||
|
||||
export function stableSessionId(credentials) {
|
||||
const key = identityKey(credentials);
|
||||
const existing = stableOpencodeSessions.get(key);
|
||||
if (existing) {
|
||||
existing.lastUsed = Date.now();
|
||||
stableOpencodeSessions.delete(key);
|
||||
stableOpencodeSessions.set(key, existing);
|
||||
return existing.sessionId;
|
||||
}
|
||||
const sessionId = generateSessionId();
|
||||
if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) {
|
||||
stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value);
|
||||
}
|
||||
stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() });
|
||||
return sessionId;
|
||||
}
|
||||
|
||||
function lastUserText(body) {
|
||||
try {
|
||||
if (!body || typeof body !== "object") return "";
|
||||
const arr = Array.isArray(body.messages)
|
||||
? body.messages
|
||||
: Array.isArray(body.input)
|
||||
? body.input
|
||||
: null;
|
||||
if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : "";
|
||||
for (let i = arr.length - 1; i >= 0; i--) {
|
||||
const msg = arr[i];
|
||||
if (!msg) continue;
|
||||
if (msg.role && msg.role !== "user") continue;
|
||||
const content = msg.content;
|
||||
if (typeof content === "string" && content.trim()) return content.trim().slice(-600);
|
||||
if (Array.isArray(content)) {
|
||||
const text = content
|
||||
.map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || ""))
|
||||
.join(" ")
|
||||
.trim();
|
||||
if (text) return text.slice(-600);
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
// The real CLI sends the current user message id (stable per turn, same on
|
||||
// retries) as x-opencode-request. Derive it deterministically from the
|
||||
// session plus the last user message so retries share the id.
|
||||
export function deriveRequestId(sessionId, body) {
|
||||
const text = lastUserText(body);
|
||||
if (!text) return generateRequestId();
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-req\0${sessionId || ""}\0${text}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
const id = `msg_${timeHex}${randomPart}`;
|
||||
return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId();
|
||||
}
|
||||
|
||||
function normalizeRequestId(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null;
|
||||
}
|
||||
|
||||
function bodyHasSessionHints(body) {
|
||||
try {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (typeof body.session_id === "string" && body.session_id.trim()) return true;
|
||||
if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true;
|
||||
if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true;
|
||||
if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true;
|
||||
if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true;
|
||||
const arr = Array.isArray(body.messages)
|
||||
? body.messages
|
||||
: Array.isArray(body.input)
|
||||
? body.input
|
||||
: null;
|
||||
if (arr) {
|
||||
let assistantText = "";
|
||||
for (const msg of arr) {
|
||||
if (msg?.role === "assistant") {
|
||||
const content = msg.content;
|
||||
if (typeof content === "string") assistantText += content;
|
||||
else if (Array.isArray(content)) {
|
||||
for (const part of content) assistantText += part?.text || part?.output || "";
|
||||
}
|
||||
if (assistantText.length >= 50) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
|
||||
@@ -31,14 +318,116 @@ function isResponsesModel(model) {
|
||||
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials) {
|
||||
function isMessagesModel(model) {
|
||||
return MESSAGES_MODELS.has(baseModelId(model));
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) {
|
||||
const headers = credentials?.rawHeaders || {};
|
||||
return resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
generate: generateSessionId,
|
||||
const native = nativeSession(headers);
|
||||
if (native) return native;
|
||||
|
||||
let incoming = null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
incoming = normalizeSession(value);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
const hinted = incoming || normalizeSession(providerSessionId);
|
||||
if (hinted) return translateSessionId(hinted, clientTool);
|
||||
|
||||
if (credentials?.connectionId || bodyHasSessionHints(body)) {
|
||||
let viaManager = null;
|
||||
try {
|
||||
viaManager = resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
});
|
||||
} catch {
|
||||
viaManager = null;
|
||||
}
|
||||
if (viaManager) return translateSessionId(viaManager, clientTool);
|
||||
}
|
||||
|
||||
return stableSessionId(credentials);
|
||||
}
|
||||
|
||||
function resolveOpencodeRequestId(body, credentials, sessionId) {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
for (const [key, value] of Object.entries(raw)) {
|
||||
if (key.toLowerCase() === "x-opencode-request") {
|
||||
const normalized = normalizeRequestId(value);
|
||||
if (normalized) return normalized;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return deriveRequestId(sessionId, body);
|
||||
}
|
||||
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials
|
||||
// (`Bearer public`) routing to an upstream OpenAI/Console account pool.
|
||||
// OpenAI Responses API strictly enforces that reasoning `encrypted_content`
|
||||
// can only be decrypted by the exact caller/account that issued it; sending it
|
||||
// across different accounts or rotating proxy relays triggers:
|
||||
// [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400).
|
||||
// Furthermore, under stateless mode (store=false), omitting encrypted_content
|
||||
// causes OpenAI to reject the referenced reasoning item as "not found or was deleted".
|
||||
// Dropping prior reasoning items allows multi-turn conversations and tool-calling
|
||||
// loops to succeed cleanly.
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -68,12 +457,35 @@ function normalizeOpencodeReasoning(model, body) {
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode", PROVIDERS.opencode);
|
||||
this._currentSessionId = null;
|
||||
}
|
||||
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool);
|
||||
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: session,
|
||||
[REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session),
|
||||
};
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
this._currentSessionId = resolveOpencodeSession(body, credentials);
|
||||
if (isResponsesModel(model)) {
|
||||
if (body && typeof body === "object" && model && !body.model) body.model = model;
|
||||
// Zen rejects non-streaming requests on free models with 403 FreeTierError;
|
||||
// always stream upstream and let the handler layer aggregate for non-stream clients.
|
||||
if (body && typeof body === "object") body.stream = true;
|
||||
if (isResponsesModel(model || body?.model) && body && typeof body === "object") {
|
||||
// ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng.
|
||||
if ("tool_choice" in body && body.tool_choice !== "auto"
|
||||
&& this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) {
|
||||
body.tool_choice = "auto";
|
||||
}
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
if (!Array.isArray(body.input) || body.input.length === 0) {
|
||||
body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses API names the output cap max_output_tokens and takes thinking
|
||||
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
|
||||
if (body.max_output_tokens === undefined) {
|
||||
@@ -83,34 +495,53 @@ export class OpenCodeExecutor extends BaseExecutor {
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
normalizeOpencodeReasoning(model, body);
|
||||
body.stream = true;
|
||||
body.store = false;
|
||||
normalizeResponsesTools(body);
|
||||
sanitizeResponsesItems(body);
|
||||
if (!Array.isArray(body.tools) || body.tools.length === 0) {
|
||||
cloakOpencodeTools(body, true);
|
||||
}
|
||||
} else if (body && typeof body === "object") {
|
||||
cloakOpencodeTools(body, false);
|
||||
}
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
return isResponsesModel(model)
|
||||
? `${base}/zen/v1/responses`
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
async execute(args) {
|
||||
return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) });
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
if (isResponsesModel(model)) return `${base}/zen/v1/responses`;
|
||||
if (isMessagesModel(model)) return `${base}/zen/v1/messages`;
|
||||
return `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url = "") {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const lower = {};
|
||||
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
|
||||
|
||||
const downstreamUa = lower["user-agent"] || "";
|
||||
const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode");
|
||||
const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa);
|
||||
|
||||
return {
|
||||
const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD];
|
||||
const downstreamReq = normalizeRequestId(lower["x-opencode-request"]);
|
||||
const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId();
|
||||
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer public",
|
||||
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
|
||||
"x-opencode-client": lower["x-opencode-client"] || "desktop",
|
||||
"x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(),
|
||||
"x-opencode-request": lower["x-opencode-request"] || generateRequestId(),
|
||||
"x-opencode-session": session,
|
||||
"x-opencode-request": requestId,
|
||||
"x-opencode-project": lower["x-opencode-project"] || "global",
|
||||
"Accept": stream ? "text/event-stream" : "*/*",
|
||||
};
|
||||
if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION;
|
||||
return headers;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,11 +29,18 @@ import {
|
||||
zedLlmFetch,
|
||||
} from "../shared/zedAuth.js";
|
||||
|
||||
// Wire values for the `provider` field of POST /completions. These are NOT
|
||||
// display names: cloud.zed.dev matches them exactly, and an unrecognized value
|
||||
// fails the whole request with `500 {"message":"An internal server error
|
||||
// occurred."}` before the model is ever looked at. Spellings come from Zed's
|
||||
// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore),
|
||||
// `x_ai` follows the same convention — so feeding a catalog value back through
|
||||
// normalizeZedProvider is identity.
|
||||
const ZED_PROVIDER = {
|
||||
anthropic: "Anthropic",
|
||||
openai: "OpenAi",
|
||||
google: "Google",
|
||||
xai: "XAi",
|
||||
anthropic: "anthropic",
|
||||
openai: "open_ai",
|
||||
google: "google",
|
||||
xai: "x_ai",
|
||||
};
|
||||
|
||||
function normalizeZedProvider(value, model) {
|
||||
@@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) {
|
||||
return openaiToClaudeRequest(model, body, true);
|
||||
}
|
||||
if (provider === ZED_PROVIDER.google) {
|
||||
return openaiToGeminiRequest(model, body, true);
|
||||
const geminiRequest = openaiToGeminiRequest(model, body, true);
|
||||
// Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the
|
||||
// public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`,
|
||||
// `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google
|
||||
// path so Zed applies its own defaults — scoped here so native Gemini/
|
||||
// Antigravity is untouched.
|
||||
delete geminiRequest.safetySettings;
|
||||
return geminiRequest;
|
||||
}
|
||||
if (provider === ZED_PROVIDER.openai) {
|
||||
return openaiToOpenAIResponsesRequest(model, body, true, credentials);
|
||||
|
||||
@@ -6,8 +6,9 @@ import {
|
||||
} from "../../utils/stream.js";
|
||||
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildStreamErrorBytes } from "../../utils/streamHelpers.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
@@ -130,13 +131,21 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
|
||||
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
// Terminal bytes when the stream aborts after HTTP 200 was already sent, so the
|
||||
// client sees a real error instead of a silently truncated stream.
|
||||
// Responses passthrough keeps its own response.failed shape; every other client
|
||||
// format gets the OpenAI error frame + [DONE], or `event: error` for Claude.
|
||||
const isResponsesPassthrough =
|
||||
sourceFormat === FORMATS.OPENAI_RESPONSES &&
|
||||
targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough
|
||||
? buildAbortedResponsesTerminalBytes
|
||||
: null;
|
||||
: (message) =>
|
||||
buildStreamErrorBytes(
|
||||
HTTP_STATUS.GATEWAY_TIMEOUT,
|
||||
message,
|
||||
sourceFormat,
|
||||
);
|
||||
const stallTimeoutMs =
|
||||
PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(
|
||||
|
||||
@@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = {
|
||||
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
|
||||
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
|
||||
// DeepSeek V4.1-Flash is natively multimodal — models.dev lists
|
||||
// opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream
|
||||
// the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the
|
||||
// same image capability as the exp id above. "deepseek-flash" is the GA id on the
|
||||
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
|
||||
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
|
||||
// short-circuits the pattern table, so a vision-only delta would drop them.
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
|
||||
|
||||
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
|
||||
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
"coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
@@ -131,6 +141,8 @@ export const MODEL_CAPABILITIES = {
|
||||
// via OpenAI Responses input_image; reasoning supports up to xhigh.
|
||||
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
// OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output
|
||||
"union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 },
|
||||
};
|
||||
|
||||
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
@@ -214,6 +226,13 @@ export const PROVIDER_CAPABILITIES = {
|
||||
// contract). maxOutput 128000 per the server's product-config payload.
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
|
||||
},
|
||||
// CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors
|
||||
// the codebuddy-cn entry (the openai-style reasoning_effort format matters:
|
||||
// the generic *deepseek-v4* pattern would otherwise pick the vendor-native
|
||||
// "deepseek" thinking shape, which the CodeBuddy gateway does not accept).
|
||||
"codebuddy-intl": {
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
|
||||
},
|
||||
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
|
||||
// registry `name` is display-only and capability lookup matches on the raw
|
||||
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
|
||||
@@ -251,6 +270,16 @@ export const PROVIDER_CAPABILITIES = {
|
||||
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
|
||||
"laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 },
|
||||
},
|
||||
// Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge
|
||||
// the library page publishes for this model (text+image in, 1M context).
|
||||
// ponytail: thinkingFormat stays "deepseek" to preserve today's body shape;
|
||||
// Ollama's native toggle is the top-level `think` field (bool or
|
||||
// low/medium/high/max), which no format in thinkingUnified.js emits yet —
|
||||
// openai-to-ollama.js drops it. Wire a "think" format when thinking on
|
||||
// Ollama Cloud is actually needed.
|
||||
"ollama": {
|
||||
"deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -349,7 +378,11 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
|
||||
|
||||
// ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─
|
||||
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } },
|
||||
// v4.1+ has real image input (probed live on Alibaba MaaS: correct color
|
||||
// read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore
|
||||
// them (answered "Unknown"), so vision stays scoped to v4.* dotted releases.
|
||||
{ pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } },
|
||||
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } },
|
||||
{ pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
|
||||
{ pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
|
||||
{ pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } },
|
||||
@@ -425,6 +458,7 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
// globalThis, which IS shared across server bundles in the same process.
|
||||
// Same reason the browser bundle is safe: it never calls a setter, so the slots
|
||||
// stay empty and every consumer below short-circuits.
|
||||
let catalogSource = null;
|
||||
const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
|
||||
catalog: null, // { getModalities, getLimits } — synced models.dev catalog
|
||||
userCaps: null, // (provider, model) => asserted caps — dashboard toggles
|
||||
@@ -432,10 +466,20 @@ const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
|
||||
|
||||
/**
|
||||
* Install the synced catalog reader (server only).
|
||||
* @param {{ getModalities: Function, getLimits: Function } | null} source
|
||||
* @param {{ getModalities: (provider: string, model: string) => object|null,
|
||||
* getLimits: (provider: string, model: string) => object|null } | null} source
|
||||
*/
|
||||
export function setCatalogSource(source) {
|
||||
catalogSource = source || null;
|
||||
SOURCE_SLOTS.catalog = source || null;
|
||||
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source || null;
|
||||
}
|
||||
|
||||
function getCatalogSource() {
|
||||
if (catalogSource) return catalogSource;
|
||||
if (SOURCE_SLOTS.catalog) return (catalogSource = SOURCE_SLOTS.catalog);
|
||||
if (typeof globalThis === "undefined") return null;
|
||||
return (catalogSource = globalThis.__9rCatalogSource || null);
|
||||
}
|
||||
|
||||
// Capabilities the user asserted per provider+model (dashboard "Add/Edit Model"
|
||||
@@ -478,16 +522,17 @@ function applyUserCaps(result, provider, model) {
|
||||
// flips when an outside source positively declares support.
|
||||
function refine(base, provider, model) {
|
||||
const result = { ...DEFAULT_CAPABILITIES, ...base };
|
||||
const catalogSource = SOURCE_SLOTS.catalog;
|
||||
if (catalogSource) {
|
||||
const modalities = catalogSource.getModalities(model);
|
||||
|
||||
const source = getCatalogSource();
|
||||
if (source) {
|
||||
const modalities = source.getModalities(provider, model);
|
||||
if (modalities) {
|
||||
for (const key of MODALITY_KEYS) {
|
||||
if (modalities[key] === true) result[key] = true;
|
||||
}
|
||||
}
|
||||
|
||||
const limits = catalogSource.getLimits(provider, model);
|
||||
const limits = source.getLimits(provider, model);
|
||||
if (limits) {
|
||||
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
|
||||
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
|
||||
@@ -499,12 +544,67 @@ function refine(base, provider, model) {
|
||||
return result;
|
||||
}
|
||||
|
||||
// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models
|
||||
// default to vision; only this denylist stays text-only.
|
||||
const COMMANDCODE_TEXT_ONLY = new Set([
|
||||
"deepseek/deepseek-v4-pro",
|
||||
"deepseek/deepseek-v4-flash",
|
||||
"deepseek/deepseek-v4-flash-fast",
|
||||
"zai-org/glm-5.3",
|
||||
"zai-org/glm-5.2",
|
||||
"zai-org/glm-5.2-fast",
|
||||
"zai-org/glm-5.1",
|
||||
"zai-org/glm-5",
|
||||
"minimaxai/minimax-m2.7",
|
||||
"minimax/minimax-m2.7-free",
|
||||
"minimaxai/minimax-m2.5",
|
||||
"xiaomi/mimo-v2.5-pro",
|
||||
"qwen/qwen3.6-max-preview",
|
||||
"qwen/qwen3.7-max",
|
||||
"meituan/longcat-2.0:free",
|
||||
"stepfun/step-3.5-flash",
|
||||
"tencent/hy4-preview",
|
||||
"tencent/hy3",
|
||||
"tencent/hy3-paid",
|
||||
"nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"poolside/laguna-s-2.1-free",
|
||||
"inclusionai/ling-3.0-flash-free",
|
||||
"inclusionai/ling-3.0-flash-sante:free",
|
||||
]);
|
||||
|
||||
function isCommandCodeTextOnly(model) {
|
||||
const key = String(model || "").toLowerCase();
|
||||
if (COMMANDCODE_TEXT_ONLY.has(key)) return true;
|
||||
for (const id of COMMANDCODE_TEXT_ONLY) {
|
||||
const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
|
||||
if (key === base || key.endsWith("/" + base)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
export function getCapabilitiesForModel(provider, model) {
|
||||
if (!model) return { ...DEFAULT_CAPABILITIES };
|
||||
|
||||
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
const resolve = () => {
|
||||
// CommandCode wire is /alpha/generate for every model. Family patterns
|
||||
// (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here.
|
||||
if (provider === "commandcode" || provider === "cmc") {
|
||||
const providerCaps = PROVIDER_CAPABILITIES.commandcode;
|
||||
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
return {
|
||||
...DEFAULT_CAPABILITIES,
|
||||
reasoning: true,
|
||||
thinkingFormat: "commandcode",
|
||||
thinkingEffortSupported: true,
|
||||
vision: !isCommandCodeTextOnly(model),
|
||||
contextWindow: 1000000,
|
||||
maxOutput: 384000,
|
||||
};
|
||||
}
|
||||
|
||||
// 1. Provider-specific override
|
||||
if (provider) {
|
||||
const providerCaps = PROVIDER_CAPABILITIES[provider];
|
||||
|
||||
@@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
|
||||
// Trimmed upstream catalog, read by the add-models skill (not by the router).
|
||||
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
|
||||
|
||||
// Schema of the file this module reads. The writer stamps it; a file carrying an
|
||||
// older value predates provider-scoped modality keys, and its flat keys are not
|
||||
// looked up here, so the sync rebuilds it instead of asking upstream for a 304.
|
||||
export const CATALOG_VERSION = 2;
|
||||
|
||||
const EMPTY = { models: {}, providers: {} };
|
||||
let cache = EMPTY;
|
||||
let cachedMtime = -1;
|
||||
@@ -45,14 +50,19 @@ function load() {
|
||||
return cache;
|
||||
}
|
||||
|
||||
// Modality is a property of the model itself — any gateway serving it inherits
|
||||
// the same image/video/pdf support, so this is keyed by model id alone.
|
||||
export function getCatalogModalities(model) {
|
||||
return load().models[baseId(model)] || null;
|
||||
// Modalities are recorded per gateway upstream, and gateways disagree about the
|
||||
// same weights — some do not proxy images at all — so the key is provider +
|
||||
// model, in the local provider id space, exactly like the limits below. Keying
|
||||
// by model id alone made short ids collide across vendors: "auto", "free" and
|
||||
// "efficient" are router modes in one catalog and model names in another, and a
|
||||
// request to the router mode inherited a stranger's vision.
|
||||
export function getCatalogModalities(provider, model) {
|
||||
if (!provider) return null;
|
||||
return load().models[`${provider}:${baseId(model)}`] || null;
|
||||
}
|
||||
|
||||
// Context and output limits are a property of the gateway, not the model: each
|
||||
// one truncates differently, so these stay keyed by provider + model.
|
||||
// Context and output limits are a property of the gateway too: each one
|
||||
// truncates differently, so these stay keyed by provider + model.
|
||||
export function getCatalogLimits(provider, model) {
|
||||
const byProvider = provider && load().providers[provider];
|
||||
if (!byProvider) return null;
|
||||
|
||||
@@ -111,6 +111,8 @@ export const MODEL_PRICING = {
|
||||
"deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 },
|
||||
|
||||
// === GLM ===
|
||||
|
||||
@@ -58,7 +58,9 @@ export default {
|
||||
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
|
||||
{ id: "hy3-preview", name: "Hy3 Preview" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
|
||||
// deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same
|
||||
// catalog as CN; the old endpoint still answers 200 but the list is the contract).
|
||||
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
|
||||
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
|
||||
],
|
||||
oauth: {
|
||||
|
||||
@@ -65,6 +65,9 @@ export default {
|
||||
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
|
||||
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
|
||||
// Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived
|
||||
// from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398).
|
||||
{ id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" },
|
||||
{ id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
|
||||
@@ -74,4 +74,8 @@ export default {
|
||||
{ id: "claude-opus-4-7", name: "Claude Opus 4.7" },
|
||||
{ id: "claude-haiku-4-5", name: "Claude Haiku 4.5" },
|
||||
],
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -59,6 +59,7 @@ export default {
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
|
||||
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
|
||||
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
|
||||
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
|
||||
|
||||
@@ -30,6 +30,7 @@ export default {
|
||||
{ id: "glm-4.7-flash", name: "GLM 4.7 Flash" },
|
||||
{ id: "qwen3.5", name: "Qwen3.5" },
|
||||
{ id: "minimax-m3", name: "MiniMax M3" },
|
||||
{ id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" },
|
||||
],
|
||||
serviceKinds: ["llm", "webFetch"],
|
||||
fetchConfig: {
|
||||
|
||||
@@ -13,7 +13,7 @@ export default {
|
||||
textIcon: "OC",
|
||||
website: "https://opencode.ai/auth",
|
||||
notice: {
|
||||
text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
|
||||
text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
|
||||
apiKeyUrl: "https://opencode.ai/auth",
|
||||
},
|
||||
},
|
||||
|
||||
@@ -17,13 +17,17 @@ export default {
|
||||
headers: {
|
||||
"x-opencode-client": "desktop",
|
||||
},
|
||||
forceStream: true,
|
||||
noAuth: true,
|
||||
quirks: {
|
||||
forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"],
|
||||
},
|
||||
},
|
||||
models: [
|
||||
// Muse Spark models are served by /zen/v1/responses; the rest stay on
|
||||
// /chat/completions, so the format is declared per-model, not per-provider.
|
||||
// Endpoint formats differ per model, so declare non-chat models explicitly.
|
||||
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
|
||||
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
|
||||
{ id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" },
|
||||
],
|
||||
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
|
||||
passthroughModels: true,
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
// Zed provider — RSA keypair callback auth (NOT standard OAuth).
|
||||
export default {
|
||||
id: "zed",
|
||||
priority: 10,
|
||||
priority: 999,
|
||||
alias: "zd",
|
||||
uiAlias: "zd",
|
||||
hidden: true,
|
||||
display: {
|
||||
name: "Zed",
|
||||
icon: "code",
|
||||
|
||||
@@ -62,8 +62,17 @@ const ANTHROPIC_BETA_BASE = [
|
||||
const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"];
|
||||
|
||||
// Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them.
|
||||
export function selectAnthropicBeta(model = "") {
|
||||
const flags = [...ANTHROPIC_BETA_BASE];
|
||||
// `redact-thinking` asks Anthropic to return signature-only thinking blocks, which
|
||||
// is right for clients that never render thinking but blanks the summaries a
|
||||
// client explicitly requested with `thinking.display: "summarized"`.
|
||||
const ANTHROPIC_BETA_REDACT_THINKING = "redact-thinking-2026-02-12";
|
||||
|
||||
export function wantsThinkingSummaries(body) {
|
||||
return body?.thinking?.display === "summarized";
|
||||
}
|
||||
|
||||
export function selectAnthropicBeta(model = "", body = null) {
|
||||
const flags = ANTHROPIC_BETA_BASE.filter((flag) => flag !== ANTHROPIC_BETA_REDACT_THINKING || !wantsThinkingSummaries(body));
|
||||
if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT);
|
||||
return flags.join(",");
|
||||
}
|
||||
|
||||
@@ -26,6 +26,7 @@ const FORMAT_LEVELS = {
|
||||
qwen: L.base,
|
||||
kimi: L.levelMax,
|
||||
deepseek: L.hiMax,
|
||||
commandcode: ["none", "low", "medium", "high", "xhigh", "max"],
|
||||
minimax: L.onOff,
|
||||
hunyuan: L.base,
|
||||
step: L.base,
|
||||
@@ -40,6 +41,10 @@ const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
|
||||
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
|
||||
// (disable thinking instead). none kept for the picker = disable.
|
||||
{ pattern: "*deepseek-v4.*", levels: ["none", "low", "medium", "high", "xhigh", "max"] },
|
||||
// codebuddy-cn per-model effort sets — the server's product-config payload
|
||||
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
|
||||
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
|
||||
@@ -52,6 +57,8 @@ const PATTERN_THINKING = [
|
||||
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
|
||||
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
|
||||
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
|
||||
// codebuddy-intl rides the same gateway catalog, so its deepseek levels match.
|
||||
{ provider: "codebuddy-intl", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
|
||||
];
|
||||
|
||||
// The generic level set used when a model's thinking format is unknown. Exported
|
||||
|
||||
@@ -29,7 +29,7 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) {
|
||||
// Request-scoped rule: the request body itself is at fault — no cooldown,
|
||||
// no account lock. Caller must stop rotating and surface the error.
|
||||
if (rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
|
||||
return { shouldFallback: false, requestScoped: true, cooldownMs: 0 };
|
||||
return { shouldFallback: false, cooldownMs: 0 };
|
||||
}
|
||||
|
||||
// Text-based rule: match substring in error message
|
||||
@@ -51,6 +51,20 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) {
|
||||
}
|
||||
}
|
||||
|
||||
// Request-scoped client errors that matched no rule above: a 400 caused by the
|
||||
// request itself (context overflow, malformed body, unsupported parameter) says
|
||||
// nothing about the credential, so cooling the account down only removes a
|
||||
// healthy connection from rotation. With a single connection it is worse: every
|
||||
// later request in the window fails with a copy of this very error
|
||||
// ("all 1 accounts locked for <model> | lastError=[400]: ..."), which hides the
|
||||
// real cause from the caller and makes unrelated sessions look like they hit the
|
||||
// same limit. Hand the upstream error back for this request instead.
|
||||
// Account-scoped statuses keep their rules above (401/402/403/404/429), and the
|
||||
// text rules still win for rate-limit / quota / capacity wording.
|
||||
if (status >= 400 && status < 500 && status !== 401 && status !== 402 && status !== 403 && status !== 429) {
|
||||
return { shouldFallback: false, cooldownMs: 0 };
|
||||
}
|
||||
|
||||
// Default: transient cooldown for any unmatched error
|
||||
return { shouldFallback: true, cooldownMs: TRANSIENT_COOLDOWN_MS };
|
||||
}
|
||||
|
||||
@@ -124,6 +124,8 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) {
|
||||
|
||||
// Config-driven prefix → provider inference (first match wins, fallback "openai").
|
||||
const MODEL_PREFIX_PROVIDERS = [
|
||||
// Codex CLI sends this bare virtual model for auto-review — keep it on OAuth Codex (#1398).
|
||||
[/^codex-auto-review$/, "codex"],
|
||||
[/^claude-/, "anthropic"],
|
||||
[/^gemini-/, "gemini"],
|
||||
[/^gpt-/, "openai"],
|
||||
|
||||
@@ -10,6 +10,24 @@ const signatureKv = makeKv(SCOPE);
|
||||
const memorySignatures = new Map();
|
||||
let pruneCounter = 0;
|
||||
|
||||
/**
|
||||
* Model family that produced / will consume a signature. Antigravity serves Gemini and Claude
|
||||
* models behind the same API, and each backend only accepts its own signatures: a Claude
|
||||
* signature replayed to Gemini fails with 400 "Corrupted thought signature." (and vice versa).
|
||||
*/
|
||||
export function signatureFamily(model) {
|
||||
const m = typeof model === "string" ? model.toLowerCase() : "";
|
||||
if (!m) return null;
|
||||
if (m.includes("claude")) return "claude";
|
||||
if (m.includes("gemini")) return "gemini";
|
||||
return m;
|
||||
}
|
||||
|
||||
// Entries stored before families were recorded (no `family`) stay usable for any model.
|
||||
function isCompatible(entry, family) {
|
||||
return !entry.family || !family || entry.family === family;
|
||||
}
|
||||
|
||||
function pruneMemoryExpired() {
|
||||
const now = Date.now();
|
||||
for (const [key, value] of memorySignatures.entries()) {
|
||||
@@ -62,13 +80,15 @@ async function maybePrunePersisted() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async)
|
||||
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async).
|
||||
* `model` is the model that produced the signature; lookups for another model family skip it.
|
||||
*/
|
||||
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) {
|
||||
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null, model = null) {
|
||||
if (typeof toolCallId !== "string" || !toolCallId) return;
|
||||
if (typeof signature !== "string" || !signature) return;
|
||||
|
||||
const now = Date.now();
|
||||
const family = signatureFamily(model);
|
||||
pruneMemoryExpired();
|
||||
|
||||
const keys = [];
|
||||
@@ -80,12 +100,14 @@ export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = n
|
||||
for (const k of keys) {
|
||||
memorySignatures.set(k, {
|
||||
signature,
|
||||
family,
|
||||
expiresAt: now + MEMORY_TTL_MS,
|
||||
});
|
||||
|
||||
// Async persist to SQLite kv table without blocking
|
||||
signatureKv.set(k, {
|
||||
signature,
|
||||
family,
|
||||
createdAt: now,
|
||||
expiresAt: now + PERSISTED_TTL_MS,
|
||||
}).catch(() => {});
|
||||
@@ -95,23 +117,25 @@ export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = n
|
||||
}
|
||||
|
||||
/**
|
||||
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback)
|
||||
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback).
|
||||
* `model` is the target model; signatures produced by another model family are ignored.
|
||||
*/
|
||||
export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
|
||||
export async function getGeminiThoughtSignature(toolCallId, sessionId = null, model = null) {
|
||||
if (typeof toolCallId !== "string" || !toolCallId) return null;
|
||||
|
||||
const family = signatureFamily(model);
|
||||
pruneMemoryExpired();
|
||||
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
const sessionKey = `${sessionId}:${toolCallId}`;
|
||||
const sessionEntry = memorySignatures.get(sessionKey);
|
||||
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
|
||||
if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) {
|
||||
return sessionEntry.signature;
|
||||
}
|
||||
}
|
||||
|
||||
const entry = memorySignatures.get(toolCallId);
|
||||
if (entry && entry.expiresAt > Date.now()) {
|
||||
if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) {
|
||||
return entry.signature;
|
||||
}
|
||||
|
||||
@@ -119,9 +143,10 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
const sessionKey = `${sessionId}:${toolCallId}`;
|
||||
const sessionRow = await signatureKv.get(sessionKey);
|
||||
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) {
|
||||
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now()) && isCompatible(sessionRow, family)) {
|
||||
memorySignatures.set(sessionKey, {
|
||||
signature: sessionRow.signature,
|
||||
family: sessionRow.family || null,
|
||||
expiresAt: Date.now() + MEMORY_TTL_MS,
|
||||
});
|
||||
return sessionRow.signature;
|
||||
@@ -134,8 +159,10 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
|
||||
signatureKv.remove(toolCallId).catch(() => {});
|
||||
return null;
|
||||
}
|
||||
if (!isCompatible(row, family)) return null;
|
||||
memorySignatures.set(toolCallId, {
|
||||
signature: row.signature,
|
||||
family: row.family || null,
|
||||
expiresAt: Date.now() + MEMORY_TTL_MS,
|
||||
});
|
||||
return row.signature;
|
||||
@@ -148,22 +175,24 @@ export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Synchronous get from RAM cache only (for sync translators)
|
||||
* Synchronous get from RAM cache only (for sync translators).
|
||||
* `model` is the target model; signatures produced by another model family are ignored.
|
||||
*/
|
||||
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) {
|
||||
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null, model = null) {
|
||||
if (typeof toolCallId !== "string" || !toolCallId) return null;
|
||||
const family = signatureFamily(model);
|
||||
pruneMemoryExpired();
|
||||
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
const sessionKey = `${sessionId}:${toolCallId}`;
|
||||
const sessionEntry = memorySignatures.get(sessionKey);
|
||||
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
|
||||
if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) {
|
||||
return sessionEntry.signature;
|
||||
}
|
||||
}
|
||||
|
||||
const entry = memorySignatures.get(toolCallId);
|
||||
if (entry && entry.expiresAt > Date.now()) {
|
||||
if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) {
|
||||
return entry.signature;
|
||||
}
|
||||
return null;
|
||||
|
||||
@@ -22,6 +22,7 @@ import { getZedUsage } from "./usage/zed.js";
|
||||
import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js";
|
||||
import { resolveQoderCredentials } from "./qoderModels.js";
|
||||
import { getGlmUsage } from "./usage/glm.js";
|
||||
import { getCommandCodeUsage } from "./usage/commandcode.js";
|
||||
import {
|
||||
getIflowUsage,
|
||||
getOllamaUsage,
|
||||
@@ -66,6 +67,7 @@ const USAGE_HANDLERS = {
|
||||
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
|
||||
zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
"xiaomi-mimo": (c) => getXiaomiMimoUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
|
||||
};
|
||||
|
||||
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {
|
||||
|
||||
@@ -1,207 +1,134 @@
|
||||
/**
|
||||
* CommandCode usage handler
|
||||
*
|
||||
* Mirrors the official command-code CLI /usage command: it calls the alpha API
|
||||
* to surface the 5-hour + weekly usage windows, the subscription plan, and the
|
||||
* credits consumed in the current billing period.
|
||||
*
|
||||
* GET /alpha/whoami → org.id (org-scoped billing; null for personal)
|
||||
* GET /alpha/billing/credits → { credits: { monthlyCredits, purchasedCredits,
|
||||
* freeCredits }, windowLimits: { fiveHour, weekly } }
|
||||
* GET /alpha/billing/subscriptions → { data: { planId, currentPeriodStart, ... } }
|
||||
* GET /alpha/usage/summary?since= → period token/cost totals
|
||||
*
|
||||
* The CLI fetches whoami first (for orgId), then credits + subscription in
|
||||
* parallel, then the summary with since = currentPeriodStart. We keep the same
|
||||
* order/dependencies: window limits live on credits, and the plan period start
|
||||
* determines the summary window.
|
||||
* Command Code usage — billing credits + 5h/weekly rate windows.
|
||||
* Mirrors ~/cc-usage.mjs: whoami → credits + subscriptions.
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U, parseResetTime } from "./shared.js";
|
||||
import { parseResetTime, toFiniteNumber } from "./shared.js";
|
||||
|
||||
const USAGE = U("commandcode");
|
||||
const BASE = USAGE.baseUrl || "https://api.commandcode.ai";
|
||||
const WHOAMI_URL = BASE + (USAGE.whoamiUrl || "/alpha/whoami");
|
||||
const CREDITS_URL = BASE + (USAGE.creditsUrl || "/alpha/billing/credits");
|
||||
const SUBSCRIPTIONS_URL =
|
||||
BASE + (USAGE.subscriptionsUrl || "/alpha/billing/subscriptions");
|
||||
const SUMMARY_URL = BASE + (USAGE.summaryUrl || "/alpha/usage/summary");
|
||||
const BASE = (process.env.COMMAND_CODE_API_BASE_URL || "https://api.commandcode.ai").replace(/\/$/, "");
|
||||
|
||||
function buildHeaders(token) {
|
||||
return {
|
||||
Authorization: `Bearer ${token}`,
|
||||
Accept: "application/json",
|
||||
};
|
||||
const PLAN_NAMES = {
|
||||
"individual-go": "Go",
|
||||
"individual-goat": "GOAT",
|
||||
"individual-pro": "Pro",
|
||||
"individual-pro-v1": "Pro",
|
||||
"individual-provider": "Provider",
|
||||
"individual-max": "Max",
|
||||
"individual-ultra": "Ultra",
|
||||
"teams-pro": "Teams Pro",
|
||||
};
|
||||
|
||||
const PLAN_CAPS = {
|
||||
"individual-go": 10,
|
||||
"individual-goat": 70,
|
||||
"individual-pro": 30,
|
||||
"individual-pro-v1": 80,
|
||||
"individual-provider": 15,
|
||||
"individual-max": 150,
|
||||
"individual-ultra": 300,
|
||||
"teams-pro": 40,
|
||||
};
|
||||
|
||||
function qs(route, params) {
|
||||
const s = new URLSearchParams(
|
||||
Object.entries(params || {}).filter(([, v]) => v != null),
|
||||
).toString();
|
||||
return s ? `${route}?${s}` : route;
|
||||
}
|
||||
|
||||
/** Build a normalized quota row. `unit` is "$" — the API reports currency credits. */
|
||||
function makeQuota({ used, total, resetAt, unlimited = false, unit = "$" }) {
|
||||
const safeTotal = Math.max(0, Number(total) || 0);
|
||||
const safeUsed = Math.max(0, Number(used) || 0);
|
||||
if (unlimited || safeTotal === 0) {
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: 0,
|
||||
remainingPercentage: unlimited ? 100 : 0,
|
||||
resetAt: resetAt || null,
|
||||
unit,
|
||||
unlimited: true,
|
||||
};
|
||||
}
|
||||
const remaining = Math.max(0, safeTotal - safeUsed);
|
||||
const remainingPercentage = (remaining / safeTotal) * 100;
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: safeTotal,
|
||||
remainingPercentage,
|
||||
resetAt: resetAt || null,
|
||||
unit,
|
||||
unlimited: false,
|
||||
};
|
||||
function windowQuota(win) {
|
||||
if (!win || typeof win !== "object") return null;
|
||||
const used = toFiniteNumber(win.used, 0);
|
||||
const total = toFiniteNumber(win.cap, 0);
|
||||
if (total <= 0 && used <= 0) return null;
|
||||
return {
|
||||
used,
|
||||
total,
|
||||
remaining: Math.max(0, total - used),
|
||||
unlimited: false,
|
||||
resetAt: parseResetTime(win.resetAt),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} apiKey - commandcode API key (user_...)
|
||||
* @param {string|null|undefined} apiKey
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getCommandCodeUsage(apiKey, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "CommandCode credential not available." };
|
||||
}
|
||||
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
|
||||
return { message: "Command Code API key not available. Add a key to view usage." };
|
||||
}
|
||||
|
||||
const headers = buildHeaders(apiKey);
|
||||
const headers = {
|
||||
Authorization: `Bearer ${apiKey.trim()}`,
|
||||
Accept: "application/json",
|
||||
};
|
||||
|
||||
try {
|
||||
// whoami resolves the org id (billing is org-scoped; null for personal).
|
||||
const whoamiRes = await proxyAwareFetch(
|
||||
WHOAMI_URL,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
|
||||
return { message: "CommandCode credential invalid or expired." };
|
||||
}
|
||||
if (!whoamiRes.ok) {
|
||||
return { message: `CommandCode whoami API error (${whoamiRes.status}).` };
|
||||
}
|
||||
const whoami = await whoamiRes.json().catch(() => null);
|
||||
const orgId = whoami?.org?.id ?? null;
|
||||
const get = async (route) => {
|
||||
const response = await proxyAwareFetch(
|
||||
BASE + route,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
return response;
|
||||
};
|
||||
|
||||
const orgQuery = orgId ? `?orgId=${encodeURIComponent(orgId)}` : "";
|
||||
try {
|
||||
const whoamiRes = await get(qs("/alpha/whoami", { limits: "1" }));
|
||||
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
|
||||
return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." };
|
||||
}
|
||||
if (!whoamiRes.ok) {
|
||||
return { plan: "Command Code", message: `Command Code usage API error (${whoamiRes.status})` };
|
||||
}
|
||||
const whoami = await whoamiRes.json().catch(() => ({}));
|
||||
const orgId = whoami?.org?.id ?? null;
|
||||
|
||||
const [creditsRes, subsRes] = await Promise.all([
|
||||
proxyAwareFetch(
|
||||
CREDITS_URL + orgQuery,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
proxyAwareFetch(
|
||||
SUBSCRIPTIONS_URL + orgQuery,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
]);
|
||||
const [creditsRes, subsRes] = await Promise.all([
|
||||
get(qs("/alpha/billing/credits", { orgId })),
|
||||
get(qs("/alpha/billing/subscriptions", { orgId })),
|
||||
]);
|
||||
|
||||
if (
|
||||
creditsRes.status === 401 ||
|
||||
creditsRes.status === 403 ||
|
||||
subsRes.status === 401 ||
|
||||
subsRes.status === 403
|
||||
) {
|
||||
return { message: "CommandCode credential invalid or expired." };
|
||||
}
|
||||
if (!creditsRes.ok) {
|
||||
return {
|
||||
message: `CommandCode credits API error (${creditsRes.status}).`,
|
||||
};
|
||||
}
|
||||
if (creditsRes.status === 401 || creditsRes.status === 403 || subsRes.status === 401 || subsRes.status === 403) {
|
||||
return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." };
|
||||
}
|
||||
if (!creditsRes.ok) {
|
||||
return { plan: "Command Code", message: `Command Code credits API error (${creditsRes.status})` };
|
||||
}
|
||||
if (!subsRes.ok) {
|
||||
return { plan: "Command Code", message: `Command Code subscriptions API error (${subsRes.status})` };
|
||||
}
|
||||
|
||||
const credits = await creditsRes.json().catch(() => null);
|
||||
const subs = await subsRes.json().catch(() => null);
|
||||
const creditsBody = await creditsRes.json().catch(() => ({}));
|
||||
const subsBody = await subsRes.json().catch(() => ({}));
|
||||
const planId = subsBody?.data?.planId ?? null;
|
||||
const plan = (planId && PLAN_NAMES[planId]) || planId || "Command Code";
|
||||
const cap = planId ? (PLAN_CAPS[planId] || 0) : 0;
|
||||
const c = creditsBody?.credits || {};
|
||||
const remaining =
|
||||
toFiniteNumber(c.monthlyCredits, 0) +
|
||||
toFiniteNumber(c.purchasedCredits, 0) +
|
||||
toFiniteNumber(c.freeCredits, 0);
|
||||
const used = cap > 0 ? Math.max(0, cap - remaining) : 0;
|
||||
const total = cap > 0 ? cap : remaining;
|
||||
|
||||
const subData = subs?.data;
|
||||
const planId = subData?.planId ?? null;
|
||||
const periodStart = subData?.currentPeriodStart ?? null;
|
||||
const quotas = {};
|
||||
quotas.Credits = {
|
||||
used,
|
||||
total,
|
||||
remaining,
|
||||
unlimited: cap <= 0,
|
||||
resetAt: parseResetTime(subsBody?.data?.currentPeriodEnd),
|
||||
};
|
||||
|
||||
// Summary needs `since`; the CLI falls back to first-of-month when the
|
||||
// subscription period start is unavailable.
|
||||
const since = periodStart || firstOfMonth();
|
||||
const summaryRes = await proxyAwareFetch(
|
||||
`${SUMMARY_URL}?since=${encodeURIComponent(since)}`,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
const summary = summaryRes.ok
|
||||
? await summaryRes.json().catch(() => null)
|
||||
: null;
|
||||
const fiveHour = windowQuota(creditsBody?.windowLimits?.fiveHour);
|
||||
if (fiveHour) quotas["Session (5h)"] = fiveHour;
|
||||
const weekly = windowQuota(creditsBody?.windowLimits?.weekly);
|
||||
if (weekly) quotas.Weekly = weekly;
|
||||
|
||||
const quotas = {};
|
||||
const windowLimits = credits?.windowLimits || {};
|
||||
|
||||
const fiveHour = windowLimits.fiveHour;
|
||||
if (fiveHour && Number(fiveHour.cap) > 0) {
|
||||
quotas["5-hour window"] = makeQuota({
|
||||
used: fiveHour.used,
|
||||
total: fiveHour.cap,
|
||||
resetAt: parseResetTime(fiveHour.resetAt),
|
||||
});
|
||||
}
|
||||
|
||||
const weekly = windowLimits.weekly;
|
||||
if (weekly && Number(weekly.cap) > 0) {
|
||||
quotas["Weekly window"] = makeQuota({
|
||||
used: weekly.used,
|
||||
total: weekly.cap,
|
||||
resetAt: parseResetTime(weekly.resetAt),
|
||||
});
|
||||
}
|
||||
|
||||
// The credits API reports remaining balances (monthly/purchased/free),
|
||||
// not a total. The official CLI renders the monthly line as
|
||||
// `used = summary.totalCost`, `total = totalCost + remaining` — i.e.
|
||||
// the plan ceiling is the sum of what was consumed and what is left.
|
||||
const monthlyUsed =
|
||||
typeof summary?.totalCredits === "number"
|
||||
? summary.totalCredits
|
||||
: typeof summary?.totalCost === "number"
|
||||
? summary.totalCost
|
||||
: 0;
|
||||
|
||||
const creditsObj = credits?.credits || {};
|
||||
const remaining =
|
||||
Math.max(0, Number(creditsObj.monthlyCredits) || 0) +
|
||||
Math.max(0, Number(creditsObj.purchasedCredits) || 0) +
|
||||
Math.max(0, Number(creditsObj.freeCredits) || 0);
|
||||
const monthlyTotal = monthlyUsed + remaining;
|
||||
|
||||
if (monthlyTotal > 0 || monthlyUsed > 0) {
|
||||
quotas["Monthly credits"] = makeQuota({
|
||||
used: monthlyUsed,
|
||||
total: monthlyTotal,
|
||||
resetAt: periodStart ? undefined : null,
|
||||
});
|
||||
}
|
||||
|
||||
if (Object.keys(quotas).length === 0) {
|
||||
return {
|
||||
plan: planId || "CommandCode",
|
||||
message: "CommandCode connected, but no quota was reported.",
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
plan: planId || "CommandCode",
|
||||
quotas,
|
||||
periodBasis: summary?.periodBasis || "billing-period",
|
||||
};
|
||||
} catch (error) {
|
||||
return { message: `CommandCode usage error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
function firstOfMonth() {
|
||||
const now = new Date();
|
||||
return new Date(now.getFullYear(), now.getMonth(), 1).toISOString();
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: `Command Code error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
@@ -91,14 +91,15 @@ export async function getDeepseekUsage(apiKey = null, proxyOptions = null) {
|
||||
const quotas = {};
|
||||
for (const b of balances) {
|
||||
const total = Math.max(0, b.totalBalance);
|
||||
// Credit pot: show full remaining against current balance; never set absolute
|
||||
// `remaining` — QuotaTable treats it as a 0–100 percentage.
|
||||
// Credit balance: show as "Credit: $X.XX USD" not a usage quota
|
||||
quotas[`Balance (${b.currency})`] = {
|
||||
used: 0,
|
||||
total,
|
||||
remainingPercentage: total > 0 ? 100 : 0,
|
||||
resetAt: null,
|
||||
unlimited: total > 0,
|
||||
unlimited: false,
|
||||
isCreditBalance: true,
|
||||
currency: b.currency,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -112,7 +112,14 @@ export function parseZedCallbackPayload(input) {
|
||||
url = new URL(raw);
|
||||
} catch {
|
||||
try {
|
||||
url = new URL(`http://127.0.0.1/?${raw.replace(/^\?/, "")}`);
|
||||
// Accept pathname+query (what the local proxy forwards, e.g.
|
||||
// "/?user_id=..&access_token=.." or "/callback?.."), a bare query,
|
||||
// or a lone query string. Only the query part is parsed — a leading
|
||||
// path must never become part of the first parameter name.
|
||||
const query = raw.includes("?")
|
||||
? raw.slice(raw.indexOf("?") + 1)
|
||||
: raw.replace(/^\?/, "");
|
||||
url = new URL(`http://127.0.0.1/?${query}`);
|
||||
} catch {
|
||||
throw new Error("Invalid Zed callback URL");
|
||||
}
|
||||
@@ -134,6 +141,10 @@ export function parseZedCallbackPayload(input) {
|
||||
export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier) {
|
||||
const privateKey = decodeZedPrivateKeyVerifier(privateKeyVerifier);
|
||||
const encrypted = Buffer.from(String(encryptedAccessToken), "base64url");
|
||||
const fail = (oaepError) => {
|
||||
const message = oaepError instanceof Error ? oaepError.message : String(oaepError);
|
||||
throw new Error(`Failed to decrypt Zed access token: ${message}`);
|
||||
};
|
||||
try {
|
||||
return crypto
|
||||
.privateDecrypt(
|
||||
@@ -143,15 +154,21 @@ export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier)
|
||||
.toString("utf8");
|
||||
} catch (oaepError) {
|
||||
try {
|
||||
return crypto
|
||||
const text = crypto
|
||||
.privateDecrypt(
|
||||
{ key: privateKey, padding: crypto.constants.RSA_PKCS1_PADDING },
|
||||
encrypted,
|
||||
)
|
||||
.toString("utf8");
|
||||
} catch {
|
||||
const message = oaepError instanceof Error ? oaepError.message : String(oaepError);
|
||||
throw new Error(`Failed to decrypt Zed access token: ${message}`);
|
||||
// PKCS#1 v1.5 unpadding is not integrity-checked: a wrong-key decrypt
|
||||
// can "succeed" with garbage bytes instead of throwing. Replacement
|
||||
// characters prove the output is not the real UTF-8 token — fail loudly
|
||||
// rather than storing garbage as a credential.
|
||||
if (text.includes("<22>")) fail(oaepError);
|
||||
return text;
|
||||
} catch (err) {
|
||||
if (err.message.startsWith("Failed to decrypt Zed access token")) throw err;
|
||||
fail(oaepError);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -280,6 +297,7 @@ export async function fetchZedLlmToken(credentials, options = {}) {
|
||||
body: JSON.stringify({ organization_id: organizationId }),
|
||||
signal: options.signal ?? undefined,
|
||||
},
|
||||
options.proxyOptions ?? null,
|
||||
);
|
||||
const token =
|
||||
typeof data?.token === "string" ? data.token : data?.token?.[0] || data?.token?.value;
|
||||
|
||||
@@ -5,6 +5,20 @@ import {
|
||||
} from "../../config/kiroConstants.js";
|
||||
|
||||
const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/;
|
||||
|
||||
/**
|
||||
* Kiro rejects user turns with empty `content`, so a turn that only carries
|
||||
* tool results needs placeholder text. It must not read like a user
|
||||
* instruction: with "continue", models answer the word itself ("Nothing in
|
||||
* progress to continue") and drop the task they were in the middle of.
|
||||
*/
|
||||
export const KIRO_TOOL_RESULTS_PLACEHOLDER = "Tool results provided.";
|
||||
export const KIRO_EMPTY_USER_PLACEHOLDER = "continue";
|
||||
|
||||
/** Placeholder content for a user turn with no text of its own. */
|
||||
export function kiroEmptyUserContent(hasToolResults) {
|
||||
return hasToolResults ? KIRO_TOOL_RESULTS_PLACEHOLDER : KIRO_EMPTY_USER_PLACEHOLDER;
|
||||
}
|
||||
const TOOL_NAME_PATTERN = /[^a-zA-Z0-9_-]/g;
|
||||
|
||||
function clone(value) {
|
||||
@@ -34,7 +48,6 @@ function uniqueName(rawName, index, usedNames) {
|
||||
const cleaned = String(rawName || "")
|
||||
.trim()
|
||||
.replace(TOOL_NAME_PATTERN, "_")
|
||||
.replace(/_+/g, "_")
|
||||
.replace(/^_+|_+$/g, "");
|
||||
const base = trimCodePoints(cleaned || `tool_${index + 1}`, KIRO_TOOL_NAME_MAX_LENGTH);
|
||||
let candidate = base;
|
||||
@@ -174,7 +187,8 @@ function normalizeTurns(history, currentMessage, modelId) {
|
||||
|
||||
for (const turn of turns) {
|
||||
if (turn.userInputMessage) {
|
||||
turn.userInputMessage.content = text(turn.userInputMessage.content).trim() || "continue";
|
||||
turn.userInputMessage.content = text(turn.userInputMessage.content).trim()
|
||||
|| kiroEmptyUserContent(turn.userInputMessage.userInputMessageContext?.toolResults?.length > 0);
|
||||
turn.userInputMessage.modelId ||= modelId;
|
||||
if (turn.userInputMessage.userInputMessageContext?.tools) {
|
||||
delete turn.userInputMessage.userInputMessageContext.tools;
|
||||
|
||||
@@ -8,6 +8,7 @@ import { fetchImageAsBase64, parseDataUri } from "./image.js";
|
||||
const TARGETS_NEED_BASE64 = new Set([
|
||||
FORMATS.GEMINI, FORMATS.GEMINI_CLI, FORMATS.VERTEX,
|
||||
FORMATS.ANTIGRAVITY, FORMATS.OLLAMA, FORMATS.KIRO,
|
||||
FORMATS.COMMANDCODE,
|
||||
]);
|
||||
|
||||
function isRemoteUrl(url) {
|
||||
|
||||
@@ -19,6 +19,7 @@ const FORMAT_TO_NATIVE = {
|
||||
vertex: "gemini-budget",
|
||||
antigravity: "gemini-budget",
|
||||
kiro: "kiro",
|
||||
commandcode: "commandcode",
|
||||
};
|
||||
|
||||
// Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent).
|
||||
@@ -108,6 +109,7 @@ export const captureThinking = extractThinking;
|
||||
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
|
||||
|
||||
function resolveFormat(targetFormat, model, provider) {
|
||||
if (targetFormat === "commandcode") return "commandcode";
|
||||
const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null;
|
||||
if (providerFmt) return providerFmt;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
@@ -223,10 +225,14 @@ function stripAll(body) {
|
||||
delete body.output_config;
|
||||
if (body.generationConfig) delete body.generationConfig.thinkingConfig;
|
||||
if (body.request?.generationConfig) delete body.request.generationConfig.thinkingConfig;
|
||||
if (body.params && typeof body.params === "object") {
|
||||
delete body.params.reasoning_effort;
|
||||
delete body.params.thinking;
|
||||
}
|
||||
}
|
||||
|
||||
// Apply unified thinking config to body in the resolved provider-native format.
|
||||
function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
|
||||
const none = cfg.mode === "none";
|
||||
const canDisable = caps.thinkingCanDisable !== false;
|
||||
// Model cannot disable thinking → clamp "none" to minimal effort instead.
|
||||
@@ -243,7 +249,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
// Models that can disable thinking need the explicit adaptive switch.
|
||||
// Permanently adaptive models such as Fable 5.1 accept effort directly.
|
||||
if (canDisable) body.thinking = { type: "adaptive" };
|
||||
if (canDisable) body.thinking = { type: "adaptive", ...(display ? { display } : {}) };
|
||||
else delete body.thinking;
|
||||
const level = toLevel(eff);
|
||||
body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level };
|
||||
@@ -252,7 +258,7 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
case "claude-budget": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
const budget = toBudget(eff, caps.thinkingRange);
|
||||
body.thinking = budget === -1 ? { type: "enabled" } : { type: "enabled", budget_tokens: budget || 8192 };
|
||||
body.thinking = budget === -1 ? { type: "enabled", ...(display ? { display } : {}) } : { type: "enabled", budget_tokens: budget || 8192, ...(display ? { display } : {}) };
|
||||
break;
|
||||
}
|
||||
case "gemini-level": {
|
||||
@@ -336,6 +342,17 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
case "kiro":
|
||||
// Kiro thinking handled via system-tag injection in openai-to-kiro.js; no body field here.
|
||||
break;
|
||||
case "commandcode": {
|
||||
// Native CLI sends reasoning_effort inside params of the /alpha/generate envelope.
|
||||
if (!body.params || typeof body.params !== "object") body.params = {};
|
||||
if (none && canDisable) {
|
||||
delete body.params.reasoning_effort;
|
||||
break;
|
||||
}
|
||||
const level = toLevel(eff);
|
||||
if (level) body.params.reasoning_effort = level;
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -361,7 +378,10 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
|
||||
|
||||
const fmt = resolveFormat(targetFormat, cleanModel, provider);
|
||||
const supportedLevels = getThinkingLevels(provider, cleanModel);
|
||||
// Anthropic's `display` (summarized | omitted) decides whether thinking text
|
||||
// comes back at all; keep what the client asked for instead of resetting it.
|
||||
const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined;
|
||||
stripAll(body);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels, display);
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -415,6 +415,28 @@ export function anchorClaudeCache(body) {
|
||||
// - Add thinking block for Anthropic endpoint (provider === "claude")
|
||||
// - Fix tool_use/tool_result ordering
|
||||
// - Apply cloaking (billing header + fake user ID) for OAuth tokens
|
||||
export function hoistToolResultImages(body) {
|
||||
if (!Array.isArray(body?.messages)) return body;
|
||||
let touched = false;
|
||||
const messages = body.messages.map((msg) => {
|
||||
if (msg?.role !== ROLE.USER || !Array.isArray(msg.content)) return msg;
|
||||
const hoisted = [];
|
||||
const content = msg.content.map((block) => {
|
||||
if (block?.type !== CLAUDE_BLOCK.TOOL_RESULT || !Array.isArray(block.content)) return block;
|
||||
const images = block.content.filter((c) => c?.type === CLAUDE_BLOCK.IMAGE);
|
||||
if (!images.length) return block;
|
||||
const rest = block.content.filter((c) => c?.type !== CLAUDE_BLOCK.IMAGE);
|
||||
hoisted.push({ type: CLAUDE_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` }, ...images);
|
||||
return { ...block, content: rest.length ? rest : [{ type: CLAUDE_BLOCK.TEXT, text: "(image attached below)" }] };
|
||||
});
|
||||
if (!hoisted.length) return msg;
|
||||
touched = true;
|
||||
// tool_result blocks must lead a user message; the hoisted image follows them.
|
||||
return { ...msg, content: [...content, ...hoisted] };
|
||||
});
|
||||
return touched ? { ...body, messages } : body;
|
||||
}
|
||||
|
||||
export function prepareClaudeRequest(body, provider = null, apiKey = null, connectionId = null, rawHeaders = null, sessionId = null) {
|
||||
// quirk: MiniMax's Claude-compatible endpoint rejects Anthropic's output_config (400 invalid params)
|
||||
if (PROVIDERS[provider]?.quirks?.dropOutputConfig) {
|
||||
@@ -608,6 +630,14 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
|
||||
}
|
||||
}
|
||||
|
||||
// Anthropic itself reads images inside tool_result; other Anthropic-compatible
|
||||
// endpoints (OpenCode Go, Kimi, DeepSeek, GLM, MiniMax) accept image blocks
|
||||
// only as user content and silently drop them inside a tool result. Move a
|
||||
// tool's screenshot out of the result and into the same user turn.
|
||||
if (provider !== "claude" && !provider?.startsWith("anthropic-compatible")) {
|
||||
body = hoistToolResultImages(body);
|
||||
}
|
||||
|
||||
// Apply cloaking for OAuth tokens (billing header + fake user ID)
|
||||
// session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency
|
||||
if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) {
|
||||
|
||||
@@ -34,6 +34,7 @@ import { ROLE, CLAUDE_BLOCK } from "../schema/index.js";
|
||||
import {
|
||||
canonicalizeKiroConversation,
|
||||
normalizeKiroToolSpecs,
|
||||
kiroEmptyUserContent,
|
||||
} from "../concerns/kiroConversation.js";
|
||||
|
||||
/**
|
||||
@@ -53,7 +54,8 @@ function convertClaudeMessagesToKiro(messages, model) {
|
||||
|
||||
const flushPending = () => {
|
||||
if (currentRole === ROLE.USER) {
|
||||
const content = pendingUserContent.join("\n\n").trim() || "continue";
|
||||
const content = pendingUserContent.join("\n\n").trim()
|
||||
|| kiroEmptyUserContent(pendingToolResults.length > 0);
|
||||
const userMsg = { userInputMessage: { content, modelId: model } };
|
||||
|
||||
if (pendingImages.length > 0) {
|
||||
@@ -97,11 +99,21 @@ function convertClaudeMessagesToKiro(messages, model) {
|
||||
if (typeof block.content === "string") {
|
||||
resultContent = block.content;
|
||||
} else if (Array.isArray(block.content)) {
|
||||
// Images a tool returned (screenshots) ride along as user images;
|
||||
// Kiro tool results are text-only.
|
||||
let hasImage = false;
|
||||
for (const c of block.content) {
|
||||
if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") {
|
||||
hasImage = true;
|
||||
const imageType = c.source.media_type || DEFAULT_IMAGE_MIME;
|
||||
pendingImages.push({ format: imageType.split("/")[1] || imageType, source: { bytes: c.source.data } });
|
||||
}
|
||||
}
|
||||
resultContent =
|
||||
block.content
|
||||
.filter((c) => c.type === CLAUDE_BLOCK.TEXT)
|
||||
.map((c) => c.text)
|
||||
.join("\n") || JSON.stringify(block.content);
|
||||
.join("\n") || (hasImage ? "(image attached)" : JSON.stringify(block.content));
|
||||
} else if (block.content) {
|
||||
resultContent = JSON.stringify(block.content);
|
||||
}
|
||||
@@ -341,6 +353,13 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
|
||||
enumerable: false,
|
||||
});
|
||||
|
||||
// Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the
|
||||
// reverse map so tool calls stream back under the client's own names.
|
||||
const restoredToolNames = new Map();
|
||||
for (const [original, sanitized] of nameMap) {
|
||||
if (original !== sanitized) restoredToolNames.set(sanitized, original);
|
||||
}
|
||||
if (restoredToolNames.size) payload._toolNameMap = restoredToolNames;
|
||||
return payload;
|
||||
}
|
||||
|
||||
|
||||
@@ -196,25 +196,41 @@ function convertClaudeMessage(msg) {
|
||||
});
|
||||
break;
|
||||
|
||||
case CLAUDE_BLOCK.TOOL_RESULT:
|
||||
case CLAUDE_BLOCK.TOOL_RESULT: {
|
||||
let resultContent = "";
|
||||
const resultImages = [];
|
||||
if (typeof block.content === "string") {
|
||||
resultContent = block.content;
|
||||
} else if (Array.isArray(block.content)) {
|
||||
resultContent = block.content
|
||||
.filter(c => c.type === CLAUDE_BLOCK.TEXT)
|
||||
.map(c => c.text)
|
||||
.join("\n") || JSON.stringify(block.content);
|
||||
for (const c of block.content) {
|
||||
if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") {
|
||||
resultImages.push({
|
||||
type: OPENAI_BLOCK.IMAGE_URL,
|
||||
image_url: { url: encodeDataUri(c.source.media_type, c.source.data) }
|
||||
});
|
||||
}
|
||||
}
|
||||
const textOnly = block.content.filter(c => c?.type === CLAUDE_BLOCK.TEXT);
|
||||
resultContent = textOnly.map(c => c.text).join("\n")
|
||||
|| (resultImages.length ? "" : JSON.stringify(block.content));
|
||||
} else if (block.content) {
|
||||
resultContent = JSON.stringify(block.content);
|
||||
}
|
||||
|
||||
|
||||
toolResults.push({
|
||||
role: ROLE.TOOL,
|
||||
tool_call_id: block.tool_use_id,
|
||||
content: resultContent
|
||||
});
|
||||
// The OpenAI tool role is text-only, so a screenshot or any other image a
|
||||
// tool returned would otherwise vanish. Hand it to the model in the user
|
||||
// turn that follows the tool messages, tagged with the call it came from.
|
||||
if (resultImages.length) {
|
||||
parts.push({ type: OPENAI_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` });
|
||||
parts.push(...resultImages);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
* - params.system: STRING at top level (Anthropic-style; system messages NOT allowed in messages[])
|
||||
* - params.messages[*].role ∈ {"user","assistant","tool"}
|
||||
* - params.messages[*].content: Array of content blocks (NEVER a string)
|
||||
* - image_url / image source → {type:"image", image:"data:...;base64,...", mimeType}
|
||||
* - tool_use blocks (assistant): {type:"tool-call", toolCallId, toolName, input}
|
||||
* - tool_result blocks (role=user): {type:"tool-result", toolCallId, toolName, output}
|
||||
* - tools[*]: Anthropic plain {name, description, input_schema}
|
||||
@@ -12,10 +13,9 @@
|
||||
import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { randomUUID } from "crypto";
|
||||
import { ROLE, OPENAI_BLOCK } from "../schema/index.js";
|
||||
import { DEFAULT_IMAGE_MIME } from "../schema/index.js";
|
||||
import { parseDataUri } from "../concerns/image.js";
|
||||
import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js";
|
||||
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
|
||||
import { parseDataUri, encodeDataUri } from "../concerns/image.js";
|
||||
|
||||
function flattenText(content) {
|
||||
if (content == null) return "";
|
||||
@@ -32,6 +32,58 @@ function flattenText(content) {
|
||||
return String(content);
|
||||
}
|
||||
|
||||
function toNativeImageBlock(part) {
|
||||
if (!part || typeof part !== "object") return null;
|
||||
|
||||
if (part.type === OPENAI_BLOCK.IMAGE_URL) {
|
||||
const url = typeof part.image_url === "string" ? part.image_url : part.image_url?.url;
|
||||
if (!url) return null;
|
||||
const parsed = parseDataUri(url);
|
||||
if (parsed) {
|
||||
return {
|
||||
type: OPENAI_BLOCK.IMAGE,
|
||||
image: encodeDataUri(parsed.mimeType, parsed.base64),
|
||||
mimeType: parsed.mimeType,
|
||||
};
|
||||
}
|
||||
if (typeof url === "string" && (url.startsWith("http://") || url.startsWith("https://"))) {
|
||||
return {
|
||||
type: OPENAI_BLOCK.IMAGE,
|
||||
image: url,
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
if (part.type === OPENAI_BLOCK.IMAGE || part.type === CLAUDE_BLOCK.IMAGE) {
|
||||
if (typeof part.image === "string" && part.image.startsWith("data:")) {
|
||||
const parsed = parseDataUri(part.image);
|
||||
return {
|
||||
type: OPENAI_BLOCK.IMAGE,
|
||||
image: part.image,
|
||||
mimeType: part.mimeType || parsed?.mimeType || "image/png",
|
||||
};
|
||||
}
|
||||
if (typeof part.image === "string" && (part.image.startsWith("http://") || part.image.startsWith("https://"))) {
|
||||
return {
|
||||
type: OPENAI_BLOCK.IMAGE,
|
||||
image: part.image,
|
||||
};
|
||||
}
|
||||
const source = part.source;
|
||||
if (source?.type === "base64" && typeof source.data === "string") {
|
||||
const mime = source.media_type || "image/png";
|
||||
return {
|
||||
type: OPENAI_BLOCK.IMAGE,
|
||||
image: encodeDataUri(mime, source.data),
|
||||
mimeType: mime,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function toContentBlocks(content) {
|
||||
if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }];
|
||||
if (typeof content === "string")
|
||||
@@ -44,30 +96,12 @@ function toContentBlocks(content) {
|
||||
} else if (part && typeof part === "object") {
|
||||
if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
} else if (
|
||||
part.type === OPENAI_BLOCK.IMAGE_URL ||
|
||||
part.type === OPENAI_BLOCK.IMAGE
|
||||
) {
|
||||
// CommandCode `/alpha/generate` accepts {type:"image", image:"<data URI | url>"} —
|
||||
// same shape the official command-code CLI sends (verified from CLI source).
|
||||
const src = part.source;
|
||||
let raw = part.image_url?.url || src?.data || src?.url || "";
|
||||
let parsed = parseDataUri(raw);
|
||||
if (!parsed && src?.type === "base64" && src?.data) {
|
||||
// Claude-style base64 source without a data-URI prefix → wrap it.
|
||||
raw = `data:${src.media_type || DEFAULT_IMAGE_MIME};base64,${src.data}`;
|
||||
parsed = parseDataUri(raw);
|
||||
} else {
|
||||
const image = toNativeImageBlock(part);
|
||||
if (image) blocks.push(image);
|
||||
else if (typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
}
|
||||
if (parsed) {
|
||||
blocks.push({
|
||||
type: "image",
|
||||
image: `data:${parsed.mimeType};base64,${parsed.base64}`,
|
||||
});
|
||||
} else if (raw) {
|
||||
blocks.push({ type: "image", image: raw });
|
||||
}
|
||||
} else if (typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -119,6 +153,10 @@ function convertMessages(messages = []) {
|
||||
|
||||
if (role === ROLE.ASSISTANT) {
|
||||
const blocks = [];
|
||||
const rc = m.reasoning_content || m.thought || m.reasoning;
|
||||
if (rc || (Array.isArray(m.tool_calls) && m.tool_calls.length > 0)) {
|
||||
blocks.push({ type: "reasoning", text: rc || " " });
|
||||
}
|
||||
const text = flattenText(m.content);
|
||||
if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
|
||||
if (Array.isArray(m.tool_calls)) {
|
||||
|
||||
@@ -129,7 +129,7 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
|
||||
if (tc.type !== OPENAI_BLOCK.FUNCTION) continue;
|
||||
|
||||
const args = tryParseJSON(tc.function?.arguments || "{}");
|
||||
const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null;
|
||||
const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId, model) : null;
|
||||
// First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig
|
||||
const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined);
|
||||
firstFunctionCallSeen = true;
|
||||
@@ -341,7 +341,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
|
||||
if (block.type === CLAUDE_BLOCK.TEXT) {
|
||||
parts.push({ text: block.text });
|
||||
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||
const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null;
|
||||
const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId, model) : null;
|
||||
const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined);
|
||||
firstToolUseSeen = true;
|
||||
|
||||
|
||||
@@ -23,6 +23,7 @@ import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js";
|
||||
import {
|
||||
canonicalizeKiroConversation,
|
||||
normalizeKiroToolSpecs,
|
||||
kiroEmptyUserContent,
|
||||
} from "../concerns/kiroConversation.js";
|
||||
|
||||
/**
|
||||
@@ -51,7 +52,8 @@ function convertMessages(messages, model) {
|
||||
|
||||
const flushPending = () => {
|
||||
if (currentRole === "user") {
|
||||
const content = pendingUserContent.join("\n\n").trim() || "continue";
|
||||
const content = pendingUserContent.join("\n\n").trim()
|
||||
|| kiroEmptyUserContent(pendingToolResults.length > 0);
|
||||
const userMsg = {
|
||||
userInputMessage: {
|
||||
content: content,
|
||||
@@ -434,6 +436,13 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
|
||||
enumerable: false
|
||||
});
|
||||
|
||||
// Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the
|
||||
// reverse map so tool calls stream back under the client's own names.
|
||||
const restoredToolNames = new Map();
|
||||
for (const [original, sanitized] of nameMap) {
|
||||
if (original !== sanitized) restoredToolNames.set(sanitized, original);
|
||||
}
|
||||
if (restoredToolNames.size) payload._toolNameMap = restoredToolNames;
|
||||
return payload;
|
||||
}
|
||||
|
||||
|
||||
@@ -183,27 +183,12 @@ export function commandCodeToOpenAIResponse(chunk, state) {
|
||||
break;
|
||||
}
|
||||
case "error": {
|
||||
// Terminal upstream failure (AI SDK v5 error event) — NOT content. Emit an
|
||||
// OpenAI-shaped error chunk (chunk.error) so downstream — parseSSEToOpenAIResponse
|
||||
// for non-streaming, OpenAI SDK clients for streaming — treats the request as
|
||||
// failed instead of surfacing fake success content like "[CommandCode error: ...]".
|
||||
state.finishReason = OPENAI_FINISH.STOP;
|
||||
const errVal = event.error ?? event.message ?? "unknown";
|
||||
const errStr =
|
||||
typeof errVal === "string"
|
||||
? errVal
|
||||
: typeof errVal?.message === "string"
|
||||
? errVal.message
|
||||
: JSON.stringify(errVal);
|
||||
const errType =
|
||||
typeof errVal === "string"
|
||||
? "upstream_error"
|
||||
: errVal?.type || "upstream_error";
|
||||
const errChunk = makeChunk(state, {});
|
||||
errChunk.error = { message: errStr, type: errType };
|
||||
out.push(errChunk);
|
||||
out.push(makeChunk(state, {}, OPENAI_FINISH.STOP));
|
||||
break;
|
||||
typeof errVal === "string" ? errVal : JSON.stringify(errVal);
|
||||
// Mid-stream error: throw rather than emitting as fake content with finish_reason: "stop"
|
||||
// This ensures the downstream stream handler marks the stream as errored/aborted.
|
||||
throw new Error(`[CommandCode error: ${errStr}]`);
|
||||
}
|
||||
// Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end,
|
||||
// provider-metadata, message-metadata, etc. They carry no client-visible content.
|
||||
|
||||
@@ -22,7 +22,7 @@ function emitFunctionCall(functionCall, state, signature = null) {
|
||||
const toolCallIndex = state.functionIndex++;
|
||||
const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`;
|
||||
if (signature) {
|
||||
storeGeminiThoughtSignature(callId, signature, state.sessionId);
|
||||
storeGeminiThoughtSignature(callId, signature, state.sessionId, state.model);
|
||||
}
|
||||
const toolCall = {
|
||||
id: callId,
|
||||
@@ -52,7 +52,7 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
// Initialize state
|
||||
if (!state.messageId) {
|
||||
state.messageId = response.responseId || `msg_${Date.now()}`;
|
||||
state.model = response.modelVersion || "gemini";
|
||||
state.model = response.modelVersion || state.model || "gemini";
|
||||
state.functionIndex = 0;
|
||||
state.geminiToolCallCount = 0;
|
||||
results.push(buildChunk(chunkMeta(state), { role: ROLE.ASSISTANT }, null));
|
||||
|
||||
@@ -46,6 +46,14 @@ function convertFinishReason(reason) {
|
||||
* Convert one OpenAI-format chunk (from KiroExecutor) into Claude SSE events.
|
||||
* Returns an array of Claude events, or null when the chunk yields nothing.
|
||||
*/
|
||||
// Kiro only accepts sanitized tool names; the request translator leaves the
|
||||
// reverse map on the stream state so calls come back under the client's names.
|
||||
function restoreToolName(stateOrData, name) {
|
||||
const raw = name || "";
|
||||
const map = stateOrData?.toolNameMap || stateOrData?._toolNameMap;
|
||||
return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw;
|
||||
}
|
||||
|
||||
export function kiroToClaudeResponse(chunk, state) {
|
||||
// KiroExecutor emits chat.completion.chunk objects; tolerate string chunks
|
||||
// by attempting a parse (defensive — the direct path is always objects).
|
||||
@@ -161,7 +169,7 @@ export function kiroToClaudeResponse(chunk, state) {
|
||||
const toolBlockIndex = state.nextBlockIndex++;
|
||||
state.toolCalls.set(idx, {
|
||||
id: tc.id,
|
||||
name: tc.function?.name || "",
|
||||
name: restoreToolName(state, tc.function?.name),
|
||||
blockIndex: toolBlockIndex,
|
||||
});
|
||||
results.push({
|
||||
@@ -170,7 +178,7 @@ export function kiroToClaudeResponse(chunk, state) {
|
||||
content_block: {
|
||||
type: "tool_use",
|
||||
id: tc.id,
|
||||
name: tc.function?.name || "",
|
||||
name: restoreToolName(state, tc.function?.name),
|
||||
input: {},
|
||||
},
|
||||
});
|
||||
@@ -246,7 +254,7 @@ export function kiroToClaudeNonStreaming(data) {
|
||||
content.push({
|
||||
type: "tool_use",
|
||||
id: tc.id || `toolu_${Date.now()}`,
|
||||
name: tc.function?.name || "",
|
||||
name: restoreToolName(data, tc.function?.name),
|
||||
input,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -20,13 +20,38 @@ function chunkMeta(state) {
|
||||
* Parse Kiro SSE event and convert to OpenAI format
|
||||
* Kiro events: assistantResponseEvent, codeEvent, supplementaryWebLinksEvent, etc.
|
||||
*/
|
||||
// Kiro only accepts sanitized tool names; the request translator leaves the
|
||||
// reverse map on the stream state so calls come back under the client's names.
|
||||
function restoreToolName(state, name) {
|
||||
const raw = name || "";
|
||||
const map = state?.toolNameMap;
|
||||
return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw;
|
||||
}
|
||||
|
||||
export function kiroToOpenAIResponse(chunk, state) {
|
||||
|
||||
if (!chunk) return null;
|
||||
|
||||
// If chunk is already in OpenAI format (from executor transform), return as-is
|
||||
// If chunk is already in OpenAI format (from executor transform), return it
|
||||
// with the client's tool names restored.
|
||||
if (chunk.object === "chat.completion.chunk" && chunk.choices) {
|
||||
return chunk;
|
||||
if (!state?.toolNameMap?.size) return chunk;
|
||||
return {
|
||||
...chunk,
|
||||
choices: chunk.choices.map((choice) => {
|
||||
const calls = choice?.delta?.tool_calls;
|
||||
if (!Array.isArray(calls)) return choice;
|
||||
return {
|
||||
...choice,
|
||||
delta: {
|
||||
...choice.delta,
|
||||
tool_calls: calls.map((tc) => tc?.function?.name
|
||||
? { ...tc, function: { ...tc.function, name: restoreToolName(state, tc.function.name) } }
|
||||
: tc),
|
||||
},
|
||||
};
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
// Handle string chunk (raw SSE data)
|
||||
@@ -109,7 +134,7 @@ export function kiroToOpenAIResponse(chunk, state) {
|
||||
state.hadToolUse = true;
|
||||
const toolUse = data.toolUseEvent || data;
|
||||
const toolCallId = toolUse.toolUseId || fallbackToolCallId();
|
||||
const toolName = toolUse.name || "";
|
||||
const toolName = restoreToolName(state, toolUse.name);
|
||||
const toolInput = toolUse.input || {};
|
||||
|
||||
const openaiChunk = buildChunk(chunkMeta(state), {
|
||||
|
||||
@@ -95,6 +95,9 @@ export function createStreamController({ onDisconnect, onError, log, provider, m
|
||||
* activity), not here — output of the transform stream may be silent
|
||||
* for long periods while raw bytes still flow (e.g. Kiro EventStream
|
||||
* binary frames buffering, Claude reasoning streams).
|
||||
*
|
||||
* @param {function} [onAbortTerminal] - Receives a human-readable abort
|
||||
* message and returns terminal SSE bytes to emit downstream.
|
||||
*/
|
||||
export function createDisconnectAwareStream(transformStream, streamController, onAbortTerminal = null) {
|
||||
const reader = transformStream.readable.getReader();
|
||||
@@ -194,6 +197,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont
|
||||
let chunkCount = 0;
|
||||
let totalBytes = 0;
|
||||
let lastChunkAt = Date.now();
|
||||
let abortMessage = "upstream connection lost";
|
||||
const t0 = Date.now();
|
||||
const tag = "STREAM";
|
||||
const clearStall = () => {
|
||||
@@ -203,6 +207,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont
|
||||
clearStall();
|
||||
stallTimer = setTimeout(() => {
|
||||
stallTimer = null;
|
||||
abortMessage = "stream stall timeout";
|
||||
dbg(tag, `STALL TIMEOUT ${stallTimeoutMs}ms | chunks=${chunkCount} | bytes=${totalBytes} | sinceLast=${Date.now() - lastChunkAt}ms`);
|
||||
streamController.handleError?.(new Error("stream stall timeout"));
|
||||
streamController.abort?.();
|
||||
@@ -249,7 +254,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont
|
||||
return createDisconnectAwareStream(
|
||||
{ readable: transformedBody, writable: { getWriter: () => ({ abort: () => Promise.resolve() }) } },
|
||||
wrappedController,
|
||||
onAbortTerminal
|
||||
onAbortTerminal ? () => onAbortTerminal(abortMessage) : null
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,8 @@
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { buildErrorBody } from "./error.js";
|
||||
import { SSE_DONE } from "./sseConstants.js";
|
||||
|
||||
const sharedEncoder = new TextEncoder();
|
||||
|
||||
// Parse SSE data line
|
||||
export function parseSSELine(line, format = null) {
|
||||
@@ -120,3 +124,24 @@ export function formatSSE(data, sourceFormat) {
|
||||
|
||||
return `data: ${JSON.stringify(data)}\n\n`;
|
||||
}
|
||||
|
||||
// Terminal frames for a stream that aborted after HTTP 200 was already sent, so
|
||||
// the status code can no longer change. OpenAI-compatible clients (openai-python
|
||||
// raises APIError on any `data:` payload carrying an `error` key, checked before
|
||||
// [DONE]) need the error frame first, then [DONE]; Anthropic clients need
|
||||
// `event: error`. Never fabricate a successful finish_reason instead.
|
||||
//
|
||||
// Returns encoded bytes: onAbortTerminal callbacks are enqueued verbatim, same
|
||||
// as buildAbortedResponsesTerminalBytes.
|
||||
//
|
||||
// NOTE: non-SSE client formats (Ollama NDJSON) get an SSE frame here — dead in
|
||||
// practice because detectFormatByEndpoint never resolves to OLLAMA.
|
||||
export function buildStreamErrorBytes(statusCode, message, clientFormat) {
|
||||
const { error } = buildErrorBody(statusCode, message);
|
||||
|
||||
const sse = clientFormat === FORMATS.CLAUDE
|
||||
? formatSSE({ type: "error", error }, FORMATS.CLAUDE)
|
||||
: formatSSE({ error }, clientFormat) + SSE_DONE;
|
||||
|
||||
return sharedEncoder.encode(sse);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user