Merge origin/master (v0.5.69) into gitea/new_feature

This commit is contained in:
2026-09-07 14:10:11 +07:00
222 changed files with 12285 additions and 3371 deletions

View File

@@ -171,6 +171,13 @@ export const LOAD_CODE_ASSIST_METADATA = {
// System prompts
export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official CLI for Claude.";
// Rewrite rules applied to Antigravity system prompts: competing-client branding
// makes the backend flag the request and answer 429 Quota Exhausted.
export const ANTIGRAVITY_PROMPT_REWRITES = [
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
];
export const ANTIGRAVITY_DEFAULT_SYSTEM = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.**Absolute paths only****Proactiveness**";
// Derive từ registry oauth.refreshLeadMs

View File

@@ -3,7 +3,8 @@ import REGISTRY from "../providers/registry/index.js";
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
import { PROVIDER_MODELS } from "../providers/index.js";
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js";
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel } from "../providers/models/helpers.js";
import { FORMATS } from "../translator/formats.js";
export { PROVIDER_MODELS };
@@ -49,6 +50,9 @@ export function findModelName(aliasOrId, modelId) {
}
export function getModelTargetFormat(aliasOrId, modelId) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) {
return FORMATS.OPENAI_RESPONSES;
}
const models = PROVIDER_MODELS[aliasOrId];
if (!models) return null;
return modelTargetFormat(findModel(models, modelId, aliasOrId));

View File

@@ -1,12 +1,13 @@
import crypto from "crypto";
import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js";
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
function sanitizeFunctionName(name) {
@@ -187,6 +188,9 @@ export class AntigravityExecutor extends BaseExecutor {
};
}
const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" });
const sessionId = toNumericSessionId(rawSessionId) || rawSessionId;
// ─── Standard (non-image) request ───
// Fix contents for Claude models via Antigravity
const contents = body.request?.contents?.map(c => {
@@ -202,17 +206,31 @@ export class AntigravityExecutor extends BaseExecutor {
return true;
});
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
// don't persist thoughtSignature in their history, so backfill the default signature on any
// functionCall part that arrives without one.
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
// don't persist thoughtSignature in their history, so backfill from cache or default signature.
// In parallel function calls, only the first call needs a signature; siblings stay unsigned.
let firstFunctionCallSeen = false;
const modifiedParts = parts?.map(p => {
if (!p.functionCall) return p;
const callId = p.functionCall.id;
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null;
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
firstFunctionCallSeen = true;
if (callSig) {
return { ...p, thoughtSignature: callSig };
}
if (p.thoughtSignature && !cachedSig) {
// Unsigned sibling call
const { thoughtSignature: _, ...rest } = p;
return rest;
}
return p;
});
const partsChanged = parts?.length !== c.parts?.length || modifiedParts?.some((p, idx) => p !== c.parts[idx]);
if (role !== c.role || partsChanged) {
return {
...c, role,
parts: needsBackfill
? parts.map(p => (p.functionCall && !p.thoughtSignature)
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
: p)
: parts,
parts: modifiedParts || parts,
};
}
return c;
@@ -246,13 +264,13 @@ export class AntigravityExecutor extends BaseExecutor {
const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {};
stripBlacklisted(requestWithoutTools);
// Rewrite competitive system prompts (e.g. Zed IDE's Claude prompt) to prevent Antigravity from
// flagging the request and immediately blocking it with a 429 Quota Exhausted response.
// Rewrite competing-client branding in system prompts (e.g. Zed's Claude prompt,
// OpenCode naming) so Antigravity doesn't flag the request with a 429 Quota Exhausted.
if (requestWithoutTools.systemInstruction?.parts) {
const oldText = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
for (const part of requestWithoutTools.systemInstruction.parts) {
if (typeof part.text === "string" && part.text.includes(oldText)) {
part.text = part.text.split(oldText).join("");
if (typeof part.text !== "string") continue;
for (const { from, to } of ANTIGRAVITY_PROMPT_REWRITES) {
part.text = part.text.replaceAll(from, to);
}
}
}
@@ -267,7 +285,7 @@ export class AntigravityExecutor extends BaseExecutor {
generationConfig,
...(contents && { contents }),
...(tools && { tools }),
sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }),
sessionId,
safetySettings: undefined,
...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } })
};

View File

@@ -46,203 +46,242 @@ export class CommandCodeExecutor extends BaseExecutor {
return headers;
}
async execute(opts) {
const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result;
result.response = await peekForUpstreamError(result.response, opts.model, {
signal: opts.signal,
});
return result;
}
async execute(opts) {
const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result;
result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
return result;
}
parseError(response, bodyText) {
let parsed = null;
try {
parsed = JSON.parse(bodyText || "{}");
} catch {
parsed = null;
}
const errObj = parsed?.error || parsed;
const msg = errObj?.message || parsed?.message || bodyText || response.statusText;
const status = Number(errObj?.code || errObj?.statusCode || response.status) || response.status;
return {
status,
message: msg || `CommandCode upstream error: ${response.status}`,
};
}
}
// How long to hold the response open while peeking the first upstream events.
// An upstream error event ("Network connection lost") is emitted at stream
// start, so the peek is fast; the bound just prevents a slow-started stream
// from being held hostage. Env: COMMANDCODE_PEEK_TIMEOUT_MS.
const PEEK_TIMEOUT_MS = (() => {
const raw = process.env.COMMANDCODE_PEEK_TIMEOUT_MS;
const n = raw ? parseInt(raw, 10) : NaN;
return Number.isFinite(n) && n > 0 ? n : 10 * 1000;
})();
export function parseCommandCodeError(event) {
if (!event || typeof event !== "object") {
return {
statusCode: 503,
message: "CommandCode upstream error",
type: "server_error",
};
}
// Event types that count as "the stream has started producing". Everything
// else (start, start-step, reasoning-start, text-start, ...) is metadata and
// does not end the peek.
const MEANINGFUL_EVENT_TYPES = new Set([
"text-delta",
"reasoning-delta",
"tool-input-start",
"tool-input-delta",
"tool-input-end",
"tool-call",
"finish-step",
"finish",
]);
const errVal = event.error ?? event.message ?? "unknown";
let message = "";
let statusCode = null;
let type = "server_error";
function makeAbortError(reason) {
const error = new Error(reason?.message || reason || "Request aborted");
error.name = "AbortError";
return error;
if (typeof errVal === "object" && errVal !== null) {
message = errVal.message || errVal.error || JSON.stringify(errVal);
if (errVal.statusCode && Number.isInteger(Number(errVal.statusCode))) {
statusCode = Number(errVal.statusCode);
} else if (errVal.status && Number.isInteger(Number(errVal.status))) {
statusCode = Number(errVal.status);
}
if (errVal.type) type = errVal.type;
} else if (typeof errVal === "string") {
message = errVal;
} else {
message = JSON.stringify(errVal);
}
if (event.statusCode && Number.isInteger(Number(event.statusCode))) {
statusCode = Number(event.statusCode);
}
if (!statusCode || statusCode < 400 || statusCode > 599) {
const lower = message.toLowerCase();
if (lower.includes("rate limit") || lower.includes("too many requests")) {
statusCode = 429;
type = "rate_limit_error";
} else if (lower.includes("unauthorized") || lower.includes("invalid api key") || lower.includes("authentication")) {
statusCode = 401;
type = "authentication_error";
} else if (lower.includes("payment required") || lower.includes("billing")) {
statusCode = 402;
type = "billing_error";
} else if (lower.includes("quota") || lower.includes("forbidden") || lower.includes("permission")) {
statusCode = 403;
type = "permission_error";
} else if (lower.includes("not found")) {
statusCode = 404;
type = "invalid_request_error";
} else if (lower.includes("unavailable") || lower.includes("overloaded") || lower.includes("server error")) {
statusCode = 503;
type = "server_error";
} else {
statusCode = 503;
}
}
return { statusCode, message, type };
}
function tryParseEvent(line) {
const trimmed = line.trim();
if (!trimmed) return null;
const json = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
if (!json || json === "[DONE]") return null;
try {
return JSON.parse(json);
} catch {
return null;
}
export async function inspectAndWrapCommandCodeResponse(originalResponse, model) {
const reader = originalResponse.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
const bufferedLines = [];
let detectedError = null;
try {
while (true) {
const { value, done } = await reader.read();
if (done) {
const trimmed = buffer.trim();
if (trimmed) {
try {
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
const parsed = JSON.parse(jsonStr);
if (parsed?.type === "error") {
detectedError = parsed;
} else {
bufferedLines.push(trimmed);
}
} catch {
bufferedLines.push(trimmed);
}
}
break;
}
buffer += decoder.decode(value, { stream: true });
const lines = buffer.split("\n");
buffer = lines.pop() || "";
let stopLoop = false;
for (const line of lines) {
const trimmed = line.trim();
if (!trimmed) continue;
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
if (!jsonStr || jsonStr === "[DONE]") {
bufferedLines.push(trimmed);
stopLoop = true;
break;
}
let event;
try {
event = JSON.parse(jsonStr);
} catch {
bufferedLines.push(trimmed);
continue;
}
if (event?.type === "error") {
detectedError = event;
stopLoop = true;
break;
}
bufferedLines.push(trimmed);
if (
event?.type === "text-delta" ||
event?.type === "reasoning-delta" ||
event?.type === "tool-input-start" ||
event?.type === "tool-call" ||
event?.type === "finish" ||
event?.type === "finish-step"
) {
stopLoop = true;
break;
}
}
if (stopLoop) break;
}
} catch {
try { reader.releaseLock(); } catch { /* ignore */ }
return originalResponse;
}
if (detectedError) {
try { await reader.cancel(); } catch { /* ignore */ }
const { statusCode, message, type } = parseCommandCodeError(detectedError);
return new Response(
JSON.stringify({
error: {
message: `[CommandCode error: ${message}]`,
type,
code: statusCode,
},
}),
{
status: statusCode,
statusText: statusCode === 503 ? "Service Unavailable" : (statusCode === 429 ? "Too Many Requests" : "Bad Gateway"),
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
},
}
);
}
const combinedStream = createReplayedStream(bufferedLines, buffer, reader);
return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse);
}
function formatErrorValue(errVal) {
const errStr =
typeof errVal === "string"
? errVal
: typeof errVal?.message === "string"
? errVal.message
: JSON.stringify(errVal);
const errType =
typeof errVal === "string"
? "upstream_error"
: errVal?.type || "upstream_error";
return { message: errStr, type: errType };
function createReplayedStream(bufferedLines, remainingBuffer, reader) {
const encoder = new TextEncoder();
let replayed = false;
return new ReadableStream({
async pull(controller) {
if (!replayed) {
replayed = true;
let prefix = bufferedLines.join("\n");
if (prefix && remainingBuffer) {
prefix += "\n" + remainingBuffer;
} else if (remainingBuffer) {
prefix = remainingBuffer;
} else if (prefix) {
prefix += "\n";
}
if (prefix) {
controller.enqueue(encoder.encode(prefix));
}
}
try {
const { value, done } = await reader.read();
if (done) {
controller.close();
} else {
controller.enqueue(value);
}
} catch (err) {
controller.error(err);
}
},
async cancel(reason) {
try {
await reader.cancel(reason);
} catch {
/* ignore */
}
},
});
}
/**
* Read the first upstream events before committing the response.
*
* - `{"type":"error"}` as the first meaningful event → return a 502 Response so
* chatCore's `!response.ok` path parses the error and triggers fallback.
* - Otherwise → re-emit the buffered bytes + the rest of the stream through the
* normal NDJSON → OpenAI SSE wrapper and return it untouched in spirit.
*
* Bounded by `timeoutMs` (default PEEK_TIMEOUT_MS): if no meaningful event
* arrives in time, or the request signal aborts, we commit whatever we have and
* let the regular stream pipeline (stall detection, abort handling) take over.
*/
export async function peekForUpstreamError(
originalResponse,
model,
{ signal = null, timeoutMs = PEEK_TIMEOUT_MS } = {},
) {
const reader = originalResponse.body.getReader();
const decoder = new TextDecoder();
const abortController = new AbortController();
const forwardAbort = () => abortController.abort(signal?.reason);
if (signal?.aborted) abortController.abort(signal?.reason);
else if (signal)
signal.addEventListener("abort", forwardAbort, { once: true });
// Raw bytes for lossless re-emission; decoded text is only used for line
// parsing / error detection. Never re-encode decoded text: TextDecoder
// holds a split multi-byte char internally and flush() would replace it
// with U+FFFD, corrupting the stream.
const rawChunks = [];
let peeked = "";
let errorEvent = null;
let committed = false;
const readWithTimeout = (ms) => {
if (abortController.signal.aborted) {
return Promise.reject(makeAbortError(abortController.signal.reason));
}
const timeoutPromise = new Promise((_, reject) => {
const t = setTimeout(() => reject(new Error("peek timeout")), ms);
t.unref?.();
});
const abortPromise = new Promise((_, reject) => {
abortController.signal.addEventListener(
"abort",
() => reject(makeAbortError(abortController.signal.reason)),
{ once: true },
);
});
return Promise.race([reader.read(), timeoutPromise, abortPromise]);
};
try {
const deadline = Date.now() + timeoutMs;
while (!errorEvent && !committed && Date.now() < deadline) {
const { done, value } = await readWithTimeout(
Math.max(deadline - Date.now(), 1),
);
if (done) break;
rawChunks.push(value);
peeked += decoder.decode(value, { stream: true });
const lines = peeked.split("\n");
// The last segment may be a partial line — only parse complete ones.
for (const line of lines.slice(0, -1)) {
const event = tryParseEvent(line);
if (!event?.type) continue;
if (event.type === "error") {
errorEvent = event;
break;
}
if (MEANINGFUL_EVENT_TYPES.has(event.type)) {
committed = true;
break;
}
}
}
} catch {
// timeout / abort / read failure during the peek → commit whatever we have;
// the downstream stream pipeline (stall detection, abort handling) takes over.
}
if (signal) signal.removeEventListener("abort", forwardAbort);
if (errorEvent) {
await reader.cancel("commandcode early error detected").catch(() => {});
const { message, type } = formatErrorValue(
errorEvent.error ?? errorEvent.message ?? "unknown",
);
return new Response(JSON.stringify({ error: { message, type } }), {
status: HTTP_STATUS.BAD_GATEWAY,
statusText: message.slice(0, 200),
headers: { "Content-Type": "application/json" },
});
}
const remaining = new ReadableStream({
start(controller) {
(async () => {
try {
// Re-emit RAW bytes (never re-encoded decoded text) so split
// multi-byte UTF-8 sequences survive the peek untouched.
for (const c of rawChunks) controller.enqueue(c);
while (true) {
const { done, value } = await reader.read();
if (done) break;
controller.enqueue(value);
}
controller.close();
} catch (err) {
controller.error(err);
}
})();
},
cancel() {
reader.cancel("commandcode stream cancelled").catch(() => {});
},
});
const combined = new Response(remaining, {
status: originalResponse.status,
statusText: originalResponse.statusText,
headers: originalResponse.headers,
});
return wrapNdjsonAsOpenAISse(combined, model);
}
function wrapNdjsonAsOpenAISse(originalResponse, model) {
const decoder = new TextDecoder();
const encoder = new TextEncoder();
let buffer = "";
const state = { model };
function wrapNdjsonAsOpenAISse(streamBody, model, originalResponse = null) {
const decoder = new TextDecoder();
const encoder = new TextEncoder();
let buffer = "";
const state = { model };
const emitChunks = (chunks, controller) => {
if (!chunks) return;
@@ -253,33 +292,38 @@ function wrapNdjsonAsOpenAISse(originalResponse, model) {
}
};
const transform = new TransformStream({
transform(chunk, controller) {
buffer += decoder.decode(chunk, { stream: true });
const lines = buffer.split("\n");
buffer = lines.pop() || "";
for (const line of lines) {
const trimmed = line.trim();
if (!trimmed) continue;
// Translate AI SDK v5 NDJSON line to one or more OpenAI chunks
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
}
},
flush(controller) {
const trimmed = buffer.trim();
if (trimmed) {
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
}
controller.enqueue(encoder.encode(SSE_DONE));
},
});
const transform = new TransformStream({
transform(chunk, controller) {
buffer += decoder.decode(chunk, { stream: true });
const lines = buffer.split("\n");
buffer = lines.pop() || "";
for (const line of lines) {
const trimmed = line.trim();
if (!trimmed) continue;
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
}
},
flush(controller) {
const trimmed = buffer.trim();
if (trimmed) {
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
}
controller.enqueue(encoder.encode(SSE_DONE));
},
});
const newBody = originalResponse.body.pipeThrough(transform);
return new Response(newBody, {
status: originalResponse.status,
statusText: originalResponse.statusText,
headers: originalResponse.headers,
});
const newBody = streamBody.pipeThrough(transform);
return new Response(newBody, {
status: originalResponse?.status || 200,
statusText: originalResponse?.statusText || "OK",
headers: {
"Content-Type": "text/event-stream",
"Cache-Control": "no-cache",
"Connection": "keep-alive",
...(originalResponse?.headers ? Object.fromEntries(originalResponse.headers.entries()) : {}),
"content-type": "text/event-stream",
},
});
}
export default CommandCodeExecutor;

View File

@@ -154,7 +154,18 @@ export class DefaultExecutor extends BaseExecutor {
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
applyAuth(headers, desc, credentials);
if (this.provider === "claude" && model) {
// anthropic-compatible-* nodes serving a real Claude model sit in front of
// Anthropic itself (a rotating multi-account proxy, a corporate gateway),
// so the request needs the same beta flags the `claude` provider sends:
// without `context-management-2025-06-27` upstream rejects the
// `context_management` block Claude Code puts in every request with
// "context_management: Extra inputs are not permitted" (HTTP 400), and the
// combo silently falls through to the next model. The model id gates this:
// a node fronting Kimi or GLM answers on its own ids and never matches, so
// gateways that would choke on unknown beta flags are left untouched.
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
if (model && (this.provider === "claude"
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
}

View File

@@ -10,6 +10,7 @@ import { CodexExecutor } from "./codex.js";
import { CursorExecutor } from "./cursor.js";
import { VertexExecutor } from "./vertex.js";
import { OpenCodeExecutor } from "./opencode.js";
import { OpenCodeGoExecutor } from "./opencode-go.js";
import { GrokWebExecutor } from "./grok-web.js";
import { GrokCliExecutor } from "./grok-cli.js";
import { PerplexityWebExecutor } from "./perplexity-web.js";
@@ -40,6 +41,7 @@ const executors = {
vertex: new VertexExecutor("vertex"),
"vertex-partner": new VertexExecutor("vertex-partner"),
opencode: new OpenCodeExecutor(),
"opencode-go": new OpenCodeGoExecutor(),
"grok-web": new GrokWebExecutor(),
"grok-cli": new GrokCliExecutor(),
gcli: new GrokCliExecutor(), // Alias
@@ -84,6 +86,7 @@ export { CursorExecutor } from "./cursor.js";
export { VertexExecutor } from "./vertex.js";
export { DefaultExecutor } from "./default.js";
export { OpenCodeExecutor } from "./opencode.js";
export { OpenCodeGoExecutor } from "./opencode-go.js";
export { GrokWebExecutor } from "./grok-web.js";
export { GrokCliExecutor } from "./grok-cli.js";
export { PerplexityWebExecutor } from "./perplexity-web.js";

View File

@@ -0,0 +1,182 @@
import crypto from "node:crypto";
import { DefaultExecutor } from "./default.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../translator/formats/responsesApi.js";
const SESSION_HEADER = "x-opencode-session";
const SESSION_FIELD = "_opencodeGoSession";
const MAX_SESSION_LENGTH = 256;
const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses";
const MAX_TOOL_NAME_LEN = 128;
function normalizeSession(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return normalized;
}
function nativeSession(headers) {
if (!headers || typeof headers !== "object") return null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value);
}
return null;
}
function translatedSession(sessionId, clientTool) {
const digest = crypto
.createHash("sha256")
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
.digest("hex")
.slice(0, 32);
return `ses_${digest}`;
}
// Strip the thinking suffix "model(level)" so checks hit the base id.
function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
function isResponsesModel(model) {
return isMuseSparkModel(baseModelId(model));
}
// Flatten Chat Completions tool declarations into the Responses flat shape and
// drop hosted/nameless tools the /responses endpoint rejects.
function normalizeResponsesTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
const name = rawName.trim();
if (!name) return false;
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
? tool.parameters
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
// Mirror the request translator: {type:"object"} without properties is rejected
// by strict Responses backends, so fill in the empty properties map.
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
for (const k of Object.keys(tool)) delete tool[k];
tool.type = "function";
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
if (description) tool.description = description;
tool.parameters = parameters;
validNames.add(tool.name);
return true;
});
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
if (!n || !validNames.has(n)) delete body.tool_choice;
}
}
}
// Last line of defense for native Responses clients (sourceFormat === targetFormat
// skips translation): coerce items in place so malformed tool payloads 400 here
// with a clear shape instead of upstream as InputValidationError.
function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
item.call_id = clampResponsesCallId(item.call_id);
item.arguments = coerceResponsesArguments(item.arguments);
return true;
}
if (item.type === "function_call_output") {
item.call_id = clampResponsesCallId(item.call_id);
item.output = coerceResponsesOutput(item.output);
return true;
}
return true;
});
}
export class OpenCodeGoExecutor extends DefaultExecutor {
constructor() {
super("opencode-go");
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
return super.buildUrl(model, stream, urlIndex, credentials);
}
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
const sourceCredentials = credentials || {};
const native = nativeSession(sourceCredentials.rawHeaders);
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
headers: sourceCredentials.rawHeaders,
body,
connectionId: sourceCredentials.connectionId,
scope: "opencode-go",
});
return {
...sourceCredentials,
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
};
}
async execute(args) {
const credentials = this.prepareRequestCredentials(args);
return super.execute({ ...args, credentials });
}
buildHeaders(credentials, stream = true, url, model) {
const headers = super.buildHeaders(credentials || {}, stream, url, model);
const prepared = credentials?.[SESSION_FIELD];
if (prepared) {
headers[SESSION_HEADER] = prepared;
return headers;
}
const fallback = this.prepareRequestCredentials({ credentials });
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
return headers;
}
transformRequest(model, body, stream, credentials) {
const out = super.transformRequest(model, body);
if (!isResponsesModel(model || body?.model)) return out;
const normalized = normalizeResponsesInput(out.input);
if (normalized) out.input = normalized;
if (!Array.isArray(out.input) || out.input.length === 0) {
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
}
// Responses names the output cap max_output_tokens, not max_tokens.
if (out.max_output_tokens === undefined) {
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
}
delete out.max_tokens;
delete out.max_completion_tokens;
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
}
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
if (!out.reasoning.summary) out.reasoning.summary = "auto";
}
delete out.reasoning_effort;
out.stream = true;
out.store = false;
normalizeResponsesTools(out);
sanitizeResponsesItems(out);
return out;
}
}

View File

@@ -1,11 +1,17 @@
import crypto from "crypto";
import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js";
import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
const OPENCODE_UA = "opencode";
const MESSAGES_MODELS = new Set();
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
const RESPONSES_MODELS = new Set([
"muse-spark-1.2-contributor-free",
"muse-spark-1.3-contributor-free",
]);
function generateRequestId() {
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
@@ -15,19 +21,48 @@ function generateSessionId() {
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
}
// Normalize any resolved id into opencode's ses_ format (stable per-conversation)
function toOpencodeSession(id) {
const stripped = String(id || "").replace(/^ses_/, "").replace(/-/g, "");
return stripped ? `ses_${stripped}` : null;
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
function isResponsesModel(model) {
const base = baseModelId(model);
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
}
function resolveOpencodeSession(body, credentials) {
return toOpencodeSession(resolveSessionId({
headers: credentials?.rawHeaders,
const headers = credentials?.rawHeaders || {};
return resolveSessionId({
headers,
body,
connectionId: credentials?.connectionId,
scope: "opencode",
}));
generate: generateSessionId,
});
}
function normalizeOpencodeReasoning(model, body) {
const current = body.reasoning;
const currentReasoning = current && typeof current === "object" && !Array.isArray(current)
? current
: null;
const requestedEffort = typeof body.reasoning_effort === "string"
? body.reasoning_effort
: currentReasoning?.effort;
if (typeof requestedEffort !== "string") return;
const cleanModel = baseModelId(model || body.model);
const supportedLevels = getThinkingLevels("opencode", cleanModel);
let effort = requestedEffort.toLowerCase().trim();
if ((effort === "max" || effort === "ultra") && supportedLevels?.length && !supportedLevels.includes(effort)) {
if (effort === "ultra" && supportedLevels.includes("max")) effort = "max";
else if (supportedLevels.includes("xhigh")) effort = "xhigh";
}
body.reasoning = { ...currentReasoning, effort };
if (!body.reasoning.summary) body.reasoning.summary = "auto";
delete body.reasoning_effort;
}
export class OpenCodeExecutor extends BaseExecutor {
@@ -38,13 +73,24 @@ export class OpenCodeExecutor extends BaseExecutor {
transformRequest(model, body, stream, credentials) {
this._currentSessionId = resolveOpencodeSession(body, credentials);
if (isResponsesModel(model)) {
// Responses API names the output cap max_output_tokens and takes thinking
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
if (body.max_output_tokens === undefined) {
if (body.max_completion_tokens !== undefined) body.max_output_tokens = body.max_completion_tokens;
else if (body.max_tokens !== undefined) body.max_output_tokens = body.max_tokens;
}
delete body.max_tokens;
delete body.max_completion_tokens;
normalizeOpencodeReasoning(model, body);
}
return injectReasoningContent({ provider: this.provider, model, body });
}
buildUrl(model) {
const base = this.config.baseUrl;
return MESSAGES_MODELS.has(model)
? `${base}/zen/v1/messages`
return isResponsesModel(model)
? `${base}/zen/v1/responses`
: `${base}/zen/v1/chat/completions`;
}

View File

@@ -38,10 +38,13 @@ import {
QODER_MODEL_MAP,
} from "../shared/qoder/constants.js";
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
import { encodeDataUri } from "../translator/concerns/image.js";
/**
* Hoist role:"system" messages out of the messages array (Qoder rejects
* system in messages) and flatten any multipart content arrays.
* system in messages) and flatten multipart content arrays — EXCEPT image
* blocks, which are preserved (see normalizeContent).
*/
function normalizeMessages(messages) {
if (!Array.isArray(messages) || messages.length === 0) {
@@ -51,18 +54,72 @@ function normalizeMessages(messages) {
const out = [];
for (const msg of messages) {
if (!msg || typeof msg !== "object") continue;
const text = extractText(msg.content);
if (msg.role === "system") {
const text = extractText(msg.content);
if (text) systemParts.push(text);
continue;
}
const cloned = { ...msg };
cloned.content = text;
cloned.content = normalizeContent(msg.content);
out.push(cloned);
}
return { messages: out, systemText: systemParts.join("\n\n") };
}
/**
* Normalize one message's content for Qoder.
*
* Text-only content is flattened to a plain string (Qoder's historical
* shape). When images are present the content stays an array and image
* blocks are kept as OpenAI-style `image_url` parts — verified against the
* upstream: it accepts both http(s) URLs and inline base64 data: URIs
* directly, no pre-upload to the /image/upload OSS flow required (that is
* a qodercli client-side choice, not a protocol requirement). The legacy
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
* qodercli leaves them null too.
*
* Claude-style `{type:"image", source:{...}}` blocks are converted to
* `image_url` so claude-format clients also round-trip.
*/
function normalizeContent(content) {
if (typeof content === "string") return content;
if (content == null) return "";
if (!Array.isArray(content)) return String(content);
const blocks = [];
const textParts = [];
let hasImage = false;
for (const item of content) {
if (!item || typeof item !== "object") continue;
if (item.type === OPENAI_BLOCK.IMAGE_URL && typeof item.image_url?.url === "string" && item.image_url.url) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: item.image_url.url } });
hasImage = true;
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
// Claude base64/url image → OpenAI image_url equivalent.
const src = item.source;
const url = src.type === "base64" && src.data
? encodeDataUri(src.media_type || "image/png", src.data)
: typeof src.url === "string" && src.url ? src.url : null;
if (url) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
hasImage = true;
}
} else if (typeof item.text === "string" && item.text) {
if (hasImage || blocks.length) {
// Keep ordering faithful once images are in play.
blocks.push({ type: OPENAI_BLOCK.TEXT, text: item.text });
} else {
textParts.push(item.text);
}
}
}
if (!hasImage) return textParts.join("\n");
// Prepend any text collected before the first image block.
if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") });
return blocks;
}
function extractText(content) {
if (typeof content === "string") return content;
if (content == null) return "";
@@ -85,9 +142,9 @@ function extractText(content) {
function lastUserText(messages) {
for (let i = messages.length - 1; i >= 0; i--) {
const m = messages[i];
if (m?.role === "user" && typeof m.content === "string") {
return m.content;
}
if (m?.role !== "user") continue;
if (typeof m.content === "string") return m.content;
if (Array.isArray(m.content)) return extractText(m.content);
}
return "";
}
@@ -111,6 +168,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) {
if (m.role) { h.update("\0"); h.update(m.role); }
if (typeof m.content === "string" && m.content) {
h.update("\0"); h.update(m.content);
} else if (Array.isArray(m.content)) {
// Include image refs so the same prompt with a different image gets
// a distinct chat_record_id.
h.update("\0");
try { h.update(JSON.stringify(m.content)); } catch {}
}
}
if (tools) {

View File

@@ -28,6 +28,7 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
import { getCapabilitiesForModel } from "../providers/capabilities.js";
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { maybeRejectEarlyStreamError } from "../utils/streamErrorPeek.js";
@@ -58,7 +59,7 @@ export function stripContinuityFields(body) {
return body;
}
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, capsOverride, streamErrorPatterns }) {
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking }) {
const { provider, model } = modelInfo;
const requestStartTime = Date.now();
// Stable per-session color so all lines of one CLI conversation share a tag
@@ -91,7 +92,12 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// differ — kimi/glm only do /chat/completions). Undeclared models keep the
// upstream default (use the transport), preserving behavior for glm/deepseek/...
const useTransport = (!modelSupportedFormats || modelSupportedFormats.includes(sourceFormat)) ? runtimeTransport : null;
const targetFormat = modelTargetFormat || useTransport?.format || getTargetFormat(provider, credentials);
// A source-format-matched endpoint keeps the request lossless. Prefer it
// over a model-level targetFormat, which is only the fallback for clients
// whose wire format has no supported transport (for example MiniMax-M3:
// OpenAI clients should stay on /chat/completions; other clients can fall
// back to its declared Claude target).
const targetFormat = useTransport?.format || modelTargetFormat || getTargetFormat(provider, credentials);
if (useTransport && credentials) credentials.runtimeTransport = useTransport;
const stripList = getModelStrip(alias, model);
const upstreamModel = getModelUpstreamId(alias, model);
@@ -238,6 +244,12 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
delete translatedBody.tools;
}
// Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax)
// reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing.
if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) {
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
}
// Per-request opt-out: client can bypass all token savers via header
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
@@ -248,7 +260,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Headroom: optional external proxy compression; fail open if proxy is absent.
const headroomDiagnostics = {};
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, diagnostics: headroomDiagnostics });
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, timeoutMs: headroomTimeoutMs, diagnostics: headroomDiagnostics });
const headroomLine = formatHeadroomLog(headroomStats);
const headroomSizeLine = formatHeadroomSizeLog(headroomDiagnostics);
if (headroomLine) {
@@ -346,7 +358,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// exception: it is decoded by the executor into OpenAI-compatible output.
let providerResponseFormat = targetFormat;
try {
const result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
const result = await executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
providerResponse = result.response;
providerUrl = result.url;
providerHeaders = result.headers;
@@ -399,7 +421,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
}
try {
const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
const retryResult = await executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
if (retryResult.response.ok) {
providerResponse = retryResult.response;
providerUrl = retryResult.url;
@@ -481,7 +513,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Streaming response
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials });
}
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {

View File

@@ -25,10 +25,16 @@ export function extractUsageFromResponse(responseBody) {
if (!responseBody || typeof responseBody !== "object") return null;
// Claude format
// Note: OpenAI Responses usage ({input_tokens, input_tokens_details:{cached_tokens}})
// also matches this branch. Its prompt is cache-INCLUSIVE and its cache rides in
// input_tokens_details, so emit it as cached_tokens — the convention
// canonicalizeUsage() passes through without folding. Reading it here keeps
// cache accounting correct for /v1/responses and codex traffic.
if (responseBody.usage?.input_tokens !== undefined) {
return {
prompt_tokens: responseBody.usage.input_tokens || 0,
completion_tokens: responseBody.usage.output_tokens || 0,
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.input_tokens_details?.cached_tokens,
cache_read_input_tokens: responseBody.usage.cache_read_input_tokens,
cache_creation_input_tokens: responseBody.usage.cache_creation_input_tokens
};
@@ -39,7 +45,7 @@ export function extractUsageFromResponse(responseBody) {
return {
prompt_tokens: responseBody.usage.prompt_tokens || 0,
completion_tokens: responseBody.usage.completion_tokens || 0,
cached_tokens: responseBody.usage.prompt_tokens_details?.cached_tokens,
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.prompt_tokens_details?.cached_tokens,
reasoning_tokens: responseBody.usage.completion_tokens_details?.reasoning_tokens
};
}

View File

@@ -31,63 +31,20 @@ const CODEX_SOURCE_TO_TARGET = {
/**
* Determine which SSE transform stream to use based on provider/format.
*/
function buildTransformStream({
provider,
sourceFormat,
targetFormat,
userAgent,
reqLogger,
toolNameMap,
customToolNames,
model,
connectionId,
body,
onStreamComplete,
apiKey,
}) {
const isDroidCLI =
userAgent?.toLowerCase().includes("droid") ||
userAgent?.toLowerCase().includes("codex-cli");
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
const isResponsesProvider =
PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
const needsCodexTranslation =
isResponsesProvider &&
targetFormat === FORMATS.OPENAI_RESPONSES &&
!isDroidCLI;
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) {
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
const needsCodexTranslation = isResponsesProvider && targetFormat === FORMATS.OPENAI_RESPONSES && !isDroidCLI;
if (needsCodexTranslation) {
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
return createSSETransformStreamWithLogger(
FORMATS.OPENAI_RESPONSES,
codexTarget,
provider,
reqLogger,
toolNameMap,
model,
connectionId,
body,
onStreamComplete,
apiKey,
customToolNames,
);
}
if (needsCodexTranslation) {
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
}
if (needsTranslation(targetFormat, sourceFormat)) {
return createSSETransformStreamWithLogger(
targetFormat,
sourceFormat,
provider,
reqLogger,
toolNameMap,
model,
connectionId,
body,
onStreamComplete,
apiKey,
customToolNames,
);
}
if (needsTranslation(targetFormat, sourceFormat)) {
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
}
return createPassthroughStreamWithLogger(
provider,
@@ -103,42 +60,14 @@ function buildTransformStream({
/**
* Handle streaming response — pipe provider SSE through transform stream to client.
*/
export async function handleStreamingResponse({
providerResponse,
provider,
model,
sourceFormat,
targetFormat,
userAgent,
body,
stream,
translatedBody,
finalBody,
requestStartTime,
connectionId,
apiKey,
clientRawRequest,
onRequestSuccess,
reqLogger,
toolNameMap,
customToolNames,
streamController,
onStreamComplete,
streamDetailId,
pxpipe,
reqTag,
log,
}) {
if (onRequestSuccess) {
Promise.resolve()
.then(onRequestSuccess)
.catch((err) => {
console.error(
"[ChatCore] onRequestSuccess failed:",
err?.message || err,
);
});
}
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) {
if (onRequestSuccess) {
Promise.resolve()
.then(onRequestSuccess)
.catch(err => {
console.error("[ChatCore] onRequestSuccess failed:", err?.message || err);
});
}
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
// page), piping it through the SSE transform stream causes Next.js
@@ -197,20 +126,7 @@ export async function handleStreamingResponse({
};
}
const transformStream = buildTransformStream({
provider,
sourceFormat,
targetFormat,
userAgent,
reqLogger,
toolNameMap,
customToolNames,
model,
connectionId,
body,
onStreamComplete,
apiKey,
});
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
const isResponsesPassthrough =

View File

@@ -1,4 +1,4 @@
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama
// Returns normalized shape across all providers
const DEFAULT_TIMEOUT_MS = 15000;
@@ -56,8 +56,8 @@ function parseJinaTitle(text) {
return m ? m[1].trim() : null;
}
function buildData({ provider, url, title, format, text, costUsd, responseMs, upstreamMs }) {
return {
function buildData({ provider, url, title, format, text, links, costUsd, responseMs, upstreamMs }) {
const data = {
provider,
url,
title: title || null,
@@ -66,6 +66,8 @@ function buildData({ provider, url, title, format, text, costUsd, responseMs, up
usage: { fetch_cost_usd: costUsd ?? null },
metrics: { response_time_ms: responseMs, upstream_latency_ms: upstreamMs }
};
if (Array.isArray(links)) data.links = links;
return data;
}
async function readJsonOrText(res) {
@@ -115,6 +117,18 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
if (provider === "exa") {
return await runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt });
}
if (provider === "ollama") {
return await runOllama({
url,
fmt,
timeoutMs,
apiKey,
maxCharacters,
costPerQuery,
startedAt,
baseUrl: providerConfig?.baseUrl,
});
}
return { success: false, status: 400, error: `Unsupported provider: ${provider}` };
} catch (err) {
log?.("fetch handler error:", err?.message || err);
@@ -241,3 +255,56 @@ async function runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery
})
};
}
async function runOllama({
url,
fmt,
timeoutMs,
apiKey,
maxCharacters,
costPerQuery,
startedAt,
baseUrl,
}) {
const upstreamStart = Date.now();
const r = await tryFetch(baseUrl, {
method: "POST",
headers: {
"content-type": "application/json",
...(apiKey ? { authorization: `Bearer ${apiKey}` } : {})
},
body: JSON.stringify({ url })
}, timeoutMs);
if (!r.ok) {
return { success: false, status: r.timeout ? 504 : 502, error: r.error };
}
const upstreamMs = Date.now() - upstreamStart;
const { json, text: responseText } = await readJsonOrText(r.res);
if (!r.res.ok) {
const error = json?.error
|| json?.message
|| responseText?.slice(0, 500)
|| `Ollama error: ${r.res.status}`;
return { success: false, status: r.res.status, error };
}
if (!json || typeof json.content !== "string") {
return { success: false, status: 502, error: "Ollama returned an empty or invalid web fetch response" };
}
const text = truncate(json.content, maxCharacters);
return {
success: true,
data: buildData({
provider: "ollama",
url,
title: json.title || null,
format: fmt,
text,
links: json.links,
costUsd: costPerQuery,
responseMs: Date.now() - startedAt,
upstreamMs
})
};
}

View File

@@ -1,6 +1,6 @@
// Antigravity image adapter - delegates to the executor for correct request
// envelope (project, model, requestType, sessionId) and auth headers.
import { nowSec } from "./_base.js";
import { nowSec, sizeToAspectRatio } from "./_base.js";
import { getExecutor } from "../../executors/index.js";
// Convert image input (data URI or raw base64) to Gemini inlineData part
@@ -31,6 +31,19 @@ export default {
const executor = getExecutor("antigravity");
if (!executor) throw new Error("Antigravity executor not found");
// Ensure we use an image model for image generation
const isImageModel = (m) => /image|imagen|image-generation/i.test(m || "");
let targetModel = isImageModel(model) ? model : "gemini-3.1-flash-image";
// If body.size is provided, resolve aspect ratio and append to model
if (body.size && typeof body.size === "string") {
const ratio = sizeToAspectRatio(body.size);
const suffix = ratio.replace(":", "x");
if (!targetModel.includes(suffix)) {
targetModel = `${targetModel}-${suffix}`;
}
}
// Build parts: text prompt + optional input image for editing
const parts = [{ text: body.prompt }];
const imageInput = body.image || (Array.isArray(body.images) && body.images[0]);
@@ -44,7 +57,7 @@ export default {
};
const result = await executor.execute({
model,
model: targetModel,
body: chatBody,
stream: false,
credentials,

View File

@@ -347,6 +347,81 @@ function buildSearxngRequest(config, params) {
};
}
function buildXquikRequest(config, params) {
const apiKey = params.token;
if (!apiKey) throw new Error("Xquik requires an API key");
const queryType = getProviderSetting(params, "queryType");
if (queryType && !["Latest", "Top"].includes(queryType)) {
throw new Error("Xquik queryType must be Latest or Top");
}
const qp = new URLSearchParams({
q: params.query,
limit: String(params.maxResults),
});
const cursor = getProviderSetting(params, "cursor");
if (cursor) qp.set("cursor", cursor);
if (queryType) qp.set("queryType", queryType);
if (params.language) qp.set("language", params.language);
return {
url: `${resolveBaseUrl(config, params)}?${qp}`,
init: {
method: "GET",
headers: { Accept: "application/json", "x-api-key": apiKey },
},
};
}
// ── Ollama Cloud web_search ──────────────────────────────────────────────
// POST https://ollama.com/api/web_search { query, max_results }
// Response: { results: [{ title, url, content, published_at? }] }
function buildOllamaSearchRequest(config, params) {
const body = { query: params.query, max_results: params.maxResults };
if (params.country) body.country = params.country;
if (params.language) body.language = params.language;
return {
url: resolveBaseUrl(config, params),
init: {
method: "POST",
headers: {
"Content-Type": "application/json",
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
},
body: JSON.stringify(body),
},
};
}
// ── GLM Coding plan MCP web_search_prime ──────────────────────────────────
// POST https://api.z.ai/api/mcp/web_search_prime/mcp
// JSON-RPC envelope: { jsonrpc, id, method: "tools/call",
// params: { name: "web_search_prime", arguments: { search_query, count } } }
// Response: { result: { content: [{ type: "text", text: "<json>" }] } }
function buildGlmSearchRequest(config, params) {
const body = {
jsonrpc: "2.0",
id: `9r-${Date.now()}`,
method: "tools/call",
params: {
name: "web_search_prime",
arguments: { search_query: params.query, count: params.maxResults },
},
};
return {
url: resolveBaseUrl(config, params),
init: {
method: "POST",
headers: {
"Content-Type": "application/json",
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
},
body: JSON.stringify(body),
},
};
}
// ── Dispatcher ──────────────────────────────────────────────────────────
const BUILDERS = {
@@ -360,6 +435,9 @@ const BUILDERS = {
"searchapi": buildSearchApiRequest,
"youcom": buildYouComRequest,
"searxng": buildSearxngRequest,
"xquik": buildXquikRequest,
"ollama-search": buildOllamaSearchRequest,
"glm": buildGlmSearchRequest,
};
/**

View File

@@ -1,8 +1,10 @@
/**
* Wrap chat-completions endpoints (with built-in web search) into the unified
* /v1/search response format. Supports gemini, openai, xai, kimi, minimax, perplexity.
* /v1/search response format. Supports gemini, antigravity, openai, xai, kimi,
* minimax, perplexity.
*/
import { PROVIDER_MEDIA } from "../../providers/index.js";
import { ANTIGRAVITY_IDE_USER_AGENT } from "../../providers/shared.js";
// Default search model + endpoint derive from registry searchViaChat (single source)
const searchModel = (id) => PROVIDER_MEDIA[id]?.searchViaChat?.defaultModel;
@@ -28,13 +30,37 @@ function toResult(c, index, provider, retrievedAt) {
score: null,
published_at: null,
favicon_url: null,
content: null,
content: c.content || null,
metadata: {},
citation: { provider, retrieved_at: retrievedAt, rank: index + 1 },
provider_raw: null
};
}
// Antigravity search request envelope (mirrors the IDE client)
const AG_CLIENT_NAME = "antigravity";
const AG_SEARCH_GENERATION_CONFIG = { temperature: 1.0, maxOutputTokens: 8192 };
const AG_CONTEXT_BEFORE = 150;
const AG_CONTEXT_AFTER = 250;
/** Widen a grounded segment to its surrounding sentence(s) in the answer text. */
function expandSegment(text, segment) {
const { startIndex, endIndex } = segment || {};
if (!text || !Number.isInteger(startIndex) || !Number.isInteger(endIndex)) return "";
const start = Math.max(0, startIndex - AG_CONTEXT_BEFORE);
const end = Math.min(text.length, endIndex + AG_CONTEXT_AFTER);
let out = text.slice(start, end).trim();
// Drop the partial words the window cut off at either edge
if (start > 0) out = `...${out.replace(/^\S+/, "")}`;
if (end < text.length) out = `${out.replace(/\S+$/, "")}...`;
return out.trim();
}
/** Join deduped grounding pieces, skipping empties. */
function joinPieces(set, sep) {
return [...(set || [])].filter(Boolean).join(sep).trim();
}
/** Coerce a citation that might be a raw URL string or an object. */
function normalizeCitation(c) {
if (!c) return null;
@@ -46,6 +72,8 @@ function normalizeCitation(c) {
/**
* Provider-specific configuration map. All providers must implement:
* { endpoint, defaultModel, buildBody, buildHeaders, extractAnswer }
* Optional: requireCredentials(credentials) → error string when a provider needs
* more than a token (returns null when satisfied).
*/
const CHAT_SEARCH_CONFIG = {
gemini: {
@@ -73,6 +101,71 @@ const CHAT_SEARCH_CONFIG = {
}
},
antigravity: {
endpoint: () => searchEndpoint("antigravity"),
// Upstream 403s on a missing or fabricated project — surface the real cause
requireCredentials: (credentials) =>
credentials?.projectId ? null : "Antigravity account has no projectId — reconnect the account",
buildBody: (query, model, credentials) => ({
project: credentials.projectId,
model,
userAgent: AG_CLIENT_NAME,
requestType: "search",
request: {
contents: [{ role: "user", parts: [{ text: query }] }],
tools: [{ googleSearch: {} }],
generationConfig: AG_SEARCH_GENERATION_CONFIG
}
}),
buildHeaders: (token) => ({
"Content-Type": "application/json",
Authorization: `Bearer ${token}`,
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT
}),
extractAnswer: (data) => {
// Antigravity wraps the Gemini payload in { response: {...} }
const response = data?.response || data;
const candidate = response?.candidates?.[0];
const parts = candidate?.content?.parts || [];
const text = parts.map((p) => p?.text || "").filter(Boolean).join("");
const grounding = candidate?.groundingMetadata || {};
const chunks = grounding.groundingChunks || [];
const supports = grounding.groundingSupports || [];
// Upstream repeats the same source across chunks — key by URL so it stays one citation.
// Map, not a plain object: both the index and the URL come from upstream.
const sources = new Map();
const byIndex = chunks.map((ch) => {
const web = ch?.web;
const url = web?.uri || web?.url || "";
if (!url) return null;
if (!sources.has(url)) sources.set(url, { title: web.title || "", snippets: new Set(), contexts: new Set() });
return sources.get(url);
});
// Each support ties a sentence of the answer back to the chunks that grounded it
for (const s of supports) {
const segment = s?.segment;
const grounded = segment?.text || "";
const expanded = expandSegment(text, segment) || grounded;
for (const idx of s?.groundingChunkIndices || []) {
const source = Number.isInteger(idx) ? byIndex[idx] : null;
if (!source) continue;
if (grounded) source.snippets.add(grounded);
if (expanded) source.contexts.add(expanded);
}
}
const citations = [...sources].map(([url, src]) => {
const snippet = joinPieces(src.snippets, " | ") || src.title;
return { url, title: src.title, snippet, content: joinPieces(src.contexts, "\n\n") || snippet };
});
const tokens = response?.usageMetadata?.totalTokenCount || 0;
return { text, citations, tokens };
}
},
openai: {
endpoint: () => searchEndpoint("openai"),
buildBody: (query, model) => {
@@ -366,13 +459,18 @@ export async function handleChatSearch({
};
}
const credentialError = cfg.requireCredentials?.(credentials);
if (credentialError) {
return { success: false, status: 401, error: credentialError };
}
const limit =
Number.isFinite(maxResults) && maxResults > 0
? Math.floor(maxResults)
: DEFAULT_MAX_RESULTS;
const useModel = model || searchModel(provider);
const url = cfg.endpoint(useModel);
const body = cfg.buildBody(query, useModel);
const body = cfg.buildBody(query, useModel, credentials);
const headers = cfg.buildHeaders(token);
const controller = new AbortController();

View File

@@ -10,6 +10,7 @@
import { buildSearchRequest } from "./callers.js";
import { normalizeSearchResponse } from "./normalizers.js";
import { handleChatSearch } from "./chatSearch.js";
import { fetchPublic } from "../../../src/shared/utils/ssrfGuard.js";
const GLOBAL_TIMEOUT_MS = 15000;
const NON_RETRIABLE = new Set([400, 401, 403, 404]);
@@ -100,7 +101,7 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
log?.info?.("SEARCH", `${provider.id} | "${params.query.slice(0, 80)}" | type=${params.searchType}`);
try {
const resp = await fetch(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
const resp = await fetchPublic(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
clearTimeout(timer);
if (!resp.ok) {
const errText = await resp.text().catch(() => "");
@@ -111,6 +112,13 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType);
const results = normalized.results.slice(0, params.maxResults);
const duration = Date.now() - startTime;
const usage = {
queries_used: 1,
search_cost_usd: providerConfig.costPerQuery ?? null,
};
if (Number.isFinite(providerConfig.creditsPerResult)) {
usage.provider_credits_used = results.length * providerConfig.creditsPerResult;
}
return {
success: true,
@@ -119,7 +127,8 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
query: params.query,
results,
answer: null,
usage: { queries_used: 1, search_cost_usd: providerConfig.costPerQuery || 0 },
usage,
...(normalized.pagination ? { pagination: normalized.pagination } : {}),
metrics: { response_time_ms: duration, upstream_latency_ms: duration, total_results_available: normalized.totalResults },
errors: []
}

View File

@@ -199,6 +199,89 @@ function normalizeSearxng(data, _query, _searchType) {
return { results, totalResults: results.length };
}
function normalizeXquik(data, _query, _searchType) {
const now = new Date().toISOString();
const items = Array.isArray(data.tweets) ? data.tweets : [];
const results = items.map((item, idx) => {
const username = typeof item?.author?.username === "string" ? item.author.username : "";
const authorName = typeof item?.author?.name === "string" ? item.author.name : "";
const tweetId = typeof item?.id === "string" ? item.id : String(item?.id || "");
const url = username && tweetId
? `https://x.com/${encodeURIComponent(username)}/status/${encodeURIComponent(tweetId)}`
: tweetId
? `https://x.com/i/web/status/${encodeURIComponent(tweetId)}`
: "";
const author = username ? `@${username}` : authorName || null;
const title = author ? `${author} on X` : "X post";
const imageUrl = Array.isArray(item?.media)
? item.media.find((media) => typeof media?.mediaUrl === "string")?.mediaUrl
: null;
return makeResult("xquik", {
title,
url,
snippet: typeof item?.text === "string" ? item.text : "",
published_at: typeof item?.createdAt === "string" ? item.createdAt : null,
author,
image_url: imageUrl || null,
source_type: "x_post",
full_text: typeof item?.text === "string" ? item.text : undefined,
text_format: "text",
}, idx, now);
});
const nextCursor = typeof data.next_cursor === "string" && data.next_cursor ? data.next_cursor : null;
return {
results,
totalResults: null,
pagination: {
has_more: data.has_next_page === true,
next_cursor: nextCursor,
},
};
}
function normalizeOllamaSearch(data, _query, _searchType) {
const now = new Date().toISOString();
const items = Array.isArray(data?.results) ? data.results : (Array.isArray(data) ? data : []);
const results = items.map((item, idx) =>
makeResult("ollama-search", {
title: item.title,
url: item.url,
snippet: item.content || item.snippet || "",
full_text: item.content,
text_format: "text",
published_at: item.published_at || null,
source_type: item.source || null,
}, idx, now)
);
return { results, totalResults: results.length };
}
function normalizeGlmSearch(data, _query, _searchType) {
const now = new Date().toISOString();
// MCP envelope: { result: { content: [{ type: "text", text: "<json>" }] } }
let payload = data;
const textContent = data?.result?.content?.[0]?.text;
if (typeof textContent === "string") {
try { payload = JSON.parse(textContent); } catch { payload = {}; }
}
const items = Array.isArray(payload?.results) ? payload.results
: Array.isArray(payload?.news) ? payload.news
: Array.isArray(payload) ? payload
: [];
const results = items.map((item, idx) =>
makeResult("glm", {
title: item.title,
url: item.link || item.url,
snippet: item.content || "",
published_at: item.publish_date || item.published_at || null,
favicon_url: item.icon || null,
source_type: item.media || null,
}, idx, now)
);
return { results, totalResults: results.length };
}
const NORMALIZERS = {
"serper": normalizeSerper,
"brave-search": normalizeBrave,
@@ -210,11 +293,14 @@ const NORMALIZERS = {
"searchapi": normalizeSearchApi,
"youcom": normalizeYouCom,
"searxng": normalizeSearxng,
"xquik": normalizeXquik,
"ollama-search": normalizeOllamaSearch,
"glm": normalizeGlmSearch,
};
/**
* Dispatch to the appropriate normalizer based on providerId.
* @returns {{results: Array, totalResults: number|null}}
* @returns {{results: Array, totalResults: number|null, pagination?: object}}
*/
export function normalizeSearchResponse(providerId, data, query, searchType) {
const fn = NORMALIZERS[providerId];

View File

@@ -6,6 +6,16 @@
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
// 4. DEFAULT_CAPABILITIES — safe floor (always returned)
//
// Two extra layers then refine the result, and neither can override the hand
// written tables above (steps 1-2 short-circuit before they are consulted):
// • the synced catalog — modalities keyed by model, limits keyed by provider
// + model, refreshed from models.dev in the background. It reads a file, so
// the server installs it via setCatalogSource(); this module stays free of
// node:fs because the dashboard bundles it into the browser too.
// • visionPatterns.js — name-based vision detection, last resort so a model
// nobody has catalogued yet still accepts images.
// Both only ever turn a capability ON.
//
// ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+
// models, MIT). Each model exposes the exact fields we map below:
@@ -23,6 +33,7 @@
// 2.0+, Grok, Perplexity). Verify with: curl -s https://models.dev/api.json
import { matchPattern } from "./pricing.js";
import { looksLikeVisionModel } from "./visionPatterns.js";
/**
* Safe floor — every resolved result is merged over this so consumers
@@ -46,6 +57,7 @@ export const DEFAULT_CAPABILITIES = {
thinkingFormat: null,
thinkingCanDisable: true, // false → model cannot turn thinking off (clamp to min instead of disable)
thinkingRange: null, // { min, max } for budget formats; null = no clamp
thinkingEffortSupported: false, // zai format only: model accepts a reasoning_effort level (GLM-5.2+; older GLM ignores it)
// limits (tokens)
contextWindow: 200000,
maxOutput: 64000,
@@ -71,7 +83,8 @@ export function capabilitiesFromServiceKind(kind) {
* otherwise mis-match. Only declare deltas vs DEFAULT.
*/
export const MODEL_CAPABILITIES = {
// Claude Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
// Claude Fable 5.1, Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
"claude-fable-5-1": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
@@ -94,8 +107,14 @@ export const MODEL_CAPABILITIES = {
// Gemini image-gen / OpenAI image / xai image variants
"gpt-image-1": { imageOutput: true, tools: false },
// GLM vision variant (text GLM has no vision)
"glm-4.6v": { vision: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000 },
// GLM vision variants (text GLM has no vision) — 5.3-Flash and 5V-Turbo are
// natively multimodal per z.ai, and 5.3-Flash carries the full 1M window.
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "zai", contextWindow: 1000000, maxOutput: 131072 },
"glm-4.6v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000, maxOutput: 32768 },
"glm-4.5v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 64000, maxOutput: 16384 },
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
@@ -108,6 +127,10 @@ export const MODEL_CAPABILITIES = {
"kimi-for-coding-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
"kimi-k2.7-code": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
"kimi-k2.7-code-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
// OpenCode Free Muse Spark — multimodal (text+image per models.dev meta/muse-spark)
// via OpenAI Responses input_image; reasoning supports up to xhigh.
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
};
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
@@ -131,6 +154,7 @@ export const PROVIDER_CAPABILITIES = {
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
},
"codex": {
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
@@ -155,24 +179,69 @@ export const PROVIDER_CAPABILITIES = {
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking
// off → thinkingCanDisable:false (clamped to minimal instead of disabled).
// (see registry thinkingFormat). For thinkingCanDisable use the server's
// reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
// below; it is NOT the inverse of onlyReasoning.
"codebuddy-cn": {
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 },
// maxOutput 64000 per both the plugin-baked fallback and the live server
// table (the old 38000 had no source and truncated output).
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 },
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
// Per-model values mirror the server's product-config payload (the plugin
// fetches it from copilot.tencent.com; the `models[]` entries carry
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
// maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
// plugin-baked fallback disagree, the server table wins.
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
// their thinking is switchable; the hy* models are forced always-on.
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
// (200K) without this map. contextWindow follows the real model family's
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
// max_output_tokens arrives as 0 for every model, so outputs are
// best-guess from the real model family. Vision tags below follow the
// upstream is_vl flag per explicit request, even though the executor
// currently sends image_urls:null (image pass-through over the agent_chat
// SSE protocol is unverified). reasoning:true on all of them — every model can
// reason; the upstream is_reasoning flag only drives model_config selection.
// thinkingFormat keeps the true-model family for documentation/UI, but
// thinkingCanDisable:false everywhere: the executor only forwards
// messages/tools/max_tokens, and thinking is fixed upstream via
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
// must never be offered as an option.
"qoder": {
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
},
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
"poolside": {
@@ -205,6 +274,7 @@ export const PATTERN_CAPABILITIES = [
// ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─
{ pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } },
{ pattern: "*gemini-3.8*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
{ pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
{ pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } },
{ pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
@@ -214,6 +284,9 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
@@ -234,6 +307,8 @@ export const PATTERN_CAPABILITIES = [
// ── Grok (vision + Live Search) ──────────────────────────────────
{ pattern: "*grok*image*", caps: { imageOutput: true } },
{ pattern: "*grok-code*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 256000 } },
// Grok 4.6: 500k context, no text output limit (docs.x.ai/developers/grok-4-6)
{ pattern: "*grok-4.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 500000 } },
// Grok 4.5 (Grok CLI / Grok Build): 500k context per cli-chat-proxy /v1/models
{ pattern: "*grok-4.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 64000 } },
{ pattern: "*grok-4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
@@ -261,6 +336,10 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*kimi*", caps: { reasoning: true, thinkingFormat: "kimi", contextWindow: 262144 } },
// ── GLM / Z.ai (thinking.enabled; disable via enable_thinking:false) ─
// reasoning_effort is only read by z.ai from GLM-5.2 onward (docs.z.ai/guides/capabilities/thinking) —
// older GLM (4.x, 5.0, 5.1, 5-turbo, 5v-turbo) ignore it, so gate it per exact version, not the "*glm-5*" catch-all.
{ pattern: "*glm-5.3*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-5.2*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-5*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-4.7*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-4*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
@@ -309,6 +388,9 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*laguna-s-2.1*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 } },
{ pattern: "*laguna*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 } },
// ── OpenCode Free Muse Spark (multimodal text+image; OpenAI Responses reasoning supports up to xhigh) ─
{ pattern: "*muse*spark*", caps: { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 } },
// ── Others ───────────────────────────────────────────────────────
{ pattern: "*hunyuan*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
{ pattern: "hy3*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
@@ -330,6 +412,46 @@ const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
* @param {string} model
* @returns {object} full capabilities object
*/
const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
// Catalog lookups, installed by the server at startup. Left as no-ops in the
// browser bundle, where there is no file to read.
let catalogSource = null;
/**
* Install the synced catalog reader (server only).
* @param {{ getModalities: Function, getLimits: Function } | null} source
*/
export function setCatalogSource(source) {
catalogSource = source;
}
// Apply the synced catalog + name heuristic on top of a table-resolved result.
// Strictly additive: a capability already true stays true, and a false one only
// flips when an outside source positively declares support.
function refine(base, provider, model) {
const result = { ...DEFAULT_CAPABILITIES, ...base };
if (catalogSource) {
const modalities = catalogSource.getModalities(model);
if (modalities) {
for (const key of MODALITY_KEYS) {
if (modalities[key] === true) result[key] = true;
}
}
const limits = catalogSource.getLimits(provider, model);
if (limits) {
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
}
}
if (!result.vision && looksLikeVisionModel(model)) result.vision = true;
return result;
}
export function getCapabilitiesForModel(provider, model) {
if (!model) return { ...DEFAULT_CAPABILITIES };
@@ -347,16 +469,13 @@ export function getCapabilitiesForModel(provider, model) {
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
// 3. Pattern match (first match wins)
// 3. Pattern match (first match wins), refined by catalog + name heuristic
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
return { ...DEFAULT_CAPABILITIES, ...caps };
return refine(caps, provider, model);
}
}
// 4. Floor (upstream-validated gateways keep vision on for unknown models)
if (provider && TRUST_UPSTREAM_VISION.has(provider)) {
return { ...DEFAULT_CAPABILITIES, vision: true };
}
return { ...DEFAULT_CAPABILITIES };
// 4. Floor
return refine(null, provider, model);
}

View File

@@ -0,0 +1,72 @@
// Read side of the model catalog synced from models.dev.
//
// The file is the source of truth; the only thing held in memory is a parsed
// copy dropped as soon as the file's mtime changes. getCapabilitiesForModel is
// synchronous and runs per request, so the hot path is one stat (~1us) and the
// parse (~0.1ms on a ~18KB file) only reruns after a sync.
import fs from "node:fs";
import path from "node:path";
import { DATA_DIR } from "@/lib/dataDir.js";
export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
// Trimmed upstream catalog, read by the add-models skill (not by the router).
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
const EMPTY = { models: {}, providers: {} };
let cache = EMPTY;
let cachedMtime = -1;
// "zai-org/GLM-4.6V:free" -> "glm-4.6v"
function baseId(model) {
if (!model) return "";
const withoutVendor = model.includes("/") ? model.split("/").pop() : model;
return withoutVendor.toLowerCase().split(":")[0];
}
function load() {
let mtime;
try {
mtime = fs.statSync(CATALOG_FILE).mtimeMs;
} catch {
cache = EMPTY;
cachedMtime = -1;
return cache;
}
if (mtime === cachedMtime) return cache;
cachedMtime = mtime;
try {
const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8"));
cache = { models: parsed?.models || {}, providers: parsed?.providers || {} };
} catch {
cache = EMPTY;
}
return cache;
}
// Modality is a property of the model itself — any gateway serving it inherits
// the same image/video/pdf support, so this is keyed by model id alone.
export function getCatalogModalities(model) {
return load().models[baseId(model)] || null;
}
// Context and output limits are a property of the gateway, not the model: each
// one truncates differently, so these stay keyed by provider + model.
export function getCatalogLimits(provider, model) {
const byProvider = provider && load().providers[provider];
if (!byProvider) return null;
return byProvider[model] || byProvider[baseId(model)] || null;
}
// Force a re-read on the next lookup (called right after a sync writes the file).
export function invalidateCatalog() {
cachedMtime = -1;
}
// Hand the reader to capabilities.js. That module is bundled into the browser
// too, so it cannot import this file directly — the server pushes it in.
export async function installCatalogSource() {
const { setCatalogSource } = await import("./capabilities.js");
setCatalogSource({ getModalities: getCatalogModalities, getLimits: getCatalogLimits });
}

View File

@@ -18,3 +18,10 @@ export function withCodexReviewModels(models) {
];
});
}
export function isMuseSparkModel(modelId) {
if (!modelId || typeof modelId !== "string") return false;
const clean = modelId.replace(/\([^()]+\)\s*$/, "").trim();
const base = clean.includes("/") ? clean.split("/").pop() : clean;
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
}

View File

@@ -53,10 +53,15 @@ export const MODEL_PRICING = {
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
// === Gemini ===
"gemini-3.8-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.8-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.8-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.8-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
@@ -260,6 +265,7 @@ export const PROVIDER_PRICING = {
"z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 },
"z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 },
"z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 },
"z-ai/glm-5.3-free": { input: 0, output: 0, cached: 0, reasoning: 0 },
},
};

View File

@@ -17,7 +17,7 @@ export default {
deprecationNotice: "RISK_NOTICE",
},
category: "oauth",
serviceKinds: ["llm", "image"],
serviceKinds: ["llm", "image", "webSearch"],
transport: {
baseUrls: [ANTIGRAVITY_IDE_BASE_URL],
format: "antigravity",
@@ -36,8 +36,7 @@ export default {
},
},
usage: {
// Discovery (quota/project) on PROD; daily host rejects these.
quotaApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels",
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
tokenUrl: "https://oauth2.googleapis.com/token",
},
@@ -45,6 +44,10 @@ export default {
clientSecret: "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",
},
models: [
{ id: "gemini-3.8-flash-high", name: "Gemini 3.8 Flash (High)", upstreamModelId: "gemini-3.8-flash-high(high)" },
{ id: "gemini-3.8-flash-medium", name: "Gemini 3.8 Flash (Medium)", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
{ id: "gemini-3.8-flash-low", name: "Gemini 3.8 Flash (Low)", upstreamModelId: "gemini-3.8-flash-low(low)" },
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
{ id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" },
{ id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" },
{ id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" },
@@ -82,6 +85,11 @@ export default {
loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT,
refreshLeadMs: 300000,
},
searchViaChat: {
defaultModel: "gemini-2.5-flash",
endpoint: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:generateContent`,
freeTier: "Free — Google Search grounding through an Antigravity OAuth account.",
},
features: {
usage: true,
},

View File

@@ -1,4 +1,4 @@
import { CLAUDE_CLI_SPOOF_HEADERS } from "../shared.js";
import { CLAUDE_CLI_VERSION } from "../shared.js";
export default {
id: "claude",
@@ -25,7 +25,7 @@ export default {
"Anthropic-Version": "2023-06-01",
"Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28",
"Anthropic-Dangerous-Direct-Browser-Access": "true",
"User-Agent": "claude-cli/2.1.92 (external, sdk-cli)",
"User-Agent": `claude-cli/${CLAUDE_CLI_VERSION} (external, sdk-cli)`,
"X-App": "cli",
"X-Stainless-Helper-Method": "stream",
"X-Stainless-Retry-Count": "0",
@@ -58,6 +58,7 @@ export default {
},
models: [
{ id: "claude-opus-5", name: "Claude Opus 5" },
{ id: "claude-fable-5-1", name: "Claude Fable 5.1" },
{ id: "claude-fable-5", name: "Claude Fable 5" },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" },

View File

@@ -47,19 +47,26 @@ export default {
models: [
{ id: "glm-5.2", name: "GLM-5.2" },
{ id: "glm-5.1", name: "GLM-5.1" },
{ id: "glm-5.0", name: "GLM-5.0" },
{ id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" },
{ id: "glm-5v-turbo", name: "GLM-5v-Turbo" },
{ id: "glm-4.7", name: "GLM-4.7" },
{ id: "minimax-m3", name: "MiniMax-M3" },
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
{ id: "kimi-k2.7", name: "Kimi-K2.7-Code" },
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
{ id: "hy3-preview", name: "Hy3 Preview" },
// Catalog mirrors the server's product-config payload (the plugin fetches
// it from copilot.tencent.com). Models the server no longer publishes are
// removed even when the chat endpoint still answers them — the published
// list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x
// (endpoint returns 11102 "model service info not found"), plus
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
// deepseek-v3-2-volc (absent from the server list, though still answering
// 200) and hy3-x (paid tier, not used here).
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
{ id: "hy3", name: "Hy3" },
{ id: "hy4-preview", name: "Hy4-Preview" },
{ id: "glm-5.3", name: "GLM-5.3" },
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
{ id: "kimi-k3-1", name: "Kimi-K3" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
],
oauth: {
baseUrl: "https://copilot.tencent.com",

View File

@@ -45,6 +45,7 @@ export default {
},
},
models: [
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
@@ -59,6 +60,9 @@ export default {
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },

View File

@@ -45,6 +45,7 @@ export default {
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
{ id: "deepseek-reasoner", name: "DeepSeek V3.2 Reasoner" },
],

View File

@@ -36,6 +36,7 @@ export default {
},
},
models: [
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash" },
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" },
{ id: "gemini-3.6-flash", name: "Gemini 3.6 Flash" },
{ id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },

View File

@@ -22,10 +22,13 @@ export default {
},
models: [
{ id: "glm-5.3", name: "GLM 5.3" },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM-4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
{ id: "glm-4.6", name: "GLM-4.6" },
{ id: "glm-4.5-air", name: "GLM-4.5-Air" },
],

View File

@@ -46,12 +46,28 @@ export default {
],
models: [
{ id: "glm-5.3", name: "GLM 5.3" },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM 4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
],
serviceKinds: ["llm", "webSearch"],
// Coding plan bundles web search on the same API key as chat.
searchConfig: {
baseUrl: "https://api.z.ai/api/mcp/web_search_prime/mcp",
method: "POST",
authType: "apikey",
authHeader: "bearer",
costPerQuery: 0,
searchTypes: ["web"],
defaultMaxResults: 5,
maxMaxResults: 50,
timeoutMs: 10000,
cacheTTLMs: 300000,
},
features: {
usage: true,
usageApikey: true,

View File

@@ -17,6 +17,12 @@ export default {
transport: {
baseUrl: "https://api.groq.com/openai/v1/chat/completions",
validateUrl: "https://api.groq.com/openai/v1/models",
// No dedicated quota endpoint; rate-limit info rides on x-ratelimit-*
// response headers, always included. Reuse the models list (already
// used as validateUrl) so reading usage never costs tokens.
usage: {
url: "https://api.groq.com/openai/v1/models",
},
},
models: [
{ id: "llama-3.3-70b-versatile", name: "Llama 3.3 70B" },
@@ -34,4 +40,8 @@ export default {
authHeader: "bearer",
format: "openai",
},
features: {
usage: true,
usageApikey: true,
},
};

View File

@@ -66,6 +66,7 @@ import p63 from "./nebius.js";
import p64 from "./nvidia.js";
import p65 from "./ollama-local.js";
import p66 from "./ollama.js";
import p123 from "./ollama-search.js";
import p67 from "./openai.js";
import p68 from "./opencode-go.js";
import p69 from "./opencode.js";
@@ -121,6 +122,7 @@ import p118 from "./selfhosted-tts.js";
import p119 from "./selfhosted-embedding.js";
import p120 from "./fish-audio.js";
import p121 from "./alitp-intl.js";
import p122 from "./xquik.js";
export default [
p0,
@@ -190,6 +192,7 @@ export default [
p64,
p65,
p66,
p123,
p67,
p68,
p69,
@@ -243,4 +246,5 @@ export default [
p119,
p120,
p121,
p122,
];

View File

@@ -0,0 +1,35 @@
export default {
id: "ollama-search",
alias: "ollama-search",
display: {
name: "Ollama Search",
icon: "cloud",
color: "#ffffff",
textIcon: "OL",
website: "https://ollama.com",
notice: {
text: "Web search via Ollama Cloud subscription. Reuses the API key from the Ollama (chat) provider.",
apiKeyUrl: "https://ollama.com/settings/keys",
},
},
category: "apikey",
authType: "apikey",
authModes: ["apikey"],
serviceKinds: ["webSearch"],
// Credential fallback: reuses the API key registered under the `ollama`
// chat provider — one key, chat + search.
credentialFallback: "ollama",
searchConfig: {
baseUrl: "https://ollama.com/api/web_search",
method: "POST",
authType: "apikey",
authHeader: "bearer",
costPerQuery: 0,
freeMonthlyQuota: 1000,
searchTypes: ["web"],
defaultMaxResults: 5,
maxMaxResults: 10,
timeoutMs: 10000,
cacheTTLMs: 300000,
},
};

View File

@@ -31,7 +31,16 @@ export default {
{ id: "qwen3.5", name: "Qwen3.5" },
{ id: "minimax-m3", name: "MiniMax M3" },
],
serviceKinds: ["llm"],
serviceKinds: ["llm", "webFetch"],
fetchConfig: {
baseUrl: "https://ollama.com/api/web_fetch",
method: "POST",
authType: "apikey",
authHeader: "bearer",
formats: ["markdown"],
maxCharacters: 200000,
timeoutMs: 30000,
},
features: {
usage: true,
usageApikey: true,

View File

@@ -21,6 +21,9 @@ export default {
transport: {
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
headers: {},
usage: {
url: "https://opencode.ai/zen/go/v1/usage",
},
},
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
@@ -31,12 +34,14 @@ export default {
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
],
models: [
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
@@ -45,5 +50,13 @@ export default {
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
// Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces
// chatCore past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
],
features: {
usage: true,
usageApikey: true,
},
};

View File

@@ -19,7 +19,12 @@ export default {
},
noAuth: true,
},
models: [],
models: [
// Muse Spark models are served by /zen/v1/responses; the rest stay on
// /chat/completions, so the format is declared per-model, not per-provider.
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
],
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true,
};

View File

@@ -30,12 +30,15 @@ export default {
{ id: "auto", name: "Auto" },
{ id: "performance", name: "Performance" },
{ id: "efficient", name: "Efficient" },
{ id: "qmodel_preview", name: "Qwen3.8-Max-Preview" },
{ id: "lite", name: "Lite" },
{ id: "qmodel_38max", name: "Qwen3.8-Max" },
{ id: "qmodel_latest", name: "Qwen3.7-Max" },
{ id: "qmodel", name: "Qwen3.7-Plus" },
{ id: "qfmodel", name: "Qwen3.8-Flash" },
{ id: "kmodel_latest", name: "Kimi-K3" },
{ id: "kmodel", name: "Kimi-K2.7-Code" },
{ id: "gm51model", name: "GLM-5.2" },
{ id: "gmodel", name: "GLM-5.3" },
{ id: "gfmodel", name: "GLM-5.3-Flash" },
{ id: "dmodel", name: "DeepSeek-V4-Pro" },
{ id: "dfmodel", name: "DeepSeek-V4-Flash" },
{ id: "mmodel", name: "MiniMax-M3" },

View File

@@ -24,129 +24,31 @@ export default {
validateUrl: "https://api.tokenrouter.com/v1/models",
thinkingFormat: "tokenrouter",
},
// Seed snapshot from live /v1/models (120 entries). Latest catalogue is
// Seed snapshot from live /v1/models. Latest catalogue is
// fetched via modelsFetcher; other ids still accepted via passthroughModels.
models: [
{ id: "MiniMax-Hailuo-2.3", name: "Minimax Hailuo 2.3", kind: "video" },
{ id: "MiniMax-M3", name: "Minimax M3" },
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5" },
{ id: "anthropic/claude-haiku-4.5", name: "Claude Haiku 4.5" },
{ id: "anthropic/claude-opus-4.5", name: "Claude Opus 4.5" },
{ id: "anthropic/claude-opus-4.6", name: "Claude Opus 4.6" },
{ id: "anthropic/claude-opus-4.7", name: "Claude Opus 4.7" },
{ id: "anthropic/claude-opus-4.7-fast", name: "Claude Opus 4.7 Fast" },
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
{ id: "anthropic/claude-opus-4.8", name: "Claude Opus 4.8" },
{ id: "anthropic/claude-opus-4.8-fast", name: "Claude Opus 4.8 Fast" },
{ id: "anthropic/claude-opus-5", name: "Claude Opus 5" },
{ id: "anthropic/claude-opus-5-fast", name: "Claude Opus 5 Fast" },
{ id: "anthropic/claude-sonnet-4", name: "Claude Sonnet 4" },
{ id: "anthropic/claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
{ id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "bytedance-seed/seedream-4.5", name: "Seedream 4.5", kind: "image" },
{ id: "bytedance-seed/seedream-5.0-lite", name: "Seedream 5.0 Lite", kind: "image" },
{ id: "bytedance-seed/seedream-5.0-pro", name: "Seedream 5.0 Pro", kind: "image" },
{ id: "claude-haiku-4-5", name: "Claude Haiku 4 5" },
{ id: "claude-opus-4-8-m-aws", name: "Claude Opus 4 8 M Aws" },
{ id: "deepseek/deepseek-v3.2", name: "Deepseek V3.2" },
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
{ id: "deepseek/deepseek-v4-flash-0731", name: "Deepseek V4 Flash 0731" },
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
{ id: "ex/gpt-5.4", name: "Gpt 5.4" },
{ id: "google/gemini-2.5-flash-image", name: "Gemini 2.5 Flash Image" },
{ id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
{ id: "google/gemini-3-pro-image-preview", name: "Gemini 3 Pro Image Preview" },
{ id: "google/gemini-3.1-flash-image-preview", name: "Gemini 3.1 Flash Image Preview" },
{ id: "google/gemini-3.1-flash-lite-image", name: "Gemini 3.1 Flash Lite Image" },
{ id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
{ id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
{ id: "google/gemini-embedding-2", name: "Gemini Embedding 2" },
{ id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B It" },
{ id: "happyhorse-1.0-t2v", name: "Happyhorse 1.0 T2V", kind: "video" },
{ id: "kling-3.0-turbo", name: "Kling 3.0 Turbo", kind: "video" },
{ id: "kling-v2-6", name: "Kling V2 6", kind: "video" },
{ id: "kling-v3", name: "Kling V3", kind: "video" },
{ id: "kling-v3-omni", name: "Kling V3 Omni", kind: "video" },
{ id: "microsoft/mai-image-2.5", name: "Mai Image 2.5" },
{ id: "minimax/minimax-m2-her", name: "Minimax M2 Her" },
{ id: "minimax/minimax-m2.1", name: "Minimax M2.1" },
{ id: "minimax/minimax-m2.1-highspeed", name: "Minimax M2.1 Highspeed" },
{ id: "minimax/minimax-m2.5", name: "Minimax M2.5" },
{ id: "minimax/minimax-m2.7", name: "Minimax M2.7" },
{ id: "minimax/minimax-m2.7-highspeed", name: "Minimax M2.7 Highspeed" },
{ id: "miromind/mirothinker-1-7-deepresearch", name: "Mirothinker 1 7 Deepresearch" },
{ id: "miromind/mirothinker-1-7-deepresearch-mini", name: "Mirothinker 1 7 Deepresearch Mini" },
{ id: "mistralai/devstral-2512", name: "Devstral 2512" },
{ id: "mistralai/mistral-medium-3-5", name: "Mistral Medium 3 5" },
{ id: "mistralai/mistral-small-2603", name: "Mistral Small 2603" },
{ id: "mistralai/voxtral-small-24b-2507", name: "Voxtral Small 24B 2507" },
{ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5" },
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
{ id: "moonshotai/kimi-k3", name: "Kimi K3" },
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
{ id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", name: "Nemotron 3 Nano Omni 30B A3B Reasoning:Free" },
{ id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B" },
{ id: "openai/gpt-4o-mini", name: "Gpt 4O Mini" },
{ id: "openai/gpt-5", name: "Gpt 5" },
{ id: "openai/gpt-5-image", name: "Gpt 5 Image" },
{ id: "openai/gpt-5-image-mini", name: "Gpt 5 Image Mini" },
{ id: "openai/gpt-5-mini", name: "Gpt 5 Mini" },
{ id: "openai/gpt-5.2", name: "Gpt 5.2" },
{ id: "openai/gpt-5.4", name: "Gpt 5.4" },
{ id: "openai/gpt-5.4-image-2", name: "Gpt 5.4 Image 2", kind: "image" },
{ id: "openai/gpt-5.4-mini", name: "Gpt 5.4 Mini" },
{ id: "openai/gpt-5.4-nano", name: "Gpt 5.4 Nano" },
{ id: "openai/gpt-5.4-pro", name: "Gpt 5.4 Pro" },
{ id: "openai/gpt-5.5", name: "Gpt 5.5" },
{ id: "openai/gpt-5.5-pro", name: "Gpt 5.5 Pro" },
{ id: "openai/gpt-5.6-luna", name: "Gpt 5.6 Luna" },
{ id: "openai/gpt-5.6-sol", name: "Gpt 5.6 Sol" },
{ id: "openai/gpt-5.6-terra", name: "Gpt 5.6 Terra" },
{ id: "openai/gpt-audio", name: "Gpt Audio", kind: "audio" },
{ id: "openai/gpt-audio-mini", name: "Gpt Audio Mini", kind: "audio" },
{ id: "openai/gpt-oss-120b", name: "Gpt Oss 120B" },
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
{ id: "qwen/qwen3-coder-next", name: "Qwen3 Coder Next" },
{ id: "qwen/qwen3.5-122b-a10b", name: "Qwen3.5 122B A10B" },
{ id: "qwen/qwen3.5-35b-a3b", name: "Qwen3.5 35B A3B" },
{ id: "qwen/qwen3.5-397b-a17b", name: "Qwen3.5 397B A17B" },
{ id: "qwen/qwen3.5-9b", name: "Qwen3.5 9B" },
{ id: "qwen/qwen3.5-flash", name: "Qwen3.5 Flash" },
{ id: "qwen/qwen3.5-plus-02-15", name: "Qwen3.5 Plus 02 15" },
{ id: "qwen/qwen3.6-plus", name: "Qwen3.6 Plus" },
{ id: "qwen/qwen3.7-max", name: "Qwen3.7 Max" },
{ id: "qwen/qwen3.7-plus", name: "Qwen3.7 Plus" },
{ id: "qwen/qwen3.8-max", name: "Qwen3.8 Max" },
{ id: "qwen3.5-omni-plus", name: "Qwen3.5 Omni Plus" },
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
{ id: "sakana/fugu-ultra", name: "Fugu Ultra" },
{ id: "seed-2-0-code-preview-260328", name: "Seed 2 0 Code Preview 260328" },
{ id: "seed-2-0-lite-260428", name: "Seed 2 0 Lite 260428" },
{ id: "seed-2-0-mini-260428", name: "Seed 2 0 Mini 260428" },
{ id: "seed-2-0-pro-260328", name: "Seed 2 0 Pro 260328" },
{ id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash" },
{ id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash" },
{ id: "tencent/hy3-preview", name: "Hy3 Preview" },
{ id: "x-ai/grok-4.1-fast", name: "Grok 4.1 Fast" },
{ id: "x-ai/grok-4.20-beta", name: "Grok 4.20 Beta" },
{ id: "x-ai/grok-4.3", name: "Grok 4.3" },
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
{ id: "x-ai/grok-build-0.1", name: "Grok Build 0.1" },
{ id: "xiaomi/mimo-v2-flash", name: "Mimo V2 Flash" },
{ id: "xiaomi/mimo-v2-omni", name: "Mimo V2 Omni" },
{ id: "xiaomi/mimo-v2-pro", name: "Mimo V2 Pro" },
{ id: "xiaomi/mimo-v2.5", name: "Mimo V2.5" },
{ id: "xiaomi/mimo-v2.5-pro", name: "Mimo V2.5 Pro" },
{ id: "z-ai/glm-4.5-air", name: "Glm 4.5 Air" },
{ id: "z-ai/glm-4.6", name: "Glm 4.6" },
{ id: "z-ai/glm-4.6v", name: "Glm 4.6V" },
{ id: "z-ai/glm-4.7", name: "Glm 4.7" },
{ id: "z-ai/glm-5", name: "Glm 5" },
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
{ id: "z-ai/glm-5.1", name: "Glm 5.1" },
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
{ id: "z-ai/glm-5.3-free", name: "Glm 5.3 Free" },
{ id: "z-ai/glm-5.2", name: "Glm 5.2" },
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
],
serviceKinds: ["llm", "embedding", "image"],
embeddingConfig: {

View File

@@ -36,6 +36,8 @@ export default {
},
},
models: [
{ id: "grok-4.6", name: "Grok 4.6" },
{ id: "grok-4.5", name: "Grok 4.5" },
{ id: "grok-4", name: "Grok 4" },
{ id: "grok-4-fast-reasoning", name: "Grok 4 Fast Reasoning" },
{ id: "grok-code-fast-1", name: "Grok Code Fast" },

View File

@@ -0,0 +1,35 @@
export default {
id: "xquik",
alias: "xquik",
display: {
name: "Xquik",
icon: "tag",
color: "#5C3327",
textIcon: "XQ",
website: "https://docs.xquik.com/api-reference/x/search-tweets",
notice: {
apiKeyUrl: "https://xquik.com",
text: "Searches public X posts. Billing uses 1 Xquik credit per returned post."
}
},
category: "apikey",
authType: "apikey",
serviceKinds: [
"webSearch"
],
searchConfig: {
baseUrl: "https://xquik.com/api/v1/x/tweets/search",
validateUrl: "https://xquik.com/api/v1/credits",
method: "GET",
authType: "apikey",
authHeader: "x-api-key",
searchTypes: [
"x"
],
defaultMaxResults: 5,
maxMaxResults: 100,
timeoutMs: 10000,
cacheTTLMs: 60000,
creditsPerResult: 1
}
};

View File

@@ -22,6 +22,7 @@ export function mapStainlessArch() {
// Anthropic API version (single source — reused across claude-format providers/executors)
export const ANTHROPIC_API_VERSION = "2023-06-01";
export const CLAUDE_CLI_VERSION = "2.1.258";
// Shared Claude-compatible API headers (reused across claude-format providers)
export const CLAUDE_API_HEADERS = {
@@ -34,7 +35,7 @@ export const CLAUDE_CLI_SPOOF_HEADERS = {
"Anthropic-Version": ANTHROPIC_API_VERSION,
"Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28",
"Anthropic-Dangerous-Direct-Browser-Access": "true",
"User-Agent": "claude-cli/2.1.92 (external, sdk-cli)",
"User-Agent": `claude-cli/${CLAUDE_CLI_VERSION} (external, sdk-cli)`,
"X-App": "cli",
"X-Stainless-Helper-Method": "stream",
"X-Stainless-Retry-Count": "0",
@@ -74,10 +75,10 @@ export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
export const OPENAI_COMPAT_BASE = "https://api.openai.com/v1";
export const ANTHROPIC_COMPAT_BASE = "https://api.anthropic.com/v1";
// Official Antigravity IDE Desktop 2.1.1 fingerprint captured from macOS arm64.
// Official Antigravity IDE Desktop 2.11.0 fingerprint captured from macOS arm64.
// Keep this static even when 9router runs on Linux: the provider profile is
// intentionally matching the IDE client, not the server host.
export const ANTIGRAVITY_IDE_VERSION = "2.1.1";
export const ANTIGRAVITY_IDE_VERSION = "2.11.0";
export const ANTIGRAVITY_IDE_BASE_URL = "https://daily-cloudcode-pa.googleapis.com";
export const ANTIGRAVITY_IDE_USER_AGENT = `antigravity/ide/${ANTIGRAVITY_IDE_VERSION} darwin/arm64`;

View File

@@ -35,10 +35,23 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
const PATTERN_THINKING = [
{ provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS },
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
// codebuddy-cn per-model effort sets — the server's product-config payload
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
// → all 200), but values outside a model's supportedEfforts are silently
// clamped, so the declared set stays authoritative for the picker. Models
// that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x /
// kimi-k3-1 / minimax-m3) fall through to the openai format default.
{ provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] },
{ provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
];
// Returns valid thinking levels for a model, or null when the model has no reasoning.

View File

@@ -0,0 +1,42 @@
// Name-based vision detection — last resort when neither the catalog file nor
// the capability tables know a model. Vendors put the modality in the id
// ("qwen3-vl-plus", "glm-4.6v", "deepseek-v4-flash-vision-exp"), so a custom or
// freshly released model still gets image input instead of silently dropping it.
//
// Only ever turns vision ON. Never used to turn a declared capability off.
const SEP = "[-_/:.]";
// Image GENERATION, video generation, and non-chat models also carry these
// words but take no image input — checked first so they can never match.
const NOT_VISION = new RegExp(
[
`(^|${SEP})(image|img)(${SEP}|$)`,
"stable-image", "gen[0-9]_image", "nanobanana", "imagine",
"t2v", "i2v", "flux", "dall", "sdxl", "diffusion",
"embed", "rerank", "guard", "moderation",
"tts", "stt", "whisper", "voice", "speech", "audio",
].join("|"),
"i"
);
// Explicit modality words, plus the "<digit>v" suffix vendors use for vision
// variants (glm-4.6v, glm-5v-turbo). The digit-v branch requires a dotted
// version so the never-shipped `gpt-4v` cannot match.
const VISION_NAME = new RegExp(
[
`(^|${SEP})(vision|vl|vlm|multimodal|omni|visual)(${SEP}|$)`,
`[0-9]\\.[0-9]+v(${SEP}|$)`,
`(^|${SEP})glm-[0-9]+v(${SEP}|$)`,
"(^|[-_/:.])(llava|pixtral|internvl|cogvlm|minicpm-v|moondream|idefics|fuyu)",
].join("|"),
"i"
);
// Does this model id look like a vision model? Name signal only.
export function looksLikeVisionModel(modelId) {
if (!modelId) return false;
const id = String(modelId).toLowerCase();
if (NOT_VISION.test(id)) return false;
return VISION_NAME.test(id);
}

View File

@@ -7,6 +7,12 @@ import {
const DEFAULT_TIMEOUT_MS = 3000;
function normalizeTimeout(value) {
return typeof value === "number" && Number.isFinite(value) && value > 0
? value
: DEFAULT_TIMEOUT_MS;
}
function jsonBytes(value) {
try {
return new TextEncoder().encode(JSON.stringify(value) || "").length;
@@ -240,6 +246,7 @@ async function callCompress(url, messages, model, timeoutMs, compressUserMessage
// /v1/compress only understands OpenAI shape, so Claude bodies are translated
// to OpenAI, compressed, then translated back using 9Router's own translators.
export async function compressWithHeadroom(body, { enabled, url, model, format, compressUserMessages, timeoutMs = DEFAULT_TIMEOUT_MS, diagnostics = null } = {}) {
timeoutMs = normalizeTimeout(timeoutMs);
if (!enabled) {
setDiagnostic(diagnostics, "disabled");
return null;
@@ -281,7 +288,10 @@ export async function compressWithHeadroom(body, { enabled, url, model, format,
return null;
}
const oai = openaiResponsesToOpenAIRequest(model, body, false);
if (!Array.isArray(oai?.messages)) return null;
if (!Array.isArray(oai?.messages)) {
setDiagnostic(diagnostics, "openai-responses request did not translate to messages[]");
return null;
}
const data = await callCompress(url, oai.messages, model, timeoutMs, compressUserMessages, diagnostics || {});
if (!data) return null;
// input: undefined so the translator rebuilds input from the compressed

View File

@@ -3,96 +3,335 @@
// native-passthrough flows. Used by caveman.js and ponytail.js.
import { FORMATS } from "../translator/formats.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK, RESPONSES_ITEM } from "../translator/schema/blocks.js";
import { ROLE } from "../translator/schema/roles.js";
const SEP = "\n\n";
export function injectSystemPrompt(body, format, prompt) {
if (!body || !prompt) return;
try {
if (!body || !prompt) return;
if (typeof body !== "object") return;
switch (format) {
case FORMATS.CLAUDE:
// Kiro wire shape is unique (conversationState/systemPrompt) — handle directly.
if (isKiroBody(body) || format === FORMATS.KIRO) {
injectKiroSystem(body, prompt);
return;
}
// Claude/Gemini own a dedicated system field, yet their bodies also carry
// messages[]/contents[] — decide by format label before the shape sniff below.
// Anthropic rejects a "system" role inside messages[] (no such input role).
if (format === FORMATS.CLAUDE) {
injectClaudeSystem(body, prompt);
return;
case FORMATS.GEMINI:
case FORMATS.GEMINI_CLI:
case FORMATS.VERTEX:
case FORMATS.ANTIGRAVITY:
}
if (format === FORMATS.GEMINI || format === FORMATS.GEMINI_CLI
|| format === FORMATS.VERTEX || format === FORMATS.ANTIGRAVITY) {
// Antigravity wraps Gemini shape in body.request → injectGeminiSystem handles it
injectGeminiSystem(body, prompt);
return;
default:
// OpenAI and OpenAI-shaped formats (responses/codex/cursor/kiro/ollama)
injectMessagesSystem(body, prompt);
}
}
// OpenAI-shaped: messages[] (chat) or input[] (responses) or instructions (responses string)
function injectMessagesSystem(body, prompt) {
// OpenAI Responses API: top-level string field
if (typeof body.instructions === "string") {
body.instructions = body.instructions
? `${body.instructions}${SEP}${prompt}`
: prompt;
return;
}
const arr = Array.isArray(body.messages) ? body.messages
: Array.isArray(body.input) ? body.input
: null;
if (!arr) return;
const idx = arr.findIndex(m => m && (m.role === "system" || m.role === "developer"));
if (idx >= 0) {
appendToOpenAIMessage(arr[idx], prompt);
} else {
arr.unshift({ role: "system", content: prompt });
}
}
function appendToOpenAIMessage(msg, prompt) {
if (typeof msg.content === "string") {
msg.content = `${msg.content}${SEP}${prompt}`;
} else if (Array.isArray(msg.content)) {
// Responses-style array of parts {type:"input_text"|"text", text}
msg.content.push({ type: "input_text", text: prompt });
} else {
msg.content = prompt;
}
}
// Claude shape: body.system as string | array of {type:"text", text}
// Insert before the last cache_control block to keep injection inside the cached prefix.
function injectClaudeSystem(body, prompt) {
if (typeof body.system === "string" && body.system.length > 0) {
body.system = `${body.system}${SEP}${prompt}`;
return;
}
if (Array.isArray(body.system)) {
const block = { type: "text", text: prompt };
let lastCacheIdx = -1;
for (let i = body.system.length - 1; i >= 0; i--) {
if (body.system[i]?.cache_control) { lastCacheIdx = i; break; }
}
if (lastCacheIdx >= 0) {
body.system.splice(lastCacheIdx, 0, block);
// Dispatch by actual wire shape for OpenAI-shaped formats.
// instructions string takes precedence; messages[] means Chat; input[] means Responses.
if (typeof body.instructions === "string") {
injectInstructionsSystem(body, prompt);
return;
}
if (Array.isArray(body.messages)) {
injectChatSystem(body, prompt);
return;
}
if (Array.isArray(body.input)) {
// Responses input[]: empty array already normalized elsewhere; string stays untouched here
injectResponsesInputSystem(body, prompt);
return;
}
if (typeof body.input === "string") {
// string input must stay untouched
return;
}
// OpenAI-shaped but no array (e.g. empty body) — no-op
} catch (_) {
// fail-open
}
}
function isKiroBody(body) {
if (!body || typeof body !== "object") return false;
if (typeof body.systemPrompt !== "string") return false;
const cs = body.conversationState;
if (!cs || typeof cs !== "object") return false;
return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object");
}
// Exact idempotency: prompt present as its own SEP-delimited segment (or the
// whole string), not as a substring of unrelated text.
function hasPrompt(haystack, prompt) {
if (!haystack || typeof haystack !== "string") return false;
if (haystack === prompt) return true;
return haystack.split(SEP).includes(prompt);
}
function dedupStringAppend(curr, prompt) {
if (!curr) return prompt;
if (hasPrompt(curr, prompt)) return curr;
return `${curr}${SEP}${prompt}`;
}
// ---- OpenAI instructions string ----
function injectInstructionsSystem(body, prompt) {
try {
const curr = body.instructions;
if (typeof curr !== "string") return;
if (hasPrompt(curr, prompt)) return;
const next = curr ? `${curr}${SEP}${prompt}` : prompt;
try { body.instructions = next; } catch (_) { /* frozen/proxy fail-open */ }
} catch (_) {}
}
// ---- Chat messages[] ----
function injectChatSystem(body, prompt) {
try {
const arr = body.messages;
if (!Array.isArray(arr)) return;
// Exact idempotency: scan existing system/developer content for full prompt
if (containsPromptInMessages(arr, prompt)) return;
let idx = -1;
try { idx = arr.findIndex(m => m && (m.role === ROLE.SYSTEM || m.role === ROLE.DEVELOPER)); } catch (_) { return; }
if (idx >= 0) {
appendToChatMessage(arr[idx], prompt);
} else {
body.system.push(block);
// create typed system message at index 0; fail-open on frozen/proxy
try { arr.unshift({ role: ROLE.SYSTEM, content: prompt }); } catch (_) {}
}
return;
}
body.system = prompt;
} catch (_) {}
}
// Gemini shape: body.system_instruction | body.systemInstruction | body.request.systemInstruction
// Each shape: { parts: [{ text }] }
function injectGeminiSystem(body, prompt) {
const target = body.request && typeof body.request === "object" ? body.request : body;
const useSnake = Object.prototype.hasOwnProperty.call(target, "system_instruction");
const key = useSnake ? "system_instruction" : "systemInstruction";
const sys = target[key];
if (sys && Array.isArray(sys.parts)) {
sys.parts.push({ text: prompt });
return;
}
target[key] = { parts: [{ text: prompt }] };
function containsPromptInMessages(arr, prompt) {
try {
for (const m of arr) {
if (!m || (m.role !== ROLE.SYSTEM && m.role !== ROLE.DEVELOPER)) continue;
const c = m.content;
if (typeof c === "string" && hasPrompt(c, prompt)) return true;
if (Array.isArray(c)) {
for (const part of c) {
if (part && typeof part.text === "string" && hasPrompt(part.text, prompt)) return true;
}
}
}
} catch (_) {}
return false;
}
function appendToChatMessage(msg, prompt) {
try {
if (!msg || typeof msg !== "object") return;
const c = msg.content;
if (typeof c === "string") {
const next = dedupStringAppend(c, prompt);
if (next === c) return;
// avoid partial mutation: try assignment, bail if setter throws
try { msg.content = next; } catch (_) {}
return;
}
if (Array.isArray(c)) {
// already deduped at message level; but guard block-level too
try {
if (c.some(b => b && b.text === prompt)) return;
} catch (_) {}
try { c.push({ type: OPENAI_BLOCK.TEXT, text: prompt }); } catch (_) {}
return;
}
try { msg.content = prompt; } catch (_) {}
} catch (_) {}
}
// ---- Responses input[] ----
function injectResponsesInputSystem(body, prompt) {
try {
const arr = body.input;
if (!Array.isArray(arr)) return;
// instructions already handled above
if (containsPromptInResponsesInput(arr, prompt)) return;
// find system/developer message items only (type === message)
let idx = -1;
try {
idx = arr.findIndex(m => m && m.type === RESPONSES_ITEM.MESSAGE && (m.role === ROLE.SYSTEM || m.role === ROLE.DEVELOPER));
} catch (_) { return; }
if (idx >= 0) {
appendToResponsesMessage(arr[idx], prompt);
} else {
const msg = { type: RESPONSES_ITEM.MESSAGE, role: ROLE.SYSTEM, content: [{ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }] };
try { arr.unshift(msg); } catch (_) {}
}
} catch (_) {}
}
function containsPromptInResponsesInput(arr, prompt) {
try {
for (const item of arr) {
if (!item || item.type !== RESPONSES_ITEM.MESSAGE) continue;
if (item.role !== ROLE.SYSTEM && item.role !== ROLE.DEVELOPER) continue;
const c = item.content;
if (typeof c === "string" && hasPrompt(c, prompt)) return true;
if (Array.isArray(c)) {
for (const part of c) {
if (part && typeof part.text === "string" && hasPrompt(part.text, prompt)) return true;
}
}
}
} catch (_) {}
return false;
}
function appendToResponsesMessage(msg, prompt) {
try {
if (!msg || typeof msg !== "object") return;
const c = msg.content;
if (typeof c === "string") {
const next = dedupStringAppend(c, prompt);
if (next === c) return;
try { msg.content = next; } catch (_) {}
return;
}
if (Array.isArray(c)) {
try { if (c.some(b => b && b.text === prompt)) return; } catch (_) {}
try { c.push({ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }); } catch (_) {}
return;
}
try { msg.content = [{ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }]; } catch (_) {}
} catch (_) {}
}
// ---- Claude ----
function injectClaudeSystem(body, prompt) {
try {
const sys = body.system;
if (typeof sys === "string") {
if (hasPrompt(sys, prompt)) return;
const next = sys.length > 0 ? `${sys}${SEP}${prompt}` : prompt;
try { body.system = next; } catch (_) {}
return;
}
if (Array.isArray(sys)) {
try { if (sys.some(b => b && b.text === prompt)) return; } catch (_) {}
const block = { type: CLAUDE_BLOCK.TEXT, text: prompt };
let lastCacheIdx = -1;
try {
for (let i = sys.length - 1; i >= 0; i--) {
if (sys[i]?.cache_control) { lastCacheIdx = i; break; }
}
} catch (_) {}
try {
if (lastCacheIdx >= 0) sys.splice(lastCacheIdx, 0, block);
else sys.push(block);
} catch (_) {}
return;
}
// absent/null
try { body.system = prompt; } catch (_) {}
} catch (_) {}
}
// ---- Gemini ----
function injectGeminiSystem(body, prompt) {
try {
let target = body;
try {
if (body.request && typeof body.request === "object") target = body.request;
} catch (_) {}
let useSnake = false;
try { useSnake = Object.prototype.hasOwnProperty.call(target, "system_instruction"); } catch (_) {}
const key = useSnake ? "system_instruction" : "systemInstruction";
let sys;
try { sys = target[key]; } catch (_) { sys = undefined; }
if (sys && Array.isArray(sys.parts)) {
try { if (sys.parts.some(p => p && p.text === prompt)) return; } catch (_) {}
try { sys.parts.push({ text: prompt }); } catch (_) {}
return;
}
try { target[key] = { parts: [{ text: prompt }] }; } catch (_) {}
} catch (_) {}
}
// ---- Kiro ----
// Updates top-level systemPrompt and only the mirrored leading prefix of the
// first user history turn, else current user. next = old + SEP + prompt.
// Replace old leading prefix only; preserve time context and user tail.
function injectKiroSystem(body, prompt) {
try {
let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : "";
// Repair path: a previous partial write left systemPrompt updated but user
// content still mirroring the pre-write prefix. Re-derive the effective old
// prefix from content so this pass converges instead of early-returning.
const cs0 = body.conversationState;
let firstUser0 = cs0 && Array.isArray(cs0.history)
? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null)
: null;
if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage;
if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) {
const c0 = firstUser0.content;
if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) {
// systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored
oldPrompt = "";
}
}
if (oldPrompt && hasPrompt(oldPrompt, prompt)) return;
const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt;
// Atomicity: write user content first, then systemPrompt only if content
// write succeeded (or was a no-op). If systemPrompt write then fails, the
// repair heuristic above re-derives from content on retry — no permanent
// half-applied state.
const cs = body.conversationState;
let targetMsg = null;
try {
const hist = Array.isArray(cs?.history) ? cs.history : null;
if (hist) {
for (const item of hist) {
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
}
}
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
targetMsg = cs.currentMessage.userInputMessage;
}
} catch (_) { targetMsg = null; }
let sysWritten = false;
try { body.systemPrompt = next; sysWritten = true; } catch (_) {}
const applyContent = () => {
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
if (oldPrompt === "") {
// Empty old prompt: prepend unless already at head (exact, not substring)
if (content.startsWith(prompt) || content.startsWith(next)) return;
const newContent = content ? `${next}${SEP}${content}` : next;
try { targetMsg.content = newContent; } catch (_) {}
return;
}
if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone
if (content.startsWith(next)) return; // already applied → idempotent
const tail = content.slice(oldPrompt.length);
try { targetMsg.content = `${next}${tail}`; } catch (_) {}
};
try {
if (targetMsg) applyContent();
} catch (_) {}
if (sysWritten && targetMsg) {
// verify convergence: content should now start with next (or be un-mirrored)
let ok = false;
try {
const c = targetMsg.content;
ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt);
} catch (_) {}
if (!ok) {
try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback
}
}
} catch (_) {}
}

View File

@@ -203,7 +203,8 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA };
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
const MAX_ATTEMPTS = 5;
const MAX_ATTEMPTS = Number(process.env.ONBOARD_MAX_ATTEMPTS) || 2;
const BASE_RETRY_DELAY_MS = Number(process.env.ONBOARD_RETRY_DELAY_MS) || 12_000;
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
// Bail out immediately if the connection was removed
@@ -241,9 +242,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
throw new Error("onboardUser done but no project_id in response");
}
// Server not done yet – wait and retry
// Server not done yet – wait and retry with jitter
const jitter = Math.floor(Math.random() * 5000);
console.log(`[ProjectId] Onboard attempt ${attempt}/${MAX_ATTEMPTS}: not done yet, waiting...`);
await new Promise(resolve => setTimeout(resolve, 2000));
await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter));
} catch (error) {
clearTimeout(timeoutId);
@@ -256,9 +258,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
console.warn(`[ProjectId] onboardUser failed after ${MAX_ATTEMPTS} attempts: ${error.message}`);
return null;
}
// Continue to next attempt instead of throwing (which would skip remaining retries)
// Wait with jitter before retrying
const jitter = Math.floor(Math.random() * 5000);
console.warn(`[ProjectId] onboardUser attempt ${attempt} failed: ${error.message}, retrying...`);
await new Promise(resolve => setTimeout(resolve, 2000));
await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter));
} finally {
clearTimeout(timeoutId);
externalSignal?.removeEventListener("abort", forwardAbort);

View File

@@ -0,0 +1,170 @@
import { makeKv } from "../../src/lib/db/helpers/kvStore.js";
const MAX_SIGNATURES = 2000;
const MAX_PERSISTED_SIGNATURES = 10_000;
const MEMORY_TTL_MS = 1000 * 60 * 60; // 1 hour
const PERSISTED_TTL_MS = 1000 * 60 * 60 * 24 * 7; // 7 days
const SCOPE = "gemini_thought_signatures";
const signatureKv = makeKv(SCOPE);
const memorySignatures = new Map();
let pruneCounter = 0;
function pruneMemoryExpired() {
const now = Date.now();
for (const [key, value] of memorySignatures.entries()) {
if (value.expiresAt <= now) {
memorySignatures.delete(key);
}
}
while (memorySignatures.size > MAX_SIGNATURES) {
const oldestKey = memorySignatures.keys().next().value;
if (!oldestKey) break;
memorySignatures.delete(oldestKey);
}
}
async function maybePrunePersisted() {
pruneCounter++;
if (pruneCounter % 100 !== 0) return;
try {
const all = await signatureKv.getAll();
const keys = Object.keys(all);
const now = Date.now();
const expiredKeys = [];
const valid = [];
for (const k of keys) {
const entry = all[k];
if (!entry || typeof entry.signature !== "string" || (entry.expiresAt && entry.expiresAt <= now)) {
expiredKeys.push(k);
} else {
valid.push({ key: k, createdAt: entry.createdAt || 0 });
}
}
for (const k of expiredKeys) {
await signatureKv.remove(k).catch(() => {});
}
if (valid.length > MAX_PERSISTED_SIGNATURES) {
valid.sort((a, b) => b.createdAt - a.createdAt);
const toRemove = valid.slice(MAX_PERSISTED_SIGNATURES);
for (const item of toRemove) {
await signatureKv.remove(item.key).catch(() => {});
}
}
} catch {
// Fail-open
}
}
/**
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async)
*/
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) {
if (typeof toolCallId !== "string" || !toolCallId) return;
if (typeof signature !== "string" || !signature) return;
const now = Date.now();
pruneMemoryExpired();
const keys = [];
if (sessionId && typeof sessionId === "string") {
keys.push(`${sessionId}:${toolCallId}`);
}
keys.push(toolCallId);
for (const k of keys) {
memorySignatures.set(k, {
signature,
expiresAt: now + MEMORY_TTL_MS,
});
// Async persist to SQLite kv table without blocking
signatureKv.set(k, {
signature,
createdAt: now,
expiresAt: now + PERSISTED_TTL_MS,
}).catch(() => {});
}
maybePrunePersisted().catch(() => {});
}
/**
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback)
*/
export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
if (typeof toolCallId !== "string" || !toolCallId) return null;
pruneMemoryExpired();
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionEntry = memorySignatures.get(sessionKey);
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
return sessionEntry.signature;
}
}
const entry = memorySignatures.get(toolCallId);
if (entry && entry.expiresAt > Date.now()) {
return entry.signature;
}
try {
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionRow = await signatureKv.get(sessionKey);
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) {
memorySignatures.set(sessionKey, {
signature: sessionRow.signature,
expiresAt: Date.now() + MEMORY_TTL_MS,
});
return sessionRow.signature;
}
}
const row = await signatureKv.get(toolCallId);
if (row && typeof row.signature === "string") {
if (row.expiresAt && row.expiresAt <= Date.now()) {
signatureKv.remove(toolCallId).catch(() => {});
return null;
}
memorySignatures.set(toolCallId, {
signature: row.signature,
expiresAt: Date.now() + MEMORY_TTL_MS,
});
return row.signature;
}
} catch {
// Fail-open
}
return null;
}
/**
* Synchronous get from RAM cache only (for sync translators)
*/
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) {
if (typeof toolCallId !== "string" || !toolCallId) return null;
pruneMemoryExpired();
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionEntry = memorySignatures.get(sessionKey);
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
return sessionEntry.signature;
}
}
const entry = memorySignatures.get(toolCallId);
if (entry && entry.expiresAt > Date.now()) {
return entry.signature;
}
return null;
}

View File

@@ -4,6 +4,7 @@ import {
refreshXaiToken,
refreshAccessToken,
refreshKimiToken,
refreshClineToken,
refreshClaudeOAuthToken,
refreshGoogleToken,
refreshCodexToken,
@@ -23,6 +24,7 @@ import {
export {
refreshAccessToken,
refreshKimiToken,
refreshClineToken,
refreshClaudeOAuthToken,
refreshGoogleToken,
refreshCodexToken,
@@ -145,6 +147,7 @@ const REFRESH_HANDLERS = {
"codebuddy-cn": (c, log) => refreshCodebuddyToken(c.refreshToken, log),
"codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log),
trae: (c, log) => refreshTraeToken(c.refreshToken, c, log),
cline: (c, log) => refreshClineToken(c.refreshToken, log),
zed: () => refreshZedToken(),
windsurf: (c, log) => refreshWindsurfToken(c, log),
// Kimi Code OAuth (merged into id `kimi`); legacy id still routes here

View File

@@ -147,6 +147,53 @@ export async function refreshKimiToken(refreshToken, credentials, log) {
return refreshAccessToken("kimi", refreshToken, credentials, log);
}
export async function refreshClineToken(refreshToken, log) {
if (!refreshToken) return null;
return dedupRefresh("cline", refreshToken, async () => {
try {
const response = await fetch(PROVIDERS.cline?.refreshUrl, {
method: "POST",
headers: {
"Content-Type": "application/json",
Accept: "application/json",
},
body: JSON.stringify({
refreshToken,
grantType: "refresh_token",
clientType: "extension",
}),
});
if (!response.ok) {
const errorText = await response.text();
log?.error?.("TOKEN_REFRESH", "Failed to refresh Cline token", {
status: response.status,
error: errorText,
});
return null;
}
const body = await response.json();
const tokens = body?.data || body;
if (!tokens?.accessToken) return null;
const expiresIn = tokens.expiresAt
? Math.max(1, Math.floor((new Date(tokens.expiresAt).getTime() - Date.now()) / 1000))
: (tokens.expiresIn || tokens.expires_in || 3600);
return {
accessToken: tokens.accessToken,
refreshToken: tokens.refreshToken || refreshToken,
expiresIn,
};
} catch (error) {
log?.error?.("TOKEN_REFRESH", `Error refreshing Cline token: ${error.message}`);
return null;
}
}, log);
}
// Claude OAuth: JSON body, client_id only. Delegate to refreshAccessToken("claude", ...).
export async function refreshClaudeOAuthToken(refreshToken, log) {
return refreshAccessToken("claude", refreshToken, {}, log);

View File

@@ -15,12 +15,14 @@ import { getXaiUsage } from "./usage/xai.js";
import { getGrokCliUsage } from "./usage/grok-cli.js";
import { getKimiUsage } from "./usage/kimi.js";
import { getDeepseekUsage } from "./usage/deepseek.js";
import { getCommandCodeUsage } from "./usage/commandcode.js";
import { getOpenCodeGoUsage } from "./usage/opencode-go.js";
import { getGroqUsage } from "./usage/groq.js";
import { getZedUsage } from "./usage/zed.js";
import { resolveQoderCredentials } from "./qoderModels.js";
import { getGlmUsage } from "./usage/glm.js";
import {
getIflowUsage,
getOllamaUsage,
getGlmUsage,
getVercelAiGatewayUsage,
getQoderUsage,
} from "./usage/misc.js";
@@ -56,8 +58,10 @@ const USAGE_HANDLERS = {
"codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData),
"opencode-go": (c) => getOpenCodeGoUsage(c.apiKey, c.proxyOptions),
deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions),
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
};
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {

View File

@@ -102,14 +102,34 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
quotas["weekly (7d)"] = createQuotaObject(data.seven_day);
}
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus)
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus, seven_day_fable)
const MODEL_DISPLAY_NAMES = {
fable_5_1: "fable",
fable_5: "fable",
};
for (const [key, value] of Object.entries(data)) {
if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) {
const modelName = key.replace("seven_day_", "");
const rawName = key.replace("seven_day_", "");
const modelName = MODEL_DISPLAY_NAMES[rawName] || rawName;
quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value);
} else if ((key === "fable" || key === "fable_5" || key === "fable_5_1") && hasUtilization(value)) {
quotas["weekly fable (7d)"] = createQuotaObject(value);
}
}
// Fallback: surface Fable quota row if weekly window exists but Fable was not returned yet
if (!quotas["weekly fable (7d)"] && hasUtilization(data.seven_day)) {
quotas["weekly fable (7d)"] = {
used: 0,
total: 100,
remaining: 100,
remainingPercentage: 100,
resetAt: parseResetTime(data.seven_day.resets_at),
unlimited: false,
};
}
return {
plan: "Claude Code",
extraUsage: data.extra_usage ?? null,

View File

@@ -21,6 +21,13 @@ function toIsoDate(value) {
return Number.isFinite(time) ? date.toISOString() : null;
}
function errorMessage(value, fallback) {
if (!value) return fallback;
if (typeof value === "string") return value;
if (typeof value.message === "string") return value.message;
return JSON.stringify(value);
}
function getCodexAccountId(providerSpecificData) {
return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null;
}
@@ -80,6 +87,23 @@ function getCodexReviewRateLimit(data) {
}) || null;
}
function getCodexSparkRateLimit(data) {
if (data.spark_rate_limit || data.gpt_5_3_codex_spark_rate_limit) {
return data.spark_rate_limit || data.gpt_5_3_codex_spark_rate_limit;
}
const byLimitId = data.rate_limits_by_limit_id;
if (byLimitId && typeof byLimitId === "object" && !Array.isArray(byLimitId)) {
return byLimitId["gpt-5.3-codex-spark"] || byLimitId.gpt_5_3_codex_spark || byLimitId.spark || null;
}
const additional = Array.isArray(data.additional_rate_limits) ? data.additional_rate_limits : [];
return additional.find((entry) => {
const id = String(entry?.limit_name || entry?.metered_feature || entry?.id || "").toLowerCase();
return id.includes("spark") || id.includes("5.3-codex-spark");
}) || null;
}
export async function getCodexUsage(accessToken, proxyOptions = null) {
try {
const response = await proxyAwareFetch(CODEX_CONFIG.usageUrl, {
@@ -97,16 +121,19 @@ export async function getCodexUsage(accessToken, proxyOptions = null) {
const data = await response.json();
const normalRateLimit = data.rate_limit || data.rate_limits || data.rate_limits_by_limit_id?.codex || {};
const reviewRateLimit = getCodexReviewRateLimit(data);
const sparkRateLimit = getCodexSparkRateLimit(data);
const availableResetCredits = Math.max(0, toFiniteNumber(data.rate_limit_reset_credits?.available_count, 0));
const quotas = {};
appendCodexQuotaWindows(quotas, "", normalRateLimit);
appendCodexQuotaWindows(quotas, "review", reviewRateLimit);
appendCodexQuotaWindows(quotas, "spark", sparkRateLimit);
return {
plan: data.plan_type || data.summary?.plan || "unknown",
limitReached: getCodexRateLimitBody(normalRateLimit)?.limit_reached || false,
reviewLimitReached: getCodexRateLimitBody(reviewRateLimit)?.limit_reached || false,
sparkLimitReached: getCodexRateLimitBody(sparkRateLimit)?.limit_reached || false,
resetCredits: { availableCount: availableResetCredits },
quotas,
};
@@ -142,7 +169,7 @@ export async function getCodexRateLimitResetCredits(accessToken, proxyOptions =
}
if (!response.ok) {
const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`;
const message = errorMessage(data?.message || data?.error || data?.detail, `Codex reset credits API unavailable (${response.status}).`);
throw new Error(message);
}

View File

@@ -0,0 +1,88 @@
/**
* GLM Coding Plan usage (international + China regions)
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U } from "./shared.js";
// GLM quota endpoints (region-aware) — url from registry transport.usage
const GLM_QUOTA_URLS = {
international: U("glm").url,
china: U("glm-cn").url,
};
/**
* GLM Coding Plan usage (international + China regions)
* Supports both TOKENS_LIMIT and CREDIT_LIMIT and dynamic intervals (e.g. session 5h, weekly 7d).
*/
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
if (!apiKey) {
return { message: "GLM API key not available." };
}
const region = provider === "glm-cn" ? "china" : "international";
const quotaUrl = GLM_QUOTA_URLS[region];
try {
const response = await proxyAwareFetch(
quotaUrl,
{
headers: {
Authorization: `Bearer ${apiKey}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (!response.ok) {
if (response.status === 401) {
return { message: "GLM API key invalid or expired." };
}
return { message: `GLM quota API error (${response.status}).` };
}
const json = await response.json();
const data = json?.data && typeof json.data === "object" ? json.data : {};
const limits = Array.isArray(data.limits) ? data.limits : [];
const quotas = {};
for (const limit of limits) {
// 1. Accept both TOKENS_LIMIT and CREDIT_LIMIT from GLM API
if (!limit || (limit.type !== "TOKENS_LIMIT" && limit.type !== "CREDIT_LIMIT")) continue;
const usedPercent = Number(limit.percentage) || 0;
const resetMs = Number(limit.nextResetTime) || 0;
const remaining = Math.max(0, 100 - usedPercent);
// 2. Map key dynamically based on type and period (unit) to avoid overwriting
let key = "session";
if (limit.unit === 3) {
key = `Session (${limit.number}h)`;
} else if (limit.unit === 6) {
key = "Weekly (7d)";
} else if (limit.type === "TOKENS_LIMIT") {
key = "Tokens";
} else {
key = `Limit (${limit.number})`;
}
quotas[key] = {
used: usedPercent,
total: 100,
remaining,
remainingPercentage: remaining,
resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null,
unlimited: false,
};
}
const levelRaw = typeof data.level === "string" ? data.level : "";
const plan = levelRaw
? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase()
: "Unknown";
return { plan, quotas };
} catch (error) {
return { message: `GLM error: ${error.message}` };
}
}

View File

@@ -161,6 +161,9 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
if (data.models) {
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
const importantModels = [
'gemini-3.8-flash-high',
'gemini-3.8-flash-medium',
'gemini-3.8-flash-low',
'gemini-3.7-flash-high',
'gemini-3.7-flash-medium',
'gemini-3.7-flash-low',

View File

@@ -0,0 +1,133 @@
/**
* Groq usage — no dedicated quota endpoint. Rate-limit info instead rides on
* every API response as x-ratelimit-* headers (requests + tokens, always
* included). We piggyback on the models list (already used as
* transport.validateUrl) so reading usage never costs tokens.
*
* Headers:
* x-ratelimit-limit-requests / x-ratelimit-remaining-requests
* x-ratelimit-limit-tokens / x-ratelimit-remaining-tokens
* x-ratelimit-reset-requests / x-ratelimit-reset-tokens (duration strings, e.g. "2m59.56s")
*
* Docs: https://console.groq.com/docs/rate-limits
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U } from "./shared.js";
const MODELS_URL = U("groq").url;
// Groq reset headers are Go-style duration strings ("2m59.56s", "7.66s"), not
// timestamps — parse the h/m/s/ms components and add them to now().
function parseGroqDurationMs(value) {
if (typeof value !== "string" || !value.trim()) return null;
const re = /(\d+(?:\.\d+)?)(ms|s|m|h)/g;
let match;
let totalMs = 0;
let matched = false;
while ((match = re.exec(value))) {
matched = true;
const amount = Number(match[1]);
const unit = match[2];
const unitMs = unit === "h" ? 3600000 : unit === "m" ? 60000 : unit === "ms" ? 1 : 1000;
totalMs += amount * unitMs;
}
return matched ? totalMs : null;
}
function resetAtFromDuration(value) {
const ms = parseGroqDurationMs(value);
return ms === null ? null : new Date(Date.now() + ms).toISOString();
}
function buildRateLimitQuota(headers, limitKey, remainingKey, resetKey) {
// headers.get() returns null when absent, and Number(null) is 0 (a finite
// number) — check presence explicitly so a missing header can't masquerade
// as a real "0 remaining" quota.
const limitRaw = headers.get(limitKey);
const remainingRaw = headers.get(remainingKey);
if (limitRaw === null || remainingRaw === null) return null;
const limit = Number(limitRaw);
const remaining = Number(remainingRaw);
if (!Number.isFinite(limit) || !Number.isFinite(remaining)) return null;
return {
used: Math.max(0, limit - remaining),
total: limit,
resetAt: resetAtFromDuration(headers.get(resetKey)),
unlimited: false,
};
}
/**
* @param {string|null|undefined} apiKey
* @param {object|null} proxyOptions
*/
export async function getGroqUsage(apiKey, proxyOptions = null) {
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
return { message: "Groq API key not available. Add a key to view usage." };
}
try {
const response = await proxyAwareFetch(
MODELS_URL,
{
method: "GET",
headers: {
Authorization: `Bearer ${apiKey.trim()}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (response.status === 401 || response.status === 403) {
return { plan: "Groq", message: "Groq authentication failed. Check the API key." };
}
if (!response.ok) {
const errText = await response.text().catch(() => "");
return {
plan: "Groq",
message: `Groq usage API error (${response.status})${errText ? `: ${errText.slice(0, 120)}` : ""}`,
};
}
// The quota data lives in headers, not the body — drain it so the
// connection can be released without needing the payload.
await response.text().catch(() => {});
const requests = buildRateLimitQuota(
response.headers,
"x-ratelimit-limit-requests",
"x-ratelimit-remaining-requests",
"x-ratelimit-reset-requests",
);
const tokens = buildRateLimitQuota(
response.headers,
"x-ratelimit-limit-tokens",
"x-ratelimit-remaining-tokens",
"x-ratelimit-reset-tokens",
);
if (!requests && !tokens) {
// Key is valid (request succeeded) but no rate-limit bucket reported —
// distinguish "not tracked yet" from an auth/error state.
return {
plan: "Groq",
message: "Groq connected. No rate-limit data reported for this key yet.",
quotas: {},
};
}
const quotas = {};
if (requests) quotas["Requests"] = requests;
if (tokens) quotas["Tokens"] = tokens;
return { plan: "Groq", quotas };
} catch (error) {
return { message: `Groq error: ${error.message}` };
}
}

View File

@@ -5,11 +5,8 @@
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U } from "./shared.js";
// GLM quota endpoints (region-aware) — url from registry transport.usage
const GLM_QUOTA_URLS = {
international: U("glm").url,
china: U("glm-cn").url,
};
export { getGlmUsage } from "./glm.js";
// Vercel AI Gateway credits endpoint
// Returns { balance: "95.50", total_used: "4.50" } (USD as decimal strings).
@@ -112,63 +109,7 @@ export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions
}
}
/**
* GLM Coding Plan usage (international + China regions)
*/
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
if (!apiKey) {
return { message: "GLM API key not available." };
}
const region = provider === "glm-cn" ? "china" : "international";
const quotaUrl = GLM_QUOTA_URLS[region];
try {
const response = await proxyAwareFetch(quotaUrl, {
headers: {
Authorization: `Bearer ${apiKey}`,
Accept: "application/json",
},
}, proxyOptions);
if (!response.ok) {
if (response.status === 401) {
return { message: "GLM API key invalid or expired." };
}
return { message: `GLM quota API error (${response.status}).` };
}
const json = await response.json();
const data = json?.data && typeof json.data === "object" ? json.data : {};
const limits = Array.isArray(data.limits) ? data.limits : [];
const quotas = {};
for (const limit of limits) {
if (!limit || limit.type !== "TOKENS_LIMIT") continue;
const usedPercent = Number(limit.percentage) || 0;
const resetMs = Number(limit.nextResetTime) || 0;
const remaining = Math.max(0, 100 - usedPercent);
quotas["session"] = {
used: usedPercent,
total: 100,
remaining,
remainingPercentage: remaining,
resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null,
unlimited: false,
};
}
const levelRaw = typeof data.level === "string" ? data.level : "";
const plan = levelRaw
? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase()
: "Unknown";
return { plan, quotas };
} catch (error) {
return { message: `GLM error: ${error.message}` };
}
}
/**
* Vercel AI Gateway usage — credit balance for the API key

View File

@@ -0,0 +1,107 @@
/**
* OpenCode Go usage — GET https://opencode.ai/zen/go/v1/usage
* Auth: Bearer <apiKey>
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { parseResetTime, toFiniteNumber, U } from "./shared.js";
const USAGE_URL = U("opencode-go").url;
const QUOTA_NAMES = {
rolling: "Rolling",
weekly: "Weekly",
monthly: "Monthly",
};
function parsePercent(value) {
if (typeof value === "number" && Number.isFinite(value)) return value;
if (typeof value === "string" && value.trim()) {
const parsed = Number(value);
if (Number.isFinite(parsed)) return parsed;
}
return null;
}
export async function getOpenCodeGoUsage(apiKey = null, proxyOptions = null) {
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
return {
message: "OpenCode Go API key not available. Add a key to view usage.",
};
}
try {
const response = await proxyAwareFetch(
USAGE_URL,
{
method: "GET",
headers: {
Authorization: `Bearer ${apiKey.trim()}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (response.status === 401) {
return {
plan: "OpenCode Go",
message: "OpenCode Go authentication failed. Check the API key.",
};
}
if (response.status === 403) {
const error = await response.json().catch(() => null);
const subscriptionRequired = error?.error?.type === "EntitlementError";
return {
plan: "OpenCode Go",
message: subscriptionRequired
? "OpenCode Go subscription required for this API key."
: "OpenCode Go access forbidden for this API key.",
};
}
if (!response.ok) {
return {
plan: "OpenCode Go",
message: `OpenCode Go usage API error (${response.status}).`,
};
}
const data = await response.json().catch(() => null);
if (!data?.usage || typeof data.usage !== "object") {
return {
plan: "OpenCode Go",
message: "OpenCode Go usage response did not contain quota data.",
};
}
const quotas = {};
for (const [period, name] of Object.entries(QUOTA_NAMES)) {
const quota = data.usage[period];
if (!quota || typeof quota !== "object") continue;
const percent = parsePercent(quota.percent);
if (percent === null) continue;
const used = Math.max(0, Math.min(100, toFiniteNumber(percent, 0)));
quotas[name] = {
used,
total: 100,
remaining: 100 - used,
remainingPercentage: 100 - used,
resetAt: parseResetTime(quota.resetsAt),
unlimited: false,
};
}
if (Object.keys(quotas).length === 0) {
return {
plan: "OpenCode Go",
message: "OpenCode Go usage response did not contain valid quota data.",
};
}
return { plan: "OpenCode Go", quotas };
} catch (error) {
return { message: `OpenCode Go error: ${error.message}` };
}
}

View File

@@ -0,0 +1,222 @@
/**
* Zed usage — GET https://cloud.zed.dev/client/users/me
* Auth: Authorization: {user_id} {access_token}
*
* Quota rows are derived from plan.usage (edit_predictions, optional model_requests)
* and subscription_period.ended_at for billing-cycle reset.
*/
import { fetchZedAuthenticatedUser } from "../../shared/zedAuth.js";
import { parseResetTime, toFiniteNumber } from "./shared.js";
/** Map plan_v3 ids to dashboard labels (CodexBar-compatible). */
export function formatZedPlanLabel(rawPlan) {
const raw = String(rawPlan || "").trim();
if (!raw) return "Zed";
switch (raw.toLowerCase()) {
case "zed_free":
return "Zed Free";
case "zed_pro":
return "Zed Pro";
case "zed_pro_trial":
return "Zed Pro Trial";
case "zed_student":
return "Zed Student";
case "zed_business":
return "Zed Business";
default:
return raw
.replace(/_/g, " ")
.split(/\s+/)
.map((word) => word.charAt(0).toUpperCase() + word.slice(1).toLowerCase())
.join(" ");
}
}
/**
* Parse Zed UsageLimit JSON: "unlimited", a number, or { limited: N }.
*/
export function parseZedUsageLimit(limit) {
if (limit == null) return { unlimited: false, total: 0 };
if (limit === "unlimited" || limit?.unlimited === true) {
return { unlimited: true, total: 0 };
}
if (typeof limit === "number" && Number.isFinite(limit)) {
return { unlimited: false, total: Math.max(0, limit) };
}
if (typeof limit === "string") {
const trimmed = limit.trim();
if (trimmed === "unlimited") return { unlimited: true, total: 0 };
const parsed = Number(trimmed);
if (Number.isFinite(parsed)) return { unlimited: false, total: Math.max(0, parsed) };
}
const limited = limit.limited ?? limit.Limited;
if (typeof limited === "number" && Number.isFinite(limited)) {
return { unlimited: false, total: Math.max(0, limited) };
}
return { unlimited: false, total: 0 };
}
/** limit `{ limited: 0 }` on Pro/Student means token billing, not a 0-cap request quota. */
export function isZedTokenBillingModelRequestsLimit(limitRaw) {
const info = parseZedUsageLimit(limitRaw);
return !info.unlimited && info.total === 0;
}
function makeZedQuotaRow(name, usedRaw, limitRaw, resetAt = null) {
const used = Math.max(0, toFiniteNumber(usedRaw, 0));
const limitInfo = parseZedUsageLimit(limitRaw);
if (limitInfo.unlimited) {
return {
used,
total: 0,
remainingPercentage: 100,
resetAt: resetAt || null,
unlimited: true,
};
}
const total = limitInfo.total;
if (total <= 0) {
return {
used,
total: 0,
remainingPercentage: 0,
resetAt: resetAt || null,
unlimited: false,
};
}
const clampedUsed = Math.min(used, total);
const remaining = Math.max(0, total - clampedUsed);
return {
used: clampedUsed,
total,
remainingPercentage: (remaining / total) * 100,
resetAt: resetAt || null,
unlimited: false,
};
}
function usageBucketLimit(bucket) {
if (!bucket || typeof bucket !== "object") return null;
if (bucket.limit != null) return bucket.limit;
return bucket;
}
/**
* Map /client/users/me JSON → { plan, quotas, message } for the dashboard.
*/
export function parseZedAuthenticatedUserUsage(userInfo) {
const plan = userInfo?.plan || {};
const planId =
plan.plan_v3 || plan.plan_v2 || plan.plan || userInfo?.plan_v3 || null;
const resetAt =
parseResetTime(plan.subscription_period?.ended_at) ||
parseResetTime(plan.subscriptionPeriod?.endedAt) ||
null;
const quotas = {};
const usage = plan.usage || {};
const editPredictions = usage.edit_predictions || usage.editPredictions;
if (editPredictions) {
quotas["Edit Predictions"] = makeZedQuotaRow(
"Edit Predictions",
editPredictions.used,
editPredictions.limit,
resetAt,
);
}
const modelRequests = usage.model_requests || usage.modelRequests;
if (modelRequests) {
const limitRaw =
modelRequests.limit != null
? modelRequests.limit
: usageBucketLimit(modelRequests)?.limit;
const limitInfo = parseZedUsageLimit(limitRaw);
// Token-billed plans report model_requests.limit=0 — not a request quota.
if (limitInfo.unlimited || limitInfo.total > 0) {
quotas["Hosted Model Requests"] = makeZedQuotaRow(
"Hosted Model Requests",
modelRequests.used,
limitRaw,
resetAt,
);
}
}
const tokenBillingNote =
modelRequests &&
isZedTokenBillingModelRequestsLimit(
modelRequests.limit ?? usageBucketLimit(modelRequests)?.limit,
)
? "Hosted AI models are billed per token (not request count). Edit Predictions are tracked below. Token spend is on dashboard.zed.dev."
: null;
let planLabel = formatZedPlanLabel(planId);
if (plan.trial_started_at || plan.trialStartedAt) {
if (!/trial/i.test(planLabel)) planLabel = `${planLabel} (Trial active)`;
}
let message = tokenBillingNote;
if (plan.has_overdue_invoices || plan.hasOverdueInvoices) {
message = "This Zed account has overdue invoices. Usage may be blocked until billing is resolved.";
}
return {
plan: planLabel,
quotas,
message,
hasOverdueInvoices: !!(plan.has_overdue_invoices || plan.hasOverdueInvoices),
trialStarted: !!(plan.trial_started_at || plan.trialStartedAt),
planId: planId || null,
resetAt,
};
}
/**
* @param {string|null|undefined} accessToken
* @param {object|null|undefined} providerSpecificData
* @param {object|null|undefined} proxyOptions
*/
export async function getZedUsage(
accessToken = null,
providerSpecificData = {},
proxyOptions = null,
) {
const psd = providerSpecificData || {};
const userId = psd.userId;
if (!accessToken || typeof accessToken !== "string" || !accessToken.trim()) {
return { message: "Zed access token not available. Re-connect Zed to view quota." };
}
if (!userId) {
return { message: "Zed credential is missing user id. Re-connect Zed to view quota." };
}
const credentials = {
accessToken: accessToken.trim(),
providerSpecificData: psd,
};
try {
const userInfo = await fetchZedAuthenticatedUser(credentials, { proxyOptions });
return parseZedAuthenticatedUserUsage(userInfo);
} catch (error) {
const status = error?.status;
if (status === 401 || status === 403) {
return {
message: "Zed authentication failed. Sign in again from the dashboard or Zed editor.",
};
}
return { message: `Zed error: ${error.message || "Failed to fetch quota"}` };
}
}

View File

@@ -54,10 +54,13 @@ export const QODER_MODEL_MAP = {
lite: "lite",
// Frontier models
qmodel: "qmodel",
qfmodel: "qfmodel",
qmodel_latest: "qmodel_latest",
qmodel_38max: "qmodel_38max",
dmodel: "dmodel",
dfmodel: "dfmodel",
gm51model: "gm51model",
gmodel: "gmodel",
gfmodel: "gfmodel",
kmodel: "kmodel",
mmodel: "mmodel",
};

View File

@@ -172,8 +172,8 @@ function getSystemId(credentials) {
);
}
async function fetchJson(url, options) {
const res = await proxyAwareFetch(url, options);
async function fetchJson(url, options, proxyOptions = null) {
const res = await proxyAwareFetch(url, options, proxyOptions);
const text = await res.text();
let data = null;
if (text) {
@@ -203,11 +203,15 @@ export async function fetchZedAuthenticatedUser(credentials, options = {}) {
const systemId = getSystemId(credentials);
if (systemId) headers[ZED_HEADERS.systemId] = systemId;
return fetchJson(zedUrl(config, "cloudBaseUrl", "/client/users/me", ZED_CLOUD_BASE_URL), {
method: "GET",
headers,
signal: options.signal ?? undefined,
});
return fetchJson(
zedUrl(config, "cloudBaseUrl", "/client/users/me", ZED_CLOUD_BASE_URL),
{
method: "GET",
headers,
signal: options.signal ?? undefined,
},
options.proxyOptions ?? null,
);
}
function normalizeOrganizationId(value) {

View File

@@ -58,6 +58,15 @@ export function extractThinking(body) {
return { mode: "level", level: e };
}
// OpenAI chat / Responses shape — check effort first (zai sends both thinking object and reasoning.effort)
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
if (typeof effort === "string" && effort) {
const e = effort.toLowerCase();
if (e === "none" || e === "off") return { mode: "none" };
if (e === "auto") return { mode: "auto" };
return { mode: "level", level: e };
}
// Claude shape
const t = body.thinking;
if (t && typeof t === "object") {
@@ -69,15 +78,6 @@ export function extractThinking(body) {
}
}
// OpenAI chat / Responses shape
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
if (typeof effort === "string" && effort) {
const e = effort.toLowerCase();
if (e === "none" || e === "off") return { mode: "none" };
if (e === "auto") return { mode: "auto" };
return { mode: "level", level: e };
}
// Gemini shape (top-level, generationConfig, or request envelope)
const tc = body.thinkingConfig || body.generationConfig?.thinkingConfig || body.request?.generationConfig?.thinkingConfig;
if (tc && typeof tc === "object") {
@@ -105,12 +105,16 @@ export function extractThinking(body) {
// at the call-site where intent is snapshotted before format translation.
export const captureThinking = extractThinking;
// Resolve thinking format: provider override > capability > derive(targetFormat).
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
function resolveFormat(targetFormat, model, provider) {
const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null;
if (providerFmt) return providerFmt;
const caps = getCapabilitiesForModel(provider, model);
if (caps.thinkingFormat) return caps.thinkingFormat;
const isOpenAIWire = targetFormat === "openai" || targetFormat === "openai-responses";
if (caps.thinkingFormat && !(isOpenAIWire && NATIVE_ONLY_FORMATS.has(caps.thinkingFormat))) {
return caps.thinkingFormat;
}
return FORMAT_TO_NATIVE[targetFormat] || "openai";
}
@@ -237,14 +241,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
}
case "claude-adaptive": {
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
// output_config.effort alone does NOT turn thinking on: Anthropic requires
// an explicit thinking:{type:"adaptive"} on Opus 4.6/4.7/4.8 and Sonnet 4.6
// ("thinking is off unless you explicitly set it"), and Anthropic-compatible
// shims (e.g. GitHub Copilot /v1/messages) default thinking off even for
// Sonnet 5. Send both fields — the documented adaptive-thinking shape.
body.thinking = { type: "adaptive" };
// Models that can disable thinking need the explicit adaptive switch.
// Permanently adaptive models such as Fable 5.1 accept effort directly.
if (canDisable) body.thinking = { type: "adaptive" };
else delete body.thinking;
const level = toLevel(eff);
body.output_config = { effort: level === "xhigh" ? "high" : level };
body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level };
break;
}
case "claude-budget": {
@@ -270,6 +272,18 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
// Z.ai ignores thinking.disabled → must use enable_thinking:false to turn off.
if (none && canDisable) { body.enable_thinking = false; delete body.thinking; break; }
body.thinking = { type: "enabled" };
// reasoning_effort is only read by z.ai from GLM-5.2 onward — older GLM ignores it
// (see thinkingEffortSupported in capabilities.js). Skip on unsupported models so we
// don't send a field the API doesn't recognize.
if (caps.thinkingEffortSupported) {
const zaiLvl = toLevel(eff);
// GLM-5.3 only accepts exactly low|high|max (anything else errors); GLM-5.2 accepts
// a wider set but z.ai maps low/medium->high and xhigh->max server-side anyway, so
// this 3-value mapping matches both.
body.reasoning_effort = (zaiLvl === "low" || zaiLvl === "minimal") ? "low"
: (zaiLvl === "high" || zaiLvl === "medium") ? "high"
: "max";
}
break;
}
case "qwen": {

View File

@@ -151,3 +151,17 @@ export function fixMissingToolResponses(body) {
return body;
}
// Default `type: "custom"` on Claude-format tools that arrive without one.
// Anthropic's Claude tool schema requires `type` to be explicitly set; strict gateways
// (e.g., MiniMax Anthropic-compatible endpoint, error 2013) reject legacy payloads that
// omit it with HTTP 400. Tools that already carry a truthy `type` (e.g., `computer_use`,
// `bash`, `web_search_20250305`) are passed through untouched.
//
// Spread order matters: `{ ...tool, type: "custom" }` (spread first, override last)
// ensures that falsy `type` values (null, undefined, "") in the original tool don't
// overwrite the default. `{ type: "custom", ...tool }` would let `type: null` survive.
export function defaultClaudeToolType(tools) {
if (!Array.isArray(tools)) return tools;
return tools.map(tool => tool?.type ? tool : { ...tool, type: "custom" });
}

View File

@@ -12,6 +12,18 @@ import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
const CACHE_CONTROL_5M = { type: "ephemeral" };
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
// Anthropic rejects a tool carrying BOTH defer_loading:true and cache_control
// ("Tools defer_loading cannot use prompt caching", #3567). MCP clients put
// deferred tools at the tail, which is exactly where the cache anchor lands.
// Anchor on the last tool that CAN be cached instead of dropping caching.
export function lastCacheableToolIndex(tools) {
if (!Array.isArray(tools)) return -1;
for (let i = tools.length - 1; i >= 0; i--) {
if (tools[i]?.defer_loading !== true) return i;
}
return -1;
}
// Check if message has valid non-empty content
export function hasValidContent(msg) {
if (typeof msg.content === "string" && msg.content.trim()) return true;
@@ -108,11 +120,24 @@ function buildThinkingPlaceholder(provider) {
return block;
}
// Anthropic validates server_tool_use ids against this pattern and rejects the
// whole request with a 400 when one does not match. A combo that falls back to a
// provider with its own built-in tools (z.ai/glm emits OpenAI-style `call_` ids for
// its analyze_image tool) leaves such blocks in the history, so every later Claude
// turn carries a poisoned id.
const CLAUDE_SERVER_TOOL_USE_ID = /^srvtoolu_[a-zA-Z0-9_]+$/;
function hasForeignServerToolUseId(block) {
return block?.type === CLAUDE_BLOCK.SERVER_TOOL_USE
&& !CLAUDE_SERVER_TOOL_USE_ID.test(String(block.id ?? ""));
}
// Normalize a native Claude passthrough body to match Anthropic Messages API spec.
// Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject:
// 1. thinking.type "adaptive" → unsupported on Haiku
// 2. output_config.effort → unsupported on Haiku
// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed
// 4. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright
export function normalizeClaudePassthrough(body, model = "") {
if (!body || typeof body !== "object") return body;
@@ -164,6 +189,7 @@ export function normalizeClaudePassthrough(body, model = "") {
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// so foreign signatures leak into history and Anthropic rejects them).
const thinkingEnabled = body.thinking?.type === "enabled";
const droppedServerToolUseIds = new Set();
if (Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
@@ -178,6 +204,10 @@ export function normalizeClaudePassthrough(body, model = "") {
}
continue;
}
if (hasForeignServerToolUseId(block)) {
if (block.id != null) droppedServerToolUseIds.add(String(block.id));
continue;
}
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
kept.push(block);
}
@@ -188,6 +218,35 @@ export function normalizeClaudePassthrough(body, model = "") {
}
}
// A dropped server_tool_use leaves its result behind; Anthropic rejects a
// tool_result that references an id no block declares, so both halves must go.
if (droppedServerToolUseIds.size > 0 && Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (!Array.isArray(msg.content)) continue;
const kept = msg.content.filter(block => !(
(block?.type === CLAUDE_BLOCK.TOOL_RESULT || block?.type === CLAUDE_BLOCK.WEB_SEARCH_TOOL_RESULT)
&& droppedServerToolUseIds.has(String(block.tool_use_id ?? ""))
));
if (kept.length !== msg.content.length) {
msg.content = kept;
}
}
}
// 5. Drop empty text blocks and any message left with no content at all.
// Anthropic rejects `messages.N.content` blocks with empty text (400
// "text content blocks must be non-empty"); a message whose blocks were all
// stripped above must be dropped, not padded with an empty placeholder.
if (Array.isArray(body.messages)) {
body.messages = body.messages.filter(msg => {
if (typeof msg.content === "string") return msg.content.trim().length > 0;
if (!Array.isArray(msg.content)) return true;
msg.content = msg.content.filter(block =>
!(block?.type === CLAUDE_BLOCK.TEXT && !String(block.text ?? "").trim()));
return msg.content.length > 0;
});
}
return body;
}
@@ -223,7 +282,7 @@ export function anchorClaudeCache(body) {
}
if (Array.isArray(body.tools)) {
const last = body.tools.length - 1;
const last = lastCacheableToolIndex(body.tools);
body.tools.forEach((tool, i) => {
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
else delete tool.cache_control;
@@ -417,9 +476,10 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
});
}
const lastCacheable = lastCacheableToolIndex(body.tools);
body.tools = body.tools.map((tool, i) => {
const { cache_control, ...rest } = tool;
if (i === body.tools.length - 1) {
if (i === lastCacheable) {
return { ...rest, cache_control: { type: "ephemeral", ttl: "1h" } };
}
return rest;

View File

@@ -14,6 +14,8 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
"uniqueItems", "contains",
// 2020-12 keywords with no Gemini equivalent
"unevaluatedProperties", "unevaluatedItems", "contentSchema",
// Tuple-array keywords; converted to items first, leftovers stripped
"prefixItems", "additionalItems",
// Claude rejects these in VALIDATED mode
"default", "examples",
// JSON Schema meta keywords
@@ -308,6 +310,37 @@ function ensureObjectType(obj) {
for (const v of Object.values(obj)) if (v && typeof v === "object") ensureObjectType(v);
}
// Convert prefixItems (tuple validation) to items — Gemini cannot express tuples,
// and a type:"array" schema without items is rejected with "missing field"
function convertPrefixItems(obj) {
if (!obj || typeof obj !== "object") return;
if (Array.isArray(obj.prefixItems) && obj.prefixItems.length > 0) {
const variants = obj.prefixItems.filter(s => s && s.type !== "null");
if (!obj.items && variants.length === 1) {
obj.items = variants[0];
} else if (!obj.items && variants.length > 1) {
obj.items = { anyOf: variants };
}
delete obj.prefixItems;
}
for (const value of Object.values(obj)) {
if (value && typeof value === "object") {
convertPrefixItems(value);
}
}
}
// Gemini requires items on every type:"array" schema — fill a permissive placeholder
function ensureArrayItems(obj) {
if (!obj || typeof obj !== "object") return;
if (obj.type === "array" && !obj.items) {
obj.items = { type: "string" };
}
for (const v of Object.values(obj)) if (v && typeof v === "object") ensureArrayItems(v);
}
// Clean JSON Schema for Antigravity API compatibility - removes unsupported keywords recursively
export function cleanJSONSchemaForAntigravity(schema) {
if (!schema || typeof schema !== "object") return schema;
@@ -321,11 +354,13 @@ export function cleanJSONSchemaForAntigravity(schema) {
// Phase 2: Flatten complex structures
mergeAllOf(cleaned);
convertPrefixItems(cleaned);
flattenAnyOfOneOf(cleaned);
flattenTypeArrays(cleaned);
// Phase 2.5: Infer missing type=object when properties exist (Gemini requirement)
ensureObjectType(cleaned);
ensureArrayItems(cleaned);
// Phase 3: Remove all unsupported keywords at ALL levels (including inside arrays)
removeUnsupportedKeywords(cleaned, UNSUPPORTED_SCHEMA_CONSTRAINTS);

View File

@@ -23,6 +23,59 @@ export function normalizeResponsesInput(input) {
return null;
}
// Strict Responses upstreams reject overlong call_ids with InputValidationError (#393).
export const MAX_RESPONSES_CALL_ID_LEN = 64;
// Fallback ids share one Date.now() when a batch of items is sanitized in a tight
// loop — a per-process sequence keeps same-millisecond ids unique so
// function_call ↔ function_call_output correlation never collides.
let responsesCallIdSeq = 0;
export function clampResponsesCallId(id) {
if (typeof id !== "string" || !id) return `call_${Date.now()}_${(responsesCallIdSeq += 1)}`;
return id.length > MAX_RESPONSES_CALL_ID_LEN ? id.substring(0, MAX_RESPONSES_CALL_ID_LEN) : id;
}
// Single-stringify: objects → JSON once; valid JSON strings pass through untouched;
// anything else (partial fragments, empty) falls back to "{}" instead of
// double-encoding and tripping upstream InputValidationError.
export function coerceResponsesArguments(value) {
if (value === undefined || value === null || value === "") return "{}";
if (typeof value !== "string") {
try {
return JSON.stringify(value);
} catch {
return "{}";
}
}
try {
JSON.parse(value);
return value;
} catch {
return "{}";
}
}
// function_call_output.output must be a string — never null/object.
export function coerceResponsesOutput(value) {
if (typeof value === "string") return value;
if (value === undefined || value === null) return "";
if (Array.isArray(value)) {
return value.map((c) => {
try {
return c?.text ?? JSON.stringify(c);
} catch {
return String(c);
}
}).join("");
}
try {
return JSON.stringify(value);
} catch {
return String(value);
}
}
/**
* Convert OpenAI Responses API format to standard chat completions format
* Responses API uses: { input: [...], instructions: "..." }

View File

@@ -1,7 +1,7 @@
import { FORMATS } from "./formats.js";
import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js";
import { prepareClaudeRequest } from "./formats/claude.js";
import { cloakClaudeTools } from "../utils/claudeCloaking.js";
import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js";
import { filterToOpenAIFormat } from "./formats/openai.js";
import { normalizeThinkingConfig } from "../services/provider.js";
import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js";
@@ -133,7 +133,7 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
result = prepareClaudeRequest(result, provider, apiKey, connectionId, credentials?.rawHeaders, clientSessionId);
}
// Claude cloaking: rename client tools with _cc suffix (anti-ban)
// Claude cloaking: rename client tools with CLAUDE_TOOL_SUFFIX (anti-ban)
// quirk: only providers flagged cloakToolsOnOAuth, and only with an OAuth token
if (PROVIDERS[provider]?.quirks?.cloakToolsOnOAuth) {
const apiKey = credentials?.accessToken || credentials?.apiKey || null;
@@ -161,9 +161,12 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
// Translate response chunk: target -> openai -> source
export function translateResponse(targetFormat, sourceFormat, chunk, state) {
ensureInitialized();
// If same format, return as-is
// If same format, return as-is — except the tool name may still be cloaked:
// translateRequest() suffixes client tools for OAuth-cloaked Claude providers
// even when no format conversion is needed, so streamed tool_use blocks must
// be decloaked here or the client sees an unknown ("_ide"-suffixed) tool.
if (sourceFormat === targetFormat) {
return [chunk];
return [decloakStreamChunk(chunk, state?.toolNameMap)];
}
let results = [chunk];

View File

@@ -327,7 +327,6 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
};
if (profileArn) payload.profileArn = profileArn;
if (systemPrompt) payload.systemPrompt = systemPrompt;
if (additionalModelRequestFields) {
payload.additionalModelRequestFields = additionalModelRequestFields;
}

View File

@@ -6,12 +6,15 @@
*/
import { register } from "../index.js";
import { FORMATS } from "../formats.js";
import { normalizeResponsesInput } from "../formats/responsesApi.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../formats/responsesApi.js";
import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM } from "../schema/index.js";
// Responses API enforces max 64 chars on call_id (#393)
const MAX_CALL_ID_LEN = 64;
const clampCallId = (id) => (typeof id === "string" && id.length > MAX_CALL_ID_LEN ? id.substring(0, MAX_CALL_ID_LEN) : id);
const MAX_TOOL_NAME_LEN = 128;
/**
* Convert OpenAI Responses API request to OpenAI Chat Completions format
@@ -249,6 +252,23 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
return result;
}
/**
* Extract plain text from a system/developer message for Responses instructions.
* Array content (text parts) is joined; anything else falls back to "" rather
* than leaking "[object Object]" upstream.
*/
function extractInstructionsText(content) {
if (typeof content === "string") return content;
if (Array.isArray(content)) {
return content.map((c) => {
if (typeof c?.text === "string") return c.text;
if (typeof c?.content === "string") return c.content;
return "";
}).filter(Boolean).join("\n");
}
return "";
}
/**
* Ensure object schema always has properties field (required by Codex Responses API)
*/
@@ -300,7 +320,16 @@ function buildReasoningInputItem(msg) {
*/
export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) {
// Body already in Responses API format (e.g. Cursor CLI calling /chat/completions with input[])
if (body.input) return { ...body, model, stream: true };
if (body.input) {
const out = { ...body, model, stream: true };
if (out.max_output_tokens === undefined) {
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
}
delete out.max_tokens;
delete out.max_completion_tokens;
return out;
}
const result = {
model,
@@ -318,7 +347,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
// Use the first instruction-bearing message as instructions.
// OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt.
if (!hasSystemMessage) {
result.instructions = typeof msg.content === "string" ? msg.content : "";
result.instructions = extractInstructionsText(msg.content);
hasSystemMessage = true;
}
continue; // Skip instruction messages in input
@@ -369,26 +398,24 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
// Convert tool calls
if (msg.role === ROLE.ASSISTANT && msg.tool_calls) {
for (const tc of msg.tool_calls) {
// Skip nameless calls — strict Responses upstreams reject them (#444)
const name = typeof tc.function?.name === "string" ? tc.function.name.trim() : "";
if (!name) continue;
result.input.push({
type: RESPONSES_ITEM.FUNCTION_CALL,
call_id: clampCallId(tc.id),
name: tc.function?.name || "_unknown",
arguments: tc.function?.arguments || "{}"
call_id: clampResponsesCallId(tc.id),
name: name.slice(0, MAX_TOOL_NAME_LEN),
arguments: coerceResponsesArguments(tc.function?.arguments)
});
}
}
// Convert tool results - output must be a string for Responses API
if (msg.role === ROLE.TOOL) {
const output = typeof msg.content === "string"
? msg.content
: Array.isArray(msg.content)
? msg.content.map(c => c.text || JSON.stringify(c)).join("")
: JSON.stringify(msg.content);
result.input.push({
type: RESPONSES_ITEM.FUNCTION_CALL_OUTPUT,
call_id: clampCallId(msg.tool_call_id),
output
call_id: clampResponsesCallId(msg.tool_call_id),
output: coerceResponsesOutput(msg.content)
});
}
}
@@ -402,21 +429,30 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
if (body.tools && Array.isArray(body.tools)) {
result.tools = body.tools.map(tool => {
if (tool.type === OPENAI_BLOCK.FUNCTION) {
// Strict upstreams reject nameless/overlong tool declarations
const name = typeof tool.function?.name === "string" ? tool.function.name.trim() : "";
if (!name) return null;
return {
type: OPENAI_BLOCK.FUNCTION,
name: tool.function.name,
name: name.slice(0, MAX_TOOL_NAME_LEN),
description: String(tool.function.description || ""),
parameters: normalizeToolParameters(tool.function.parameters),
strict: tool.function.strict
};
}
return tool;
});
}).filter(Boolean);
}
// Pass through other relevant fields
if (body.temperature !== undefined) result.temperature = body.temperature;
if (body.max_tokens !== undefined) result.max_tokens = body.max_tokens;
if (body.max_output_tokens !== undefined) {
result.max_output_tokens = body.max_output_tokens;
} else if (body.max_completion_tokens !== undefined) {
result.max_output_tokens = body.max_completion_tokens;
} else if (body.max_tokens !== undefined) {
result.max_output_tokens = body.max_tokens;
}
if (body.top_p !== undefined) result.top_p = body.top_p;
if (body.reasoning !== undefined) result.reasoning = body.reasoning;
if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" };

View File

@@ -2,6 +2,7 @@ import { register } from "../index.js";
import { FORMATS } from "../formats.js";
import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js";
import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js";
import { getGeminiThoughtSignatureSync } from "../../services/thoughtSignatureStore.js";
function generateUUID() {
return crypto.randomUUID();
}
@@ -46,7 +47,7 @@ function normalizeGeminiContents(contents) {
}
// Core: Convert OpenAI request to Gemini format (base for all variants)
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) {
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) {
const result = {
model: model,
contents: [],
@@ -133,18 +134,27 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
if (msg.tool_calls && Array.isArray(msg.tool_calls)) {
const toolCallIds = [];
let firstFunctionCallSeen = false;
for (const tc of msg.tool_calls) {
if (tc.type !== OPENAI_BLOCK.FUNCTION) continue;
const args = tryParseJSON(tc.function?.arguments || "{}");
parts.push({
thoughtSignature: signature,
const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null;
// First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig
const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined);
firstFunctionCallSeen = true;
const part = {
functionCall: {
id: tc.id,
name: sanitizeGeminiFunctionName(tc.function.name),
args: args
}
});
};
if (callSig) {
part.thoughtSignature = callSig;
}
parts.push(part);
toolCallIds.push(tc.id);
}
@@ -232,13 +242,13 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
}
// OpenAI -> Gemini (standard API)
export function openaiToGeminiRequest(model, body, stream) {
return openaiToGeminiBase(model, body, stream);
export function openaiToGeminiRequest(model, body, stream, credentials = null) {
return openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_AG_SIGNATURE, credentials?._clientSessionId);
}
// OpenAI -> Gemini CLI (Cloud Code Assist)
export function openaiToGeminiCLIRequest(model, body, stream) {
const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE);
export function openaiToGeminiCLIRequest(model, body, stream, credentials = null) {
const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE, credentials?._clientSessionId);
// Thinking is normalized centrally by applyThinking (thinkingUnified.js) after translation.
// Clean schema for tools
@@ -335,18 +345,26 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
const parts = [];
if (Array.isArray(msg.content)) {
let firstToolUseSeen = false;
for (const block of msg.content) {
if (block.type === CLAUDE_BLOCK.TEXT) {
parts.push({ text: block.text });
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
parts.push({
thoughtSignature: signature,
const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null;
const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined);
firstToolUseSeen = true;
const part = {
functionCall: {
id: block.id,
name: sanitizeGeminiFunctionName(block.name),
args: block.input || {}
}
});
};
if (callSig) {
part.thoughtSignature = callSig;
}
parts.push(part);
} else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) {
let content = block.content;
if (Array.isArray(content)) {

View File

@@ -420,7 +420,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
if (profileArn) {
payload.profileArn = profileArn;
}
if (systemPrompt) payload.systemPrompt = systemPrompt;
if (additionalModelRequestFields) {
payload.additionalModelRequestFields = additionalModelRequestFields;
}

View File

@@ -6,6 +6,7 @@ import { toOpenAIUsage } from "../concerns/usage.js";
import { reasoningDelta } from "../concerns/reasoning.js";
import { encodeDataUri } from "../concerns/image.js";
import { toOpenAIFinish } from "../concerns/finishReason.js";
import { storeGeminiThoughtSignature } from "../../services/thoughtSignatureStore.js";
// Build chunk meta for current gemini state
function chunkMeta(state) {
@@ -13,14 +14,18 @@ function chunkMeta(state) {
}
// Build a tool_call chunk from a gemini functionCall part (shared by sig/non-sig branches)
function emitFunctionCall(functionCall, state) {
function emitFunctionCall(functionCall, state, signature = null) {
const rawName = functionCall.name;
// Restore original tool name from mapping (AG cloaking)
const fcName = state.toolNameMap?.get(rawName) || rawName;
const fcArgs = functionCall.args || {};
const toolCallIndex = state.functionIndex++;
const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`;
if (signature) {
storeGeminiThoughtSignature(callId, signature, state.sessionId);
}
const toolCall = {
id: `${fcName}-${Date.now()}-${toolCallIndex}`,
id: callId,
index: toolCallIndex,
type: OPENAI_BLOCK.FUNCTION,
function: { name: fcName, arguments: JSON.stringify(fcArgs) },
@@ -57,13 +62,21 @@ export function geminiToOpenAIResponse(chunk, state) {
if (content?.parts) {
for (const part of content.parts) {
const hasThoughtSig = part.thoughtSignature || part.thought_signature;
if (hasThoughtSig && typeof hasThoughtSig === "string") {
state.pendingThoughtSignature = hasThoughtSig;
}
const isThought = part.thought === true;
// Handle thought signature (thinking mode)
if (hasThoughtSig) {
const hasTextContent = part.text !== undefined && part.text !== "";
const hasFunctionCall = !!part.functionCall;
// Standalone thoughtSignature part (no text, no functionCall): keep pending for next functionCall
if (!hasTextContent && !hasFunctionCall) {
continue;
}
if (hasTextContent) {
results.push(buildChunk(
chunkMeta(state),
@@ -71,9 +84,10 @@ export function geminiToOpenAIResponse(chunk, state) {
null
));
}
if (hasFunctionCall) {
results.push(emitFunctionCall(part.functionCall, state));
results.push(emitFunctionCall(part.functionCall, state, hasThoughtSig));
state.pendingThoughtSignature = null;
}
continue;
}
@@ -92,7 +106,9 @@ export function geminiToOpenAIResponse(chunk, state) {
// Function call
if (part.functionCall) {
results.push(emitFunctionCall(part.functionCall, state));
const sig = state.pendingThoughtSignature || null;
results.push(emitFunctionCall(part.functionCall, state, sig));
state.pendingThoughtSignature = null;
}
// Inline data (images)

View File

@@ -446,6 +446,13 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
state.created = Math.floor(Date.now() / 1000);
state.toolCallIndex = 0;
state.currentToolCallId = null;
// item_id → chat tool_calls index. Deltas carry item_id; keying on it (not
// stream position) keeps parallel calls separate when upstream emits all
// output_item.added events before any done/delta. Lazily created so callers
// that build their own state object (stream.js) need no changes.
state.respToolChatIndex ??= new Map();
// Indices that already received argument deltas (guards done-with-args).
state.respToolArgsEmitted ??= new Set();
}
// Text content delta
@@ -464,16 +471,29 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
return null;
}
// Function call started (standard function_call or custom_tool_call)
// Function call started (standard function_call or custom_tool_call).
// Index is assigned here (not on done): attributing deltas by stream position
// merges parallel calls into index 0 whenever upstream emits all addeds
// before dones — the client then concatenates N JSON payloads into one
// tool input and fails validation. The server item id is the correlator.
if (eventType === "response.output_item.added" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) {
const item = data.item;
state.currentToolCallId = item.call_id || fallbackToolCallId();
state.respToolChatIndex ??= new Map();
const key = item.id || data.item_id || state.currentToolCallId;
let idx;
if (key && state.respToolChatIndex.has(key)) {
idx = state.respToolChatIndex.get(key); // duplicate added (retry) — reuse
} else {
idx = state.toolCallIndex++;
if (key) state.respToolChatIndex.set(key, idx);
}
return buildChunk(
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
{
tool_calls: [{
index: state.toolCallIndex,
index: idx,
id: state.currentToolCallId,
type: OPENAI_BLOCK.FUNCTION,
function: { name: item.name || "", arguments: "" }
@@ -482,20 +502,39 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
);
}
// Function call arguments delta (standard or custom_tool_call variant)
// Function call arguments delta (standard or custom_tool_call variant).
// Routed by item_id so interleaved parallel fragments stay on their own call.
if (eventType === "response.function_call_arguments.delta" || eventType === "response.custom_tool_call_input.delta") {
const argsDelta = data.delta || "";
if (!argsDelta) return null;
const known = data.item_id ? state.respToolChatIndex?.get(data.item_id) : undefined;
const idx = known ?? Math.max(0, (state.toolCallIndex || 1) - 1);
state.respToolArgsEmitted ??= new Set();
state.respToolArgsEmitted.add(idx);
return buildChunk(
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
{ tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsDelta } }] }
{ tool_calls: [{ index: idx, function: { arguments: argsDelta } }] }
);
}
// Function call done (standard or custom_tool_call variant)
// Function call done (standard or custom_tool_call variant).
// Index was assigned at added-time; nothing to advance. Some upstreams send
// complete arguments only here (no deltas) — emit them once in that case.
if (eventType === "response.output_item.done" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) {
state.toolCallIndex++;
const key = data.item?.id || data.item_id;
const idx = (key && state.respToolChatIndex?.get(key)) ?? Math.max(0, (state.toolCallIndex || 1) - 1);
const fullArgs = data.item?.arguments;
if (typeof fullArgs === "string" && fullArgs) {
state.respToolArgsEmitted ??= new Set();
if (!state.respToolArgsEmitted.has(idx)) {
state.respToolArgsEmitted.add(idx);
return buildChunk(
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
{ tool_calls: [{ index: idx, function: { arguments: fullArgs } }] }
);
}
}
return null;
}

View File

@@ -20,6 +20,8 @@ export const CLAUDE_BLOCK = {
TOOL_RESULT: "tool_result",
THINKING: "thinking",
REDACTED_THINKING: "redacted_thinking",
SERVER_TOOL_USE: "server_tool_use",
WEB_SEARCH_TOOL_RESULT: "web_search_tool_result",
};
// OpenAI Responses API item types.

View File

@@ -1,16 +1,16 @@
import { createHash, randomBytes, randomUUID } from "crypto";
import { CLAUDE_TOOL_SUFFIX, CC_DEFAULT_TOOLS } from "../config/appConstants.js";
import { CLAUDE_CLI_VERSION } from "../providers/shared.js";
const CLAUDE_VERSION = "2.1.92";
const CC_ENTRYPOINT = "sdk-cli";
// Generate billing header matching real Claude Code 2.1.92+ format:
// Generate the billing header expected from current Claude Code clients.
// x-anthropic-billing-header: cc_version=<ver>.<build>; cc_entrypoint=sdk-cli; cch=<hash>;
function generateBillingHeader(payload) {
const content = JSON.stringify(payload);
const cch = createHash("sha256").update(content).digest("hex").slice(0, 5);
const buildHash = randomBytes(2).toString("hex").slice(0, 3);
return `x-anthropic-billing-header: cc_version=${CLAUDE_VERSION}.${buildHash}; cc_entrypoint=${CC_ENTRYPOINT}; cch=${cch};`;
return `x-anthropic-billing-header: cc_version=${CLAUDE_CLI_VERSION}.${buildHash}; cc_entrypoint=${CC_ENTRYPOINT}; cch=${cch};`;
}
// Derive a deterministic UUID-v4-shaped string from a seed (stable per account)
@@ -19,7 +19,7 @@ function deriveUuid(seed) {
return `${h.slice(0, 8)}-${h.slice(8, 12)}-4${h.slice(13, 16)}-${((parseInt(h[16], 16) & 0x3) | 0x8).toString(16)}${h.slice(17, 20)}-${h.slice(20, 32)}`;
}
// Generate fake user ID in Claude Code 2.1.92+ JSON format:
// Generate fake user ID in the current Claude Code JSON format:
// {"device_id":"<64hex>","account_uuid":"<uuid>","session_id":"<uuid>"}
// device_id/account_uuid derive from apiKey (stable per account), session_id per-conversation
function generateFakeUserID(sessionId, apiKey) {
@@ -31,8 +31,8 @@ function generateFakeUserID(sessionId, apiKey) {
/**
* Cloak tools before sending to Claude provider (anti-ban):
* - Rename non-CC client tools with _cc suffix in tools[] and messages[]
* - Skip tools that are already CC default names (they become decoys as-is)
* - Rename client tools with the CLAUDE_TOOL_SUFFIX ("_ide") in tools[] and messages[]
* - Skip tools that carry a `type` (server-side built-ins) — sent as-is
* - Inject CC_DECOY_TOOLS after client tools
* Returns { body, toolNameMap } where toolNameMap maps suffixed → original
* @param {object} body - Claude API request body
@@ -101,6 +101,33 @@ export function decloakToolNames(body, toolNameMap) {
return { ...body, content };
}
/**
* Decloak the tool name inside a single streamed Claude SSE event.
*
* Streaming counterpart of decloakToolNames(). Required for claude→claude
* proxying: translateResponse() returns same-format chunks untouched, so
* without this the client receives the cloaked ("_ide"-suffixed) tool name
* and rejects the call as an unknown tool. In a Claude SSE stream a tool
* name appears exactly once per call — on the content_block_start event of
* a tool_use block; argument deltas carry no name.
*
* Unknown names (e.g. a CC decoy tool the model called anyway) pass through
* unchanged, matching the non-streaming decloak behavior.
*
* @param {object|null} chunk - Parsed SSE event (may be null on stream flush)
* @param {Map|null} toolNameMap - Suffixed → original name map from cloakClaudeTools()
* @returns {object|null} The chunk, with the tool_use name restored when cloaked
*/
export function decloakStreamChunk(chunk, toolNameMap) {
if (!toolNameMap?.size || !chunk || typeof chunk !== "object") return chunk;
if (chunk.type !== "content_block_start") return chunk;
const block = chunk.content_block;
if (block?.type !== "tool_use" || typeof block.name !== "string") return chunk;
const original = toolNameMap.get(block.name);
if (!original) return chunk;
return { ...chunk, content_block: { ...block, name: original } };
}
// CC decoy tools — Claude Code native tool names, marked unavailable
const CC_DECOY_TOOLS = [
{ name: "Task", description: "This tool is currently unavailable.", input_schema: { type: "object", properties: {} } },

View File

@@ -0,0 +1,117 @@
// Claude tool type default self-check.
// Run: node open-sse/utils/claudeToolTypeSelfCheck.mjs
// No framework, no deps. Uses assert. Mirrors toolPairingSelfCheck.mjs style.
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
const results = [];
function run(name, fn) {
try {
fn();
results.push({ name, ok: true });
} catch (err) {
results.push({ name, ok: false, err: err.message });
}
}
const assert = {
equal(a, b, msg) { if (a !== b) throw new Error(`${msg || ""} expected ${b}, got ${a}`); },
ok(v, msg) { if (!v) throw new Error(msg || "expected truthy"); },
};
// 1. Tool without `type` property → defaults to "custom"
run("Tool without type property defaults to custom", () => {
const tools = [{ name: "foo", description: "bar", input_schema: {} }];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "custom", "type defaulted");
assert.equal(out[0].name, "foo", "other fields preserved");
});
// 2. Tool with type:null → defaults to "custom" (the spread-order bug case)
run("Tool with type:null defaults to custom", () => {
const tools = [{ name: "foo", type: null, input_schema: {} }];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "custom", "null type overwritten to custom");
});
// 3. Tool with type:undefined → defaults to "custom"
run("Tool with type:undefined defaults to custom", () => {
const tools = [{ name: "foo", type: undefined, input_schema: {} }];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "custom", "undefined type overwritten to custom");
});
// 4. Tool with type:"" (empty string) → defaults to "custom"
run("Tool with type:empty-string defaults to custom", () => {
const tools = [{ name: "foo", type: "", input_schema: {} }];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "custom", "empty-string type overwritten to custom");
});
// 5. Built-in tool with type:"computer_use" → passed through untouched
run("Built-in tool (computer_use) passed through", () => {
const tools = [{ type: "computer_use", name: "computer", display_width: 1024 }];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "computer_use", "built-in type preserved");
assert.equal(out[0], tools[0], "same reference — not cloned");
});
// 6. Tool already with type:"custom" → passed through untouched
run("Tool already with type:custom passed through", () => {
const tools = [{ type: "custom", name: "foo", input_schema: {} }];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "custom", "existing custom type preserved");
assert.equal(out[0], tools[0], "same reference — not cloned");
});
// 7. Mixed: built-in + function tool → only function tool gets default
run("Mixed: built-in kept, function tool defaulted", () => {
const tools = [
{ type: "computer_use", name: "computer", display_width: 1024 },
{ name: "search", description: "search the web", input_schema: {} },
{ type: "web_search_20250305", name: "web_search" },
];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "computer_use", "built-in preserved");
assert.equal(out[1].type, "custom", "function tool defaulted");
assert.equal(out[2].type, "web_search_20250305", "web_search preserved");
});
// 8. Non-array input → returned unchanged
run("Non-array input returned unchanged", () => {
assert.equal(defaultClaudeToolType(null), null, "null returned as-is");
assert.equal(defaultClaudeToolType(undefined), undefined, "undefined returned as-is");
assert.equal(defaultClaudeToolType("not array"), "not array", "string returned as-is");
});
// 9. Empty array → empty array
run("Empty array returns empty array", () => {
const out = defaultClaudeToolType([]);
assert.equal(Array.isArray(out), true, "returns array");
assert.equal(out.length, 0, "empty array");
});
// 10. Original tools not mutated by reference (new objects for defaulted tools)
run("Original tools not mutated by reference", () => {
const original = { name: "foo", input_schema: {} };
const tools = [original];
defaultClaudeToolType(tools);
assert.equal(original.type, undefined, "original tool not mutated");
assert.ok(!("type" in original), "type property not added to original");
});
// 11. Array with null entry → defaults to { type: "custom" } (optional chaining guard)
// tool?.type returns undefined for null, and { ...null, type: "custom" } === { type: "custom" }
run("Array with null entry defaults to custom", () => {
const tools = [null];
const out = defaultClaudeToolType(tools);
assert.equal(out[0].type, "custom", "null tool gets type custom");
assert.equal(Object.keys(out[0]).length, 1, "no other keys from spread of null");
});
// Summary
const passed = results.filter(r => r.ok).length;
const total = results.length;
for (const r of results) {
console.log(`${r.ok ? "ok" : "FAIL"} - ${r.name}${r.ok ? "" : ` :: ${r.err}`}`);
}
console.log(`\n${passed}/${total} checks passed`);
if (passed !== total) process.exit(1);

View File

@@ -0,0 +1,20 @@
// Claude Code appends a bracketed context marker to the model name when the
// 1M-context beta is toggled on: `claude-opus-5` becomes `claude-opus-5[1m]`.
// The marker is a client-side annotation, not part of any model id: it never
// matches a combo name, an alias or a `provider/model` pair, so a request that
// carries it dies at model resolution with "Invalid model format".
//
// The capability itself travels in the `anthropic-beta: context-1m-2025-08-07`
// header, which is forwarded untouched — stripping the marker is enough to let
// the request route normally and still reach the upstream as a 1M request.
const CONTEXT_MARKER = /\[1m\]$/i;
// Returns { model, contextMarker } — contextMarker is null when there is none.
export function stripModelContextMarker(modelStr) {
if (typeof modelStr !== "string") return { model: modelStr, contextMarker: null };
const trimmed = modelStr.trim();
const match = trimmed.match(CONTEXT_MARKER);
if (!match) return { model: modelStr, contextMarker: null };
return { model: trimmed.slice(0, -match[0].length), contextMarker: match[0].slice(1, -1).toLowerCase() };
}

View File

@@ -94,6 +94,7 @@ const MAX_CONTINUATION_SESSIONS = 5000;
// Client headers/body fields that carry an upstream session id (priority order)
const SESSION_HEADER_KEYS = ["x-session-id", "session-id", "session_id", "x-amp-thread-id"];
const CLAUDE_CODE_SESSION_RE = /_session_([a-f0-9-]+)$/;
const CLAUDE_CODE_SESSION_HEADER = "x-claude-code-session-id";
function sha16(text) {
return crypto.createHash("sha256").update(text).digest("hex").slice(0, 16);
@@ -135,7 +136,10 @@ function extractAntigravitySession(body) {
}
function extractClientSessionId(headers, body, scope = "") {
const claude = extractClaudeCodeSession(body?.metadata?.user_id);
// Claude Code sends the session in a header AND in metadata.user_id; the header
// survives translation to formats that drop metadata (e.g. Responses API).
const claude = extractClaudeCodeSession(body?.metadata?.user_id)
|| headerValue(headers, CLAUDE_CODE_SESSION_HEADER);
if (claude) return `claude:${claude}`;
const antigravity = extractAntigravitySession(body);
if (antigravity) return `antigravity:${antigravity}`;

View File

@@ -49,7 +49,8 @@ export function createSSEStream(options = {}) {
connectionId = null,
body = null,
onStreamComplete = null,
apiKey = null
apiKey = null,
credentials = null
} = options;
let buffer = "";
@@ -59,7 +60,7 @@ export function createSSEStream(options = {}) {
const decoder = new TextDecoder("utf-8", { fatal: false });
const state = mode === STREAM_MODE.TRANSLATE
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model }
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null }
: null;
let totalContentLength = 0;
@@ -75,6 +76,35 @@ export function createSSEStream(options = {}) {
let openAIResponsesTerminalSeen = false;
let openAIResponsesDoneSent = false;
let streamDoneSent = false; // track duplicate [DONE] across transform + flush
let finalized = false;
// Usage/logging tail, callable from transform() as well as flush(): a client that
// closes right after the terminal event cancels the reader, and flush() never runs.
const finalizeStream = () => {
if (finalized) return;
finalized = true;
const isPassthrough = mode === STREAM_MODE.PASSTHROUGH;
let finalUsage = isPassthrough ? usage : state?.usage;
if (!hasValidUsage(finalUsage) && totalContentLength > 0) {
finalUsage = estimateUsage(body, totalContentLength, isPassthrough ? FORMATS.OPENAI : sourceFormat);
if (isPassthrough) usage = finalUsage; else state.usage = finalUsage;
}
if (hasValidUsage(finalUsage)) {
logUsage(isPassthrough ? provider : (state?.provider || targetFormat), finalUsage, model, connectionId, apiKey);
} else {
appendRequestLog({ model, provider, connectionId, tokens: null, status: "200 OK" }).catch(() => { });
}
if (onStreamComplete) {
onStreamComplete({
content: accumulatedContent,
thinking: accumulatedThinking
}, finalUsage, ttftAt);
}
};
return new TransformStream({
transform(chunk, controller) {
@@ -105,6 +135,7 @@ export function createSSEStream(options = {}) {
if (mode === STREAM_MODE.PASSTHROUGH) {
let output;
let injectedUsage = false;
let responsesTerminal = false;
if (trimmed.startsWith("data:") && trimmed.slice(5).trim() !== "[DONE]") {
try {
@@ -168,6 +199,8 @@ export function createSSEStream(options = {}) {
usage = mergeUsage(usage, extracted);
}
responsesTerminal = isOpenAIResponsesTerminalEvent(currentOpenAIResponsesEvent, parsed);
const isFinishChunk = parsed.choices?.[0]?.finish_reason;
if (isFinishChunk && !hasValidUsage(parsed.usage)) {
const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI);
@@ -202,6 +235,8 @@ export function createSSEStream(options = {}) {
reqLogger?.appendConvertedChunk?.(output);
controller.enqueue(sharedEncoder.encode(output));
// Responses clients (codex CLI) close on response.completed instead of [DONE]
if (responsesTerminal) finalizeStream();
continue;
}
@@ -292,6 +327,8 @@ export function createSSEStream(options = {}) {
controller.enqueue(sharedEncoder.encode(output));
currentOpenAIResponsesEvent = null;
sseEmittedCount++;
// Responses clients (codex) close on response.completed instead of [DONE]
if (openAIResponsesTerminalSeen) finalizeStream();
continue;
}
@@ -355,13 +392,6 @@ export function createSSEStream(options = {}) {
controller.enqueue(sharedEncoder.encode(output));
}
if (!hasValidUsage(usage) && totalContentLength > 0) {
usage = estimateUsage(body, totalContentLength, FORMATS.OPENAI);
}
if (hasValidUsage(usage)) {
logUsage(provider, usage, model, connectionId, apiKey);
}
// IMPORTANT: In passthrough mode we still must terminate the SSE stream.
// Some clients (e.g. OpenClaw) expect the OpenAI-style sentinel:
// data: [DONE]\n\n
@@ -374,18 +404,26 @@ export function createSSEStream(options = {}) {
controller.enqueue(sharedEncoder.encode(doneOutput));
}
if (onStreamComplete) {
onStreamComplete({
content: accumulatedContent,
thinking: accumulatedThinking
}, usage, ttftAt);
}
finalizeStream();
return;
}
if (buffer.trim()) {
const parsed = parseSSELine(buffer.trim());
if (parsed && !parsed.done) {
// Same parse as the transform loop: without targetFormat this only
// accepts "data: " lines, so an NDJSON provider (Ollama) lost whatever
// arrived without its closing newline.
const parsed = parseSSELine(buffer.trim(), targetFormat);
// parseSSELine turns the SSE sentinel "data: [DONE]" into { done: true },
// which must not be translated. An Ollama chunk also carries done:true,
// but it is the real final chunk — it holds finish_reason and the token
// counts — so it has to go through.
const isDoneSentinel = parsed?.done && targetFormat !== FORMATS.OLLAMA;
if (parsed && !isDoneSentinel) {
// Same accumulation the transform loop does, so finalizeStream() can
// log a tail chunk's tokens instead of falling back to null.
const extracted = extractUsage(parsed);
if (extracted) state.usage = mergeUsage(state.usage, extracted);
const translated = translateResponse(targetFormat, sourceFormat, parsed, state);
if (translated?._openaiIntermediate) {
@@ -441,26 +479,16 @@ export function createSSEStream(options = {}) {
streamDoneSent = true;
}
if (!hasValidUsage(state?.usage) && totalContentLength > 0) {
state.usage = estimateUsage(body, totalContentLength, sourceFormat);
}
if (hasValidUsage(state?.usage)) {
logUsage(state.provider || targetFormat, state.usage, model, connectionId, apiKey);
}
if (onStreamComplete) {
onStreamComplete({
content: accumulatedContent,
thinking: accumulatedThinking
}, state?.usage, ttftAt);
}
finalizeStream();
} catch (error) {
console.log("Error in flush:", error);
finalizeStream();
}
}
});
}
export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null) {
export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null, credentials = null) {
return createSSEStream({
mode: STREAM_MODE.TRANSLATE,
targetFormat,
@@ -473,7 +501,8 @@ export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, p
connectionId,
body,
onStreamComplete,
apiKey
apiKey,
credentials
});
}

View File

@@ -41,7 +41,8 @@ export function createStreamController({ onDisconnect, onError, log, provider, m
if (disconnected) return;
disconnected = true;
logStream("⚡", `DISCONNECT: ${reason}`);
// Debug-only: Responses API has no [DONE] sentinel, so codex/droid close the
// socket on every completed request. "📊 done" is the authoritative outcome line.
dbg("CTRL", `${provider}/${model} | disconnect=${reason} | dur=${Date.now() - startTime}ms`);
// Delay abort to allow cleanup

View File

@@ -190,7 +190,10 @@ export function canonicalizeUsage(usage) {
prompt = prompt + cached + cacheCreation;
} else {
// OpenAI/Gemini path (or already-canonical input): prompt already includes cached_tokens.
cached = num(usage.cached_tokens);
// Mirror the cacheCreation fallback above: buildUsage() only ever emits the
// nested prompt_tokens_details.cached_tokens shape, so without this the
// cache-read count is silently dropped on every buildUsage()-derived usage.
cached = num(usage.cached_tokens ?? usage.prompt_tokens_details?.cached_tokens);
}
const result = {