This commit is contained in:
2026-07-06 00:01:34 +07:00
parent 2729408ef3
commit e7470e955e
108 changed files with 4131 additions and 529 deletions

View File

@@ -6,6 +6,7 @@ import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
function sanitizeFunctionName(name) {
@@ -177,8 +178,19 @@ export class AntigravityExecutor extends BaseExecutor {
if (p.thoughtSignature && !p.functionCall && !p.text) return false;
return true;
});
if (role !== c.role || parts?.length !== c.parts?.length) {
return { ...c, role, parts };
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
// don't persist thoughtSignature in their history, so backfill the default signature on any
// functionCall part that arrives without one.
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
return {
...c, role,
parts: needsBackfill
? parts.map(p => (p.functionCall && !p.thoughtSignature)
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
: p)
: parts,
};
}
return c;
});

View File

@@ -226,6 +226,7 @@ export class DefaultExecutor extends BaseExecutor {
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
cline: () => this.refreshCline(credentials.refreshToken, proxyOptions),
clinepass: () => this.refreshCline(credentials.refreshToken, proxyOptions),
"kimi-coding": () => this.refreshKimiCoding(credentials.refreshToken, proxyOptions),
kilocode: () => this.refreshKilocode(credentials.refreshToken, proxyOptions)
};
@@ -299,7 +300,11 @@ export class DefaultExecutor extends BaseExecutor {
const data = payload?.data || payload;
const expiresAtIso = data?.expiresAt;
const expiresIn = expiresAtIso ? Math.max(1, Math.floor((new Date(expiresAtIso).getTime() - Date.now()) / 1000)) : undefined;
return { accessToken: data?.accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
let accessToken = data?.accessToken;
if (accessToken && !accessToken.startsWith("workos:")) {
accessToken = `workos:${accessToken}`;
}
return { accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
}
async refreshKimiCoding(refreshToken, proxyOptions = null) {

View File

@@ -5,6 +5,7 @@ import { GithubExecutor } from "./github.js";
import { IFlowExecutor } from "./iflow.js";
import { QoderExecutor } from "./qoder.js";
import { KiroExecutor } from "./kiro.js";
import { KimchiExecutor } from "./kimchi.js";
import { CodexExecutor } from "./codex.js";
import { CursorExecutor } from "./cursor.js";
import { VertexExecutor } from "./vertex.js";
@@ -28,6 +29,7 @@ const executors = {
iflow: new IFlowExecutor(),
qoder: new QoderExecutor(),
kiro: new KiroExecutor(),
kimchi: new KimchiExecutor(),
codex: new CodexExecutor(),
cursor: new CursorExecutor(),
cu: new CursorExecutor(), // Alias for cursor
@@ -66,6 +68,7 @@ export { GithubExecutor } from "./github.js";
export { IFlowExecutor } from "./iflow.js";
export { QoderExecutor } from "./qoder.js";
export { KiroExecutor } from "./kiro.js";
export { KimchiExecutor } from "./kimchi.js";
export { CodexExecutor } from "./codex.js";
export { CursorExecutor } from "./cursor.js";
export { VertexExecutor } from "./vertex.js";

View File

@@ -0,0 +1,123 @@
import { DefaultExecutor } from "./default.js";
import { getCachedKimchiModelMetadata } from "../services/kimchiModels.js";
const TOP_LEVEL_OPENAI_GATEWAY_DROPS = [
"anthropic_version",
"anthropic_beta",
"client_metadata",
"mcp_servers",
"stop_sequences",
"thinking",
"top_k",
];
function systemToText(system) {
if (typeof system === "string") return system;
if (Array.isArray(system)) {
return system
.map((part) => {
if (typeof part === "string") return part;
if (typeof part?.text === "string") return part.text;
return "";
})
.filter(Boolean)
.join("\n");
}
return "";
}
function mergeTopLevelSystem(body) {
if (!body?.system || !Array.isArray(body.messages)) return;
const text = systemToText(body.system).trim();
if (!text) return;
const existing = body.messages.find((msg) => msg?.role === "system");
if (!existing) {
body.messages.unshift({ role: "system", content: text });
return;
}
if (typeof existing.content === "string") {
existing.content = `${text}\n\n${existing.content}`;
} else if (Array.isArray(existing.content)) {
existing.content.unshift({ type: "text", text });
}
}
function stripMessageArtifacts(body) {
if (!Array.isArray(body?.messages)) return;
for (const msg of body.messages) {
if (!msg || typeof msg !== "object") continue;
delete msg.cache_control;
if (!Array.isArray(msg.content)) continue;
msg.content = msg.content.map((part) => {
if (!part || typeof part !== "object") return part;
const { cache_control, signature, ...clean } = part;
return clean;
});
}
}
function stripToolArtifacts(body) {
if (!Array.isArray(body?.tools)) return;
body.tools = body.tools.map((tool) => {
if (!tool || typeof tool !== "object") return tool;
const { cache_control, ...clean } = tool;
return clean;
});
}
// Strip `reasoning_content` echoed by clients on assistant messages — but
// only when it's a real thinking block. `DefaultExecutor.transformRequest`
// runs `injectReasoningContent` first and may inject a 1-char placeholder
// (" ") for upstream validation; the placeholder is small (no token cost
// worth stripping) and stripping it would re-trigger upstream to complain
// about missing reasoning on the next turn. Threshold matches the
// placeholder length with a safety margin.
const REASONING_PLACEHOLDER_MAX_LEN = 8;
export function stripReasoningContent(body) {
if (!Array.isArray(body?.messages)) return;
for (const msg of body.messages) {
if (msg && msg.role === "assistant" && typeof msg.reasoning_content === "string"
&& msg.reasoning_content.length > REASONING_PLACEHOLDER_MAX_LEN) {
delete msg.reasoning_content;
}
}
}
function isAnthropicBackedKimchiModel(model) {
const meta = getCachedKimchiModelMetadata(model);
if (meta?.provider === "anthropic" || meta?.upstreamProvider === "anthropic") return true;
return /(^|[-_/])(?:claude|anthropic)(?:[-_/]|$)/i.test(String(model || ""));
}
export class KimchiExecutor extends DefaultExecutor {
constructor() {
super("kimchi");
}
transformRequest(model, body, stream, credentials) {
const transformed = super.transformRequest(model, body, stream, credentials);
if (!transformed || typeof transformed !== "object") return transformed;
mergeTopLevelSystem(transformed);
for (const key of TOP_LEVEL_OPENAI_GATEWAY_DROPS) {
if (transformed[key] !== undefined) delete transformed[key];
}
delete transformed.system;
if (isAnthropicBackedKimchiModel(model)) {
delete transformed.reasoning_effort;
delete transformed.reasoning;
delete transformed.thinking;
}
stripMessageArtifacts(transformed);
stripToolArtifacts(transformed);
stripReasoningContent(transformed);
return transformed;
}
}
export default KimchiExecutor;

View File

@@ -64,9 +64,22 @@ export class KiroExecutor extends BaseExecutor {
getOrderedBaseUrls(credentials) {
const baseUrls = this.getBaseUrls();
const authMethod = credentials?.providerSpecificData?.authMethod;
const isCodeWhispererSurface = authMethod === "api_key" || authMethod === "external_idp";
// IAM Identity Center (idc) tokens are AWS SSO access tokens — the same
// family as external_idp/api_key. The kiro.dev gateway rejects them with
// 403 "bearer token invalid", so they must hit the CodeWhisperer
// *.amazonaws.com surface, and in the region the token was minted in
// (the baseUrls are hardcoded us-east-1).
const isCodeWhispererSurface =
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
if (!isCodeWhispererSurface) return baseUrls;
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com"));
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
const regionalize = (u) =>
region && region !== "us-east-1" && u.includes("amazonaws.com")
? u.replace(/([a-z]+)\.[a-z0-9-]+\.amazonaws\.com/, `$1.${region}.amazonaws.com`)
: u;
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
}
@@ -122,7 +135,8 @@ export class KiroExecutor extends BaseExecutor {
hasReasoningContent: false,
reasoningChunkCount: 0,
toolCallIndex: 0,
seenToolIds: new Map()
seenToolIds: new Map(),
inThinking: false
};
const transformStream = new TransformStream({
@@ -159,7 +173,36 @@ export class KiroExecutor extends BaseExecutor {
// Handle assistantResponseEvent
if (eventType === "assistantResponseEvent" && event.payload?.content) {
const content = event.payload.content;
let content = event.payload.content;
// Kiro Claude models can leak <thinking> blocks into the content stream.
// We strip these literal tags to prevent duplication, as the reasoning
// is already routed correctly via reasoningContentEvent.
if (state.inThinking) {
if (content.includes("</thinking>")) {
state.inThinking = false;
const after = content.split("</thinking>").slice(1).join("</thinking>");
content = after.startsWith("\n") ? after.substring(1) : after;
} else {
content = ""; // Drop entirely while inside thinking block
}
} else if (content.includes("<thinking>")) {
state.inThinking = true;
if (content.includes("</thinking>")) {
state.inThinking = false;
const before = content.split("<thinking>")[0];
const after = content.split("</thinking>").slice(1).join("</thinking>");
content = before + (after.startsWith("\n") ? after.substring(1) : after);
} else {
content = content.split("<thinking>")[0];
}
}
if (!content && state.hasReasoningContent) {
// If we stripped everything, skip emitting an empty content chunk
continue;
}
state.totalContentLength += content.length;
const chunk = {
@@ -348,6 +391,11 @@ export class KiroExecutor extends BaseExecutor {
if (metrics && typeof metrics === 'object') {
const inputTokens = metrics.inputTokens || 0;
const outputTokens = metrics.outputTokens || 0;
// ponytail: Amazon Q upstream does not expose cache fields today,
// but pick up cache_read_input_tokens / cache_creation_input_tokens
// if the event shape grows them so cost tracking stays accurate.
const cachedTokens = metrics.cacheReadInputTokens || metrics.cache_read_input_tokens || 0;
const cacheCreationInputTokens = metrics.cacheCreationInputTokens || metrics.cache_creation_input_tokens || 0;
if (inputTokens > 0 || outputTokens > 0) {
state.usage = {
@@ -355,6 +403,12 @@ export class KiroExecutor extends BaseExecutor {
completion_tokens: outputTokens,
total_tokens: inputTokens + outputTokens
};
// Kiro is Claude-backed: inputTokens EXCLUDES cache (Claude convention),
// not inclusive like OpenAI's cached_tokens. Emit cache_read_input_tokens
// (not cached_tokens) so canonicalizeUsage takes the Claude fold path and
// correctly adds cache back into prompt_tokens instead of undercharging.
if (cachedTokens > 0) state.usage.cache_read_input_tokens = cachedTokens;
if (cacheCreationInputTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationInputTokens;
}
}
}

View File

@@ -326,8 +326,8 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
}
// Streaming response
const { onStreamComplete } = buildOnStreamComplete({ ...sharedCtx });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, streamController, onStreamComplete });
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId });
}
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {

View File

@@ -1,5 +1,6 @@
import { FORMATS } from "../../translator/formats.js";
import { needsTranslation } from "../../translator/index.js";
import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
import { createErrorResult } from "../../utils/error.js";
@@ -9,11 +10,65 @@ import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, sav
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
import { decloakToolNames } from "../../utils/claudeCloaking.js";
function parseToolArguments(value) {
if (!value) return {};
if (typeof value === "object") return value;
try {
return JSON.parse(value);
} catch {
return {};
}
}
function openAICompletionToClaudeMessage(responseBody) {
if (!responseBody?.choices?.[0]) return responseBody;
const choice = responseBody.choices[0];
const message = choice.message || {};
const content = [];
const reasoning = message.reasoning_content || message.provider_specific_fields?.reasoning_content || "";
if (reasoning) {
content.push({ type: "thinking", thinking: reasoning });
}
if (typeof message.content === "string" && message.content.length > 0) {
content.push({ type: "text", text: message.content });
}
for (const toolCall of message.tool_calls || []) {
const fn = toolCall.function || {};
content.push({
type: "tool_use",
id: toolCall.id || `toolu_${Date.now()}_${content.length}`,
name: fn.name || toolCall.name || "",
input: parseToolArguments(fn.arguments || toolCall.arguments),
});
}
if (content.length === 0) content.push({ type: "text", text: "" });
const usage = responseBody.usage || {};
return {
id: String(responseBody.id || `msg_${Date.now()}`).replace(/^chatcmpl-/, ""),
type: "message",
role: "assistant",
model: responseBody.model || "unknown",
content,
stop_reason: fromOpenAIFinish(choice.finish_reason, FORMATS.CLAUDE),
stop_sequence: null,
usage: {
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
},
};
}
/**
* Translate non-streaming response body from provider format → OpenAI format.
*/
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
if (targetFormat === sourceFormat || targetFormat === FORMATS.OPENAI) return responseBody;
if (targetFormat === sourceFormat) return responseBody;
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
return openAICompletionToClaudeMessage(responseBody);
}
if (targetFormat === FORMATS.OPENAI) return responseBody;
// Gemini / Antigravity
if (targetFormat === FORMATS.GEMINI || targetFormat === FORMATS.ANTIGRAVITY || targetFormat === FORMATS.GEMINI_CLI || targetFormat === FORMATS.VERTEX) {
@@ -185,6 +240,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
: responseBody;
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
if (translatedResponse?.choices?.[0]) {
@@ -197,13 +253,17 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
}
// Ensure OpenAI-required fields
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
if (!isClaudeMessageResponse) {
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
}
// Strip Azure-specific fields
delete translatedResponse.prompt_filter_results;
if (translatedResponse?.choices) {
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
if (!isClaudeMessageResponse) {
delete translatedResponse.prompt_filter_results;
if (translatedResponse?.choices) {
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
}
}
if (translatedResponse?.usage) {
@@ -213,7 +273,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
// Strip reasoning_content only when content is non-empty.
// When content is empty (e.g. thinking models that used all tokens for reasoning),
// reasoning_content is the only useful output and must be preserved.
if (translatedResponse?.choices) {
if (!isClaudeMessageResponse && translatedResponse?.choices) {
for (const choice of translatedResponse.choices) {
if (choice?.message?.reasoning_content && choice.message.content) {
delete choice.message.reasoning_content;

View File

@@ -1,5 +1,6 @@
import { saveRequestUsage, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
import { COLORS } from "../../utils/stream.js";
import { canonicalizeUsage } from "../../utils/usageTracking.js";
const OPTIONAL_PARAMS = [
"temperature", "top_p", "top_k",
@@ -48,7 +49,8 @@ export function extractUsageFromResponse(responseBody) {
return {
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
};
}
@@ -96,8 +98,14 @@ export function saveUsageStats({ provider, model, tokens, connectionId, apiKey,
msg += `${COLORS.reset}`;
console.log(msg);
<<<<<<< HEAD
// Normalize to OpenAI token shape for storage (include all token types)
const normalized = {
=======
// Canonicalize to one storage convention (prompt_tokens cache-inclusive) so
// cached/cache-creation tokens survive to cost calc + stats. See canonicalizeUsage.
const normalized = canonicalizeUsage(tokens) || {
>>>>>>> 7f436e2792be4fa5a4d1c4d6b8e9bc85eaaa6a3d
prompt_tokens: tokens.prompt_tokens ?? tokens.input_tokens ?? 0,
completion_tokens: tokens.completion_tokens ?? tokens.output_tokens ?? 0,
cache_read_input_tokens: cacheRead,

View File

@@ -43,7 +43,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
/**
* Handle streaming response — pipe provider SSE through transform stream to client.
*/
export function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete }) {
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId }) {
if (onRequestSuccess) {
Promise.resolve()
.then(onRequestSuccess)
@@ -52,12 +52,30 @@ export function handleStreamingResponse({ providerResponse, provider, model, sou
});
}
// Warn when upstream returns unexpected Content-Type for a streaming response.
// This often means the provider returned an HTML error page or plain-text error
// that the SSE transform stream would forward as garbage to the client.
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
// page), piping it through the SSE transform stream causes Next.js
// "failed to pipe response" and crashes the chat router. Read the body,
// pull a short human-readable message from the <title>, sanitize it, and
// return a clean JSON error instead. The message is stripped of HTML tags
// and clamped so untrusted upstream text never reaches the client verbatim
// (the UI may render error.message as HTML).
const upstreamContentType = (providerResponse.headers.get('content-type') || '').toLowerCase();
if (upstreamContentType && !upstreamContentType.includes('text/event-stream') && !upstreamContentType.includes('application/json')) {
console.warn('[STREAM] ' + provider + ' | ' + model + ' | unexpected Content-Type: ' + upstreamContentType);
const bodyText = await providerResponse.text().catch(() => '');
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
const sanitizedTitle = (titleMatch?.[1] || '').replace(/<[^>]*>/g, '').replace(/[\r\n]+/g, ' ').trim().slice(0, 160);
const shortMsg = sanitizedTitle
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
const status = providerResponse.status || 502;
console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
return {
success: false,
response: new Response(JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }), {
status,
headers: { 'Content-Type': 'application/json', 'Access-Control-Allow-Origin': '*' },
}),
};
}
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey });
@@ -68,7 +86,6 @@ export function handleStreamingResponse({ providerResponse, provider, model, sou
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
saveRequestDetail(buildRequestDetail({
provider, model, connectionId,
latency: { ttft: 0, total: Date.now() - requestStartTime },

View File

@@ -71,7 +71,7 @@ export function capabilitiesFromServiceKind(kind) {
* otherwise mis-match. Only declare deltas vs DEFAULT.
*/
export const MODEL_CAPABILITIES = {
// Claude 4.6/4.7/4.8 have 1M context + adaptive thinking (override generic claude pattern)
// Claude 4.6/4.7/4.8 and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
"claude-opus-4.6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-4.7": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-4-7": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
@@ -82,6 +82,10 @@ export const MODEL_CAPABILITIES = {
"claude-opus-4-8-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-4.6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-4-6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-5-thinking-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
// Gemini image-gen / OpenAI image / xai image variants
"gpt-image-1": { imageOutput: true, tools: false },
@@ -98,6 +102,15 @@ export const MODEL_CAPABILITIES = {
* Provider-specific capability overrides. Keyed by provider alias/id.
*/
export const PROVIDER_CAPABILITIES = {
// NVIDIA NIM is OpenAI-compatible → rejects MiniMax/GLM native `thinking` field.
// Force openai reasoning_effort format for its reasoning models. #issue
"nvidia": {
"minimaxai/minimax-m2.7": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 131072 },
"minimaxai/minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 131072 },
"z-ai/glm-5.2": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 },
"deepseek-ai/deepseek-v4-pro": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
},
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
@@ -177,12 +190,16 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*grok-3*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 131072 } },
{ pattern: "*grok*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
// ── Qwen (enable_thinking + thinking_budget; QwQ = thinking-only) ─
// ── Qwen (3.5+ = native vision/video; coder & max = text-only; QwQ = thinking-only) ─
{ pattern: "*qwen*vl*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
{ pattern: "*qwen*max*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen*omni*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144, maxOutput: 65536 } },
{ pattern: "*qwen*coder*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 } },
{ pattern: "*qwen*max*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen3.5*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen3.6*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen3.7*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen*plus*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
{ pattern: "*qwen*235b*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
{ pattern: "*qwen*coder*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 } },
{ pattern: "*qwq*", caps: { reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 131072 } },
{ pattern: "*qwen*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },

View File

@@ -279,7 +279,10 @@ export function calculateCostFromTokens(tokens, pricing) {
const inputTokens = tokens.prompt_tokens || tokens.input_tokens || 0;
const cachedTokens = tokens.cached_tokens || tokens.cache_read_input_tokens || 0;
const nonCachedInput = Math.max(0, inputTokens - cachedTokens);
const cacheCreationTokens = tokens.cache_creation_input_tokens || 0;
// prompt_tokens is cache-inclusive (see canonicalizeUsage): cached + cache_creation
// are subsets, so subtract both to avoid charging them at the full input rate.
const nonCachedInput = Math.max(0, inputTokens - cachedTokens - cacheCreationTokens);
cost += nonCachedInput * (pricing.input / 1000000);
@@ -295,7 +298,6 @@ export function calculateCostFromTokens(tokens, pricing) {
cost += reasoningTokens * ((pricing.reasoning || pricing.output) / 1000000);
}
const cacheCreationTokens = tokens.cache_creation_input_tokens || 0;
if (cacheCreationTokens > 0) {
cost += cacheCreationTokens * ((pricing.cache_creation || pricing.input) / 1000000);
}

View File

@@ -16,6 +16,7 @@ export default {
transport: {
baseUrl: "https://coding-intl.dashscope.aliyuncs.com/v1/chat/completions",
headers: {},
quirks: { preserveCacheControl: true },
},
models: [
{ id: "qwen3.5-plus", name: "Qwen3.5 Plus" },

View File

@@ -16,6 +16,7 @@ export default {
transport: {
baseUrl: "https://coding.dashscope.aliyuncs.com/v1/chat/completions",
headers: {},
quirks: { preserveCacheControl: true },
},
models: [
{ id: "qwen3.5-plus", name: "Qwen3.5 Plus" },

View File

@@ -0,0 +1,57 @@
export default {
id: "clinepass",
priority: 85,
alias: "clinepass",
uiAlias: "clinepass",
display: {
name: "ClinePass",
icon: "vpn_key",
color: "#5B9BD5",
textIcon: "CP",
website: "https://cline.bot",
notice: {
signupUrl: "https://app.cline.bot",
},
},
category: "oauth",
authModes: ["oauth", "apikey"],
hasOAuth: true,
transport: {
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
headers: {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
auth: {
combined: true,
header: "Authorization",
scheme: "bearer",
hooks: [
"clineHeaders",
],
},
},
models: [
{ id: "cline-pass/glm-5.2", name: "GLM-5.2 (ClinePass)" },
{ id: "cline-pass/kimi-k2.7-code", name: "Kimi K2.7 Code (ClinePass)" },
{ id: "cline-pass/kimi-k2.6", name: "Kimi K2.6 (ClinePass)" },
{ id: "cline-pass/deepseek-v4-pro", name: "DeepSeek V4 Pro (ClinePass)" },
{ id: "cline-pass/deepseek-v4-flash", name: "DeepSeek V4 Flash (ClinePass)" },
{ id: "cline-pass/mimo-v2.5", name: "MiMo-V2.5 (ClinePass)" },
{ id: "cline-pass/mimo-v2.5-pro", name: "MiMo-V2.5-Pro (ClinePass)" },
{ id: "cline-pass/minimax-m3", name: "MiniMax M3 (ClinePass)" },
{ id: "cline-pass/qwen3.7-max", name: "Qwen3.7 Max (ClinePass)" },
{ id: "cline-pass/qwen3.7-plus", name: "Qwen3.7 Plus (ClinePass)" },
],
oauth: {
appBaseUrl: "https://app.cline.bot",
apiBaseUrl: "https://api.cline.bot",
authorizeUrl: "https://api.cline.bot/api/v1/auth/authorize",
tokenUrl: "https://api.cline.bot/api/v1/auth/token",
refreshUrl: "https://api.cline.bot/api/v1/auth/refresh",
},
thinkingConfig: {
options: ["auto", "on", "off"],
defaultMode: "auto",
},
};

View File

@@ -40,6 +40,7 @@ export default {
},
usage: {
url: "https://chatgpt.com/backend-api/wham/usage",
resetCreditsUrl: "https://chatgpt.com/backend-api/wham/rate-limit-reset-credits",
resetCreditsConsumeUrl: "https://chatgpt.com/backend-api/wham/rate-limit-reset-credits/consume",
},
},

View File

@@ -15,85 +15,87 @@ import p12 from "./cerebras.js";
import p13 from "./chutes.js";
import p14 from "./claude.js";
import p15 from "./cline.js";
import p16 from "./cloudflare-ai.js";
import p17 from "./codebuddy-cn.js";
import p18 from "./codex.js";
import p19 from "./cohere.js";
import p20 from "./comfyui.js";
import p21 from "./commandcode.js";
import p22 from "./coqui.js";
import p23 from "./cursor.js";
import p24 from "./deepgram.js";
import p25 from "./deepseek.js";
import p26 from "./edge-tts.js";
import p27 from "./elevenlabs.js";
import p28 from "./exa.js";
import p29 from "./fal-ai.js";
import p30 from "./firecrawl.js";
import p31 from "./fireworks.js";
import p32 from "./gemini-cli.js";
import p33 from "./gemini.js";
import p34 from "./github.js";
import p35 from "./gitlab.js";
import p36 from "./glm-cn.js";
import p37 from "./glm.js";
import p38 from "./google-pse.js";
import p39 from "./google-tts.js";
import p40 from "./grok-web.js";
import p41 from "./groq.js";
import p42 from "./huggingface.js";
import p43 from "./hyperbolic.js";
import p44 from "./iflow.js";
import p45 from "./inworld.js";
import p46 from "./jina-ai.js";
import p47 from "./jina-reader.js";
import p48 from "./kilocode.js";
import p49 from "./kimi-coding.js";
import p50 from "./kimi.js";
import p51 from "./kiro.js";
import p52 from "./linkup.js";
import p53 from "./local-device.js";
import p54 from "./mimo-free.js";
import p55 from "./minimax-cn.js";
import p56 from "./minimax.js";
import p57 from "./mistral.js";
import p58 from "./mmf.js";
import p59 from "./nanobanana.js";
import p60 from "./nebius.js";
import p61 from "./nvidia.js";
import p62 from "./ollama-local.js";
import p63 from "./ollama.js";
import p64 from "./openai.js";
import p65 from "./opencode-go.js";
import p66 from "./opencode.js";
import p67 from "./openrouter.js";
import p68 from "./perplexity-web.js";
import p69 from "./perplexity.js";
import p70 from "./playht.js";
import p71 from "./qoder.js";
import p72 from "./qwen.js";
import p73 from "./recraft.js";
import p74 from "./runwayml.js";
import p75 from "./sdwebui.js";
import p76 from "./searchapi.js";
import p77 from "./searxng.js";
import p78 from "./serper.js";
import p79 from "./siliconflow.js";
import p80 from "./stability-ai.js";
import p81 from "./tavily.js";
import p82 from "./together.js";
import p83 from "./topaz.js";
import p84 from "./tortoise.js";
import p85 from "./venice.js";
import p86 from "./vercel-ai-gateway.js";
import p87 from "./vertex-partner.js";
import p88 from "./vertex.js";
import p89 from "./volcengine-ark.js";
import p90 from "./voyage-ai.js";
import p91 from "./xai.js";
import p92 from "./xiaomi-mimo.js";
import p93 from "./xiaomi-tokenplan.js";
import p94 from "./youcom.js";
import p16 from "./clinepass.js";
import p17 from "./cloudflare-ai.js";
import p18 from "./codebuddy-cn.js";
import p19 from "./codex.js";
import p20 from "./cohere.js";
import p21 from "./comfyui.js";
import p22 from "./commandcode.js";
import p23 from "./coqui.js";
import p24 from "./cursor.js";
import p25 from "./deepgram.js";
import p26 from "./deepseek.js";
import p27 from "./edge-tts.js";
import p28 from "./elevenlabs.js";
import p29 from "./exa.js";
import p30 from "./fal-ai.js";
import p31 from "./firecrawl.js";
import p32 from "./fireworks.js";
import p33 from "./gemini-cli.js";
import p34 from "./gemini.js";
import p35 from "./github.js";
import p36 from "./gitlab.js";
import p37 from "./glm-cn.js";
import p38 from "./glm.js";
import p39 from "./google-pse.js";
import p40 from "./google-tts.js";
import p41 from "./grok-web.js";
import p42 from "./groq.js";
import p43 from "./huggingface.js";
import p44 from "./hyperbolic.js";
import p45 from "./iflow.js";
import p46 from "./inworld.js";
import p47 from "./jina-ai.js";
import p48 from "./jina-reader.js";
import p49 from "./kilocode.js";
import p50 from "./kimchi.js";
import p51 from "./kimi-coding.js";
import p52 from "./kimi.js";
import p53 from "./kiro.js";
import p54 from "./linkup.js";
import p55 from "./local-device.js";
import p56 from "./mimo-free.js";
import p57 from "./minimax-cn.js";
import p58 from "./minimax.js";
import p59 from "./mistral.js";
import p60 from "./mmf.js";
import p61 from "./nanobanana.js";
import p62 from "./nebius.js";
import p63 from "./nvidia.js";
import p64 from "./ollama-local.js";
import p65 from "./ollama.js";
import p66 from "./openai.js";
import p67 from "./opencode-go.js";
import p68 from "./opencode.js";
import p69 from "./openrouter.js";
import p70 from "./perplexity-web.js";
import p71 from "./perplexity.js";
import p72 from "./playht.js";
import p73 from "./qoder.js";
import p74 from "./qwen.js";
import p75 from "./recraft.js";
import p76 from "./runwayml.js";
import p77 from "./sdwebui.js";
import p78 from "./searchapi.js";
import p79 from "./searxng.js";
import p80 from "./serper.js";
import p81 from "./siliconflow.js";
import p82 from "./stability-ai.js";
import p83 from "./tavily.js";
import p84 from "./together.js";
import p85 from "./topaz.js";
import p86 from "./tortoise.js";
import p87 from "./venice.js";
import p88 from "./vercel-ai-gateway.js";
import p89 from "./vertex-partner.js";
import p90 from "./vertex.js";
import p91 from "./volcengine-ark.js";
import p92 from "./voyage-ai.js";
import p93 from "./xai.js";
import p94 from "./xiaomi-mimo.js";
import p95 from "./xiaomi-tokenplan.js";
import p96 from "./youcom.js";
export default [
p0,
@@ -190,5 +192,7 @@ export default [
p91,
p92,
p93,
p94
p94,
p95,
p96
];

View File

@@ -36,6 +36,13 @@ export default {
{ id: "deepseek/deepseek-chat", name: "DeepSeek Chat" },
{ id: "deepseek/deepseek-reasoner", name: "DeepSeek Reasoner" },
],
// Kilo Code proxies the OpenRouter catalog (334 models at time of writing),
// so the hardcoded list above is only a fallback. Surfacing the full catalog
// requires a fetcher + passthroughModels, matching how openrouter.js is set up.
// Without these, only the 8 hardcoded models appear in the combo model picker,
// hiding dynamic models like cohere/north-mini-code:free and poolside/laguna-m.1:free.
modelsFetcher: { url: "https://api.kilo.ai/api/gateway/models", type: "openrouter-free" },
passthroughModels: true,
oauth: {
apiBaseUrl: "https://api.kilo.ai",
initiateUrl: "https://api.kilo.ai/api/device-auth/codes",

View File

@@ -0,0 +1,49 @@
export default {
id: "kimchi",
priority: 95,
alias: "kimchi",
uiAlias: "kimchi",
display: {
name: "Kimchi",
icon: "restaurant",
color: "#FF521D",
textIcon: "KC",
website: "https://kimchi.dev",
notice: {
signupUrl: "https://app.kimchi.dev",
},
},
category: "oauth",
authModes: ["oauth"],
hasOAuth: true,
transport: {
baseUrl: "https://llm.kimchi.dev/openai/v1/chat/completions",
format: "openai",
headers: {
"User-Agent": "kimchi/0.1.50",
},
auth: {
combined: true,
header: "Authorization",
scheme: "bearer",
},
},
models: [
{ id: "minimax-m3", name: "MiniMax-M3" },
{ id: "kimi-k2.7", name: "Kimi-K2.7" },
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
{ id: "nemotron-3-ultra-fp4", name: "Nemotron 3 Ultra FP4" },
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
{ id: "claude-opus-4-6", name: "Claude Opus 4.6" },
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6" },
],
serviceKinds: ["llm", "imageToText"],
oauth: {
webAppUrl: "https://app.kimchi.dev",
validationUrl: "https://api.cast.ai/v1/llm/openai/supported-providers",
userInfoUrl: "https://app.kimchi.dev/api/v1/me",
modelsUrl: "https://llm.kimchi.dev/v1/models/metadata?include_in_cli=true",
},
passthroughModels: true,
};

View File

@@ -42,16 +42,20 @@ export default {
},
},
models: [
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
{ id: "claude-haiku-4.5", name: "Claude Haiku 4.5" },
{ id: "deepseek-3.2", name: "DeepSeek 3.2", strip: ["image","audio"] },
{ id: "qwen3-coder-next", name: "Qwen3 Coder Next", strip: ["image","audio"] },
{ id: "glm-5", name: "GLM 5" },
{ id: "MiniMax-M2.5", name: "MiniMax M2.5" },
{ id: "claude-sonnet-5-thinking", name: "Claude Sonnet 5 (Thinking)" },
{ id: "claude-sonnet-4.5-thinking", name: "Claude Sonnet 4.5 (Thinking)" },
{ id: "claude-haiku-4.5-thinking", name: "Claude Haiku 4.5 (Thinking)" },
{ id: "claude-sonnet-5-agentic", name: "Claude Sonnet 5 (Agentic)" },
{ id: "claude-sonnet-4.5-agentic", name: "Claude Sonnet 4.5 (Agentic)" },
{ id: "claude-haiku-4.5-agentic", name: "Claude Haiku 4.5 (Agentic)" },
{ id: "claude-sonnet-5-thinking-agentic", name: "Claude Sonnet 5 (Thinking + Agentic)" },
{ id: "claude-sonnet-4.5-thinking-agentic", name: "Claude Sonnet 4.5 (Thinking + Agentic)" },
{ id: "claude-haiku-4.5-thinking-agentic", name: "Claude Haiku 4.5 (Thinking + Agentic)" },
],

View File

@@ -20,8 +20,13 @@ export default {
validateUrl: "https://integrate.api.nvidia.com/v1/models",
},
models: [
{ id: "minimaxai/minimax-m2.7", name: "Minimax M2.7" },
{ id: "z-ai/glm4.7", name: "GLM 4.7" },
{ id: "minimaxai/minimax-m2.7", name: "MiniMax M2.7" },
{ id: "minimaxai/minimax-m3", name: "MiniMax M3" },
{ id: "z-ai/glm-5.2", name: "GLM 5.2" },
{ id: "deepseek-ai/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "deepseek-ai/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
{ id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra" },
{ id: "nvidia/nv-embedqa-e5-v5", name: "NV EmbedQA E5 v5", kind: "embedding" },
{ id: "nvidia/parakeet-ctc-1.1b-asr", name: "Parakeet CTC 1.1B", params: ["language"], kind: "stt" },
{ id: "fastpitch", name: "FastPitch", kind: "tts" },

View File

@@ -21,6 +21,11 @@ export default {
},
category: "apikey",
hasProviderSpecificData: true,
regions: [
{ id: "sgp", label: "Singapore (新加坡)" },
{ id: "cn", label: "China (中国大陆)" },
{ id: "ams", label: "Amsterdam (阿姆斯特丹)" },
],
defaultRegion: "sgp",
transport: {
baseUrl: "https://token-plan-sgp.xiaomimimo.com/v1/chat/completions",

View File

@@ -73,6 +73,14 @@ function maskEndpoint(endpoint) {
}
}
function hasUnsafeResponsesInputForCompression(body) {
if (!Array.isArray(body?.input)) return false;
return body.input.some((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return false;
return typeof item.type === "string" && item.type !== "message";
});
}
// POST messages to Headroom /v1/compress; returns compressed messages + stats or null.
async function callCompress(url, messages, model, timeoutMs, compressUserMessages, diagnostics) {
const endpoint = buildCompressEndpoint(url);
@@ -143,6 +151,10 @@ export async function compressWithHeadroom(body, { enabled, url, model, format,
// messages. Translate input -> OpenAI -> compress -> translate back to input so
// body.input keeps the Responses contract (the proxy only understands OpenAI). (#1998)
if (format === "openai-responses") {
if (hasUnsafeResponsesInputForCompression(body)) {
setDiagnostic(diagnostics, "skipped: openai-responses tool/reasoning input is not safe to compress");
return null;
}
const oai = openaiResponsesToOpenAIRequest(model, body, false);
if (!Array.isArray(oai?.messages)) return null;
const data = await callCompress(url, oai.messages, model, timeoutMs, compressUserMessages, diagnostics || {});

View File

@@ -0,0 +1,63 @@
import { buildClineHeaders } from "../shared/clineAuth.js";
const CLINEPASS_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/models";
const FETCH_TIMEOUT_MS = 5000;
/**
* Build request headers for the ClinePass /models endpoint (Cline's upstream API).
* - API keys are sent as plain Bearer tokens.
* - OAuth access tokens must carry the WorkOS `workos:` prefix (handled by buildClineHeaders).
*/
function buildModelListHeaders(token, isApiKey) {
if (isApiKey) {
return {
Accept: "application/json",
Authorization: `Bearer ${token}`,
};
}
return buildClineHeaders(token, { Accept: "application/json" });
}
/**
* Fetch ClinePass live model catalog from Cline's /models endpoint.
*
* @param {object} credentials - Connection credentials ({ accessToken, apiKey })
* @returns {Promise<{ models: { id: string, name: string }[] } | null>}
*/
export async function resolveClinepassModels(credentials) {
const isApiKey = Boolean(credentials?.apiKey);
const token = isApiKey ? credentials.apiKey : credentials?.accessToken;
if (!token) return null;
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
try {
const headers = buildModelListHeaders(token, isApiKey);
const response = await fetch(CLINEPASS_MODELS_ENDPOINT, {
method: "GET",
headers,
signal: controller.signal,
});
if (!response.ok) return null;
const json = await response.json();
const rawList = Array.isArray(json) ? json : json?.data;
if (!Array.isArray(rawList)) return null;
const models = rawList
.filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/"))
.map((m) => ({
id: m.id,
name: m.name || m.id,
}));
return models.length ? { models } : null;
} catch {
return null;
} finally {
clearTimeout(timer);
}
}

View File

@@ -0,0 +1,176 @@
import { createHash } from "crypto";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
export const KIMCHI_API = "https://llm.kimchi.dev";
export const KIMCHI_USER_AGENT = "kimchi/0.1.40";
const FETCH_TIMEOUT_MS = 20_000;
const CACHE_TTL_MS = 5 * 60 * 1000;
const RETRYABLE_STATUSES = new Set([429, 500, 502, 503, 504]);
/** @type {Map<string, { expiresAt: number, models: object[], rawModels: object[] }>} */
const catalogCache = new Map();
/** @type {Map<string, object>} */
const metadataByModelId = new Map();
function normalizeKimchiEndpoint(endpoint) {
const raw = typeof endpoint === "string" ? endpoint.trim() : "";
return (raw || KIMCHI_API).replace(/\/+$/, "");
}
export function buildKimchiModelsUrl(endpoint) {
return `${normalizeKimchiEndpoint(endpoint)}/v1/models/metadata?include_in_cli=true`;
}
function readToken(credentials) {
return (
credentials?.accessToken
|| credentials?.apiKey
|| credentials?.providerSpecificData?.apiKey
|| null
);
}
function cacheKey(credentials, endpoint) {
const psd = credentials?.providerSpecificData || {};
const seed = psd.userId || psd.username || credentials?.refreshToken || readToken(credentials) || "anonymous";
return createHash("sha256")
.update(`kimchi:${normalizeKimchiEndpoint(endpoint)}:${seed}`)
.digest("hex");
}
function toModelKind(inputModalities) {
return Array.isArray(inputModalities) && inputModalities.includes("image")
? "imageToText"
: "llm";
}
export function normalizeKimchiModel(item) {
if (!item || typeof item !== "object") return null;
const id = item.slug || item.id || item.model || item.name;
if (typeof id !== "string" || id.trim() === "") return null;
const inputModalities = Array.isArray(item.input_modalities)
? item.input_modalities.filter((value) => value === "text" || value === "image")
: [];
const limits = item.limits && typeof item.limits === "object" ? item.limits : {};
const contextLength = Number(limits.context_window || item.contextLength || item.context_length) || undefined;
const maxOutputTokens = Number(limits.max_output_tokens || item.maxOutputTokens || item.max_output_tokens) || undefined;
const upstreamProvider = typeof item.provider === "string" ? item.provider : "";
const reasoning = item.reasoning === true;
const kind = toModelKind(inputModalities);
const model = {
...item,
id: id.trim(),
name: String(item.display_name || item.displayName || item.name || id).trim(),
provider: upstreamProvider,
upstreamProvider,
reasoning,
inputModalities,
kind,
type: kind,
capabilities: {
vision: inputModalities.includes("image"),
reasoning,
...(contextLength ? { contextWindow: contextLength } : {}),
...(maxOutputTokens ? { maxOutput: maxOutputTokens } : {}),
...(upstreamProvider ? { upstreamProvider } : {}),
},
...(contextLength ? { contextLength } : {}),
...(maxOutputTokens ? { maxOutputTokens } : {}),
};
if (upstreamProvider === "anthropic") {
model.compat = { supportsReasoningEffort: false, cacheControlFormat: "anthropic" };
}
return model;
}
function rememberModels(models) {
for (const model of models || []) {
if (!model?.id) continue;
metadataByModelId.set(model.id, model);
metadataByModelId.set(model.id.toLowerCase(), model);
}
}
export function getCachedKimchiModelMetadata(modelId) {
if (typeof modelId !== "string" || modelId.trim() === "") return null;
const raw = modelId.includes("/") ? modelId.split("/").pop() : modelId;
return metadataByModelId.get(raw) || metadataByModelId.get(raw.toLowerCase()) || null;
}
async function fetchKimchiCatalogRaw(token, endpoint, options = {}) {
const url = buildKimchiModelsUrl(endpoint);
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(new Error("Kimchi models fetch timeout")), FETCH_TIMEOUT_MS);
const signal = options.signal
? AbortSignal.any([options.signal, controller.signal])
: controller.signal;
try {
const response = await proxyAwareFetch(url, {
method: "GET",
headers: {
"Accept": "application/json",
"Authorization": `Bearer ${token}`,
"User-Agent": KIMCHI_USER_AGENT,
},
cache: "no-store",
signal,
}, options.proxyOptions || null);
if (!response.ok) {
const error = new Error(`Kimchi models ${response.status}: ${response.statusText}`);
error.status = response.status;
error.retryable = RETRYABLE_STATUSES.has(response.status);
throw error;
}
const data = await response.json();
return Array.isArray(data?.models) ? data.models : [];
} finally {
clearTimeout(timeout);
}
}
export async function resolveKimchiModels(credentials, options = {}) {
const token = readToken(credentials);
if (!token) return null;
const endpoint = credentials?.providerSpecificData?.kimchiEndpoint || options.endpoint || KIMCHI_API;
const key = cacheKey(credentials, endpoint);
const now = Date.now();
if (!options.forceRefresh) {
const cached = catalogCache.get(key);
if (cached && cached.expiresAt > now) return cached;
}
let rawModels;
try {
rawModels = await fetchKimchiCatalogRaw(token, endpoint, options);
} catch (error) {
options.log?.warn?.("KIMCHI_MODELS", error.message);
return null;
}
const models = rawModels.map(normalizeKimchiModel).filter(Boolean);
if (models.length === 0) return null;
rememberModels(models);
const entry = {
expiresAt: Date.now() + CACHE_TTL_MS,
models,
rawModels,
};
catalogCache.set(key, entry);
return entry;
}
export function clearKimchiCatalog() {
catalogCache.clear();
metadataByModelId.clear();
}

View File

@@ -5,9 +5,9 @@
import { getGitHubUsage } from "./usage/github.js";
import { getGeminiUsage, getAntigravityUsage } from "./usage/google.js";
import { getClaudeUsage } from "./usage/claude.js";
import { getCodexUsage, consumeCodexRateLimitResetCredit } from "./usage/codex.js";
import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits } from "./usage/codex.js";
export { consumeCodexRateLimitResetCredit };
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
import { getKiroUsage } from "./usage/kiro.js";
import { getMiniMaxUsage } from "./usage/minimax.js";
import { getCodeBuddyCnUsage } from "./usage/codebuddy-cn.js";

View File

@@ -109,15 +109,22 @@ export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificD
total: num(acc.CycleCapacitySizePrecise, acc.CycleCapacitySize),
resetAt: parseResetTime(acc.CycleEndTime),
unlimited: false,
// Recurring allowance: the CycleEndTime is the next refresh, not the
// final expiry. The UI must show "Resets in", not "Expires in".
recurring: true,
};
});
// Bonus packs: use the lifetime Capacity balance; resetAt is the expiry.
// These are one-shot credits (CycleEndTime == DeductionEndTime), so they
// never replenish — mark recurring:false so the UI shows "Expires in"
// instead of implying a monthly refill.
bonuses.forEach((acc, i) => {
quotas[`Bonus Pack ${i + 1}`] = {
used: num(acc.CapacityUsedPrecise, acc.CapacityUsed),
total: num(acc.CapacitySizePrecise, acc.CapacitySize),
resetAt: parseResetTime(acc.CycleEndTime),
unlimited: false,
recurring: false,
};
});

View File

@@ -8,9 +8,23 @@ import { U, parseResetTime, toFiniteNumber } from "./shared.js";
// Codex (OpenAI) API config
const CODEX_CONFIG = {
usageUrl: U("codex").url,
resetCreditsUrl: U("codex").resetCreditsUrl,
resetCreditsConsumeUrl: U("codex").resetCreditsConsumeUrl,
};
function toIsoDate(value) {
if (!value) return null;
const date = value instanceof Date
? value
: new Date(typeof value === "number" && value < 1e12 ? value * 1000 : value);
const time = date.getTime();
return Number.isFinite(time) ? date.toISOString() : null;
}
function getCodexAccountId(providerSpecificData) {
return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null;
}
function getCodexRateLimitBody(snapshot) {
if (!snapshot || typeof snapshot !== "object" || Array.isArray(snapshot)) return null;
return snapshot.rate_limit && typeof snapshot.rate_limit === "object"
@@ -101,6 +115,48 @@ export async function getCodexUsage(accessToken, proxyOptions = null) {
}
}
export async function getCodexRateLimitResetCredits(accessToken, proxyOptions = null, providerSpecificData = null) {
if (!accessToken) {
throw new Error("No Codex access token available. Please re-authorize the connection.");
}
const accountId = getCodexAccountId(providerSpecificData);
const headers = {
"Authorization": `Bearer ${accessToken}`,
"Accept": "application/json",
"OpenAI-Beta": "codex-1",
"originator": "codex_cli_rs",
};
if (accountId) headers["ChatGPT-Account-ID"] = accountId;
const response = await proxyAwareFetch(CODEX_CONFIG.resetCreditsUrl, {
method: "GET",
headers,
}, proxyOptions);
let data = null;
try {
data = await response.json();
} catch {
data = null;
}
if (!response.ok) {
const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`;
throw new Error(message);
}
const credits = Array.isArray(data?.credits) ? data.credits : [];
return {
availableCount: Math.max(0, toFiniteNumber(data?.available_count ?? data?.availableCount, 0)),
credits: credits.map((credit) => ({
status: String(credit?.status || "unknown"),
grantedAt: toIsoDate(credit?.granted_at ?? credit?.grantedAt),
expiresAt: toIsoDate(credit?.expires_at ?? credit?.expiresAt),
})),
};
}
// Consume one Codex rate-limit reset credit (irreversible, spends 1 credit)
export async function consumeCodexRateLimitResetCredit(accessToken, redeemRequestId, proxyOptions = null) {
if (!accessToken) {

View File

@@ -39,7 +39,16 @@ const USAGE_EXTRACTORS = {
},
kiro(raw) {
const input = n(raw.inputTokens), output = n(raw.outputTokens);
return { promptTokens: input, completionTokens: output, totalTokens: input + output };
// ponytail: Amazon Q (Kiro upstream) does not expose cache fields today,
// but pass through any cache_read/cache_creation/cached_tokens if the
// event shape grows them later so cost tracking keeps working without
// a second pass.
const cached = n(raw.cache_read_input_tokens) || n(raw.cachedTokens) || n(raw.cached_tokens);
const cacheCreation = n(raw.cache_creation_input_tokens);
const out = { promptTokens: input, completionTokens: output, totalTokens: input + output };
if (cached > 0) out.cachedTokens = cached;
if (cacheCreation > 0) out.cacheCreationTokens = cacheCreation;
return out;
},
ollama(raw) {
const input = n(raw.prompt_eval_count), output = n(raw.eval_count);

View File

@@ -150,6 +150,33 @@ export function normalizeClaudePassthrough(body, model = "") {
}
}
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// so foreign signatures leak into history and Anthropic rejects them).
const thinkingEnabled = body.thinking?.type === "enabled";
if (Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
let hasToolUse = false;
let hasKeptThinking = false;
const kept = [];
for (const block of msg.content) {
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
if (isValidClaudeSignature(block.signature)) {
hasKeptThinking = true;
kept.push(block);
}
continue;
}
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
kept.push(block);
}
msg.content = kept;
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
msg.content.unshift(buildThinkingPlaceholder("claude"));
}
}
}
return body;
}

View File

@@ -12,6 +12,8 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
"default", "examples",
// JSON Schema meta keywords
"$schema", "$defs", "definitions", "const", "$ref", "$comment",
// Annotation keywords (rejected by Gemini/Antigravity - e.g. MCP tool schemas set these)
"deprecated", "readOnly", "writeOnly",
// Object validation keywords (not supported)
"additionalProperties", "propertyNames", "patternProperties", "enumDescriptions",
// Complex schema keywords (handled by flattenAnyOfOneOf/mergeAllOf)
@@ -19,7 +21,7 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
// Dependency keywords (not supported)
"dependencies", "dependentSchemas", "dependentRequired",
// Other unsupported keywords
"title", "optional", "if", "then", "else", "contentMediaType", "contentEncoding",
"title", "optional", "deprecated", "if", "then", "else", "contentMediaType", "contentEncoding",
// UI/Styling properties (from Cursor tools - NOT JSON Schema standard)
"cornerRadius", "fillColor", "fontFamily", "fontSize", "fontWeight",
"gap", "padding", "strokeColor", "strokeThickness", "textColor"

View File

@@ -6,42 +6,46 @@ export { VALID_OPENAI_CONTENT_TYPES, VALID_OPENAI_MESSAGE_TYPES };
// Filter messages to OpenAI standard format
// Remove: thinking, redacted_thinking, signature, and other non-OpenAI blocks
export function filterToOpenAIFormat(body) {
// opts.preserveCacheControl: keep cache_control on content blocks (e.g. for DashScope/alicode)
export function filterToOpenAIFormat(body, opts = {}) {
if (!body.messages || !Array.isArray(body.messages)) return body;
const keepCache = !!opts.preserveCacheControl;
function stripBlock(block) {
const { signature, cache_control, ...rest } = block;
return keepCache && cache_control ? { ...rest, cache_control } : rest;
}
body.messages = body.messages.map(msg => {
// Normalize developer role to system (many providers don't support developer)
if (msg.role === ROLE.DEVELOPER) msg = { ...msg, role: ROLE.SYSTEM };
// Keep tool messages as-is (OpenAI format)
if (msg.role === ROLE.TOOL) return msg;
// Keep assistant messages with tool_calls as-is
if (msg.role === ROLE.ASSISTANT && msg.tool_calls) return msg;
// Handle string content
if (typeof msg.content === "string") return msg;
// Handle array content
if (Array.isArray(msg.content)) {
const filteredContent = [];
for (const block of msg.content) {
// Skip thinking blocks
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
// Only keep valid OpenAI content types
if (VALID_OPENAI_CONTENT_TYPES.includes(block.type)) {
// Remove signature field if exists
const { signature, cache_control, ...cleanBlock } = block;
filteredContent.push(cleanBlock);
filteredContent.push(stripBlock(block));
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
// Convert tool_use to tool_calls format (handled separately)
continue;
} else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) {
// Keep tool_result but clean it
const { signature, cache_control, ...cleanBlock } = block;
filteredContent.push(cleanBlock);
filteredContent.push(stripBlock(block));
}
}

View File

@@ -109,7 +109,9 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
// Always normalize to clean OpenAI format when target is OpenAI
// This handles hybrid requests (e.g., OpenAI messages + Claude tools)
if (targetFormat === FORMATS.OPENAI) {
result = filterToOpenAIFormat(result);
result = filterToOpenAIFormat(result, {
preserveCacheControl: !!PROVIDERS[provider]?.quirks?.preserveCacheControl,
});
}
// Final step: prepare request for Claude format endpoints

View File

@@ -138,12 +138,14 @@ function convertContent(content) {
// Text with thoughtSignature = regular text after thinking
if (part.thoughtSignature && part.text !== undefined) {
textParts.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
if (part.text) {
textParts.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
}
continue;
}
// Regular text
if (part.text !== undefined) {
if (part.text !== undefined && part.text !== "") {
textParts.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
}
@@ -180,8 +182,22 @@ function convertContent(content) {
}
}
// Content with only functionResponses → return array of tool messages
// Content with functionResponses — return array of tool result messages,
// plus an assistant message for any co-located tool calls / text.
if (toolResults.length > 0) {
if (toolCalls.length > 0 || textParts.length > 0 || reasoningContent) {
const assistantMsg = { role: ROLE.ASSISTANT };
if (textParts.length > 0) {
assistantMsg.content = collapseTextParts(textParts);
}
if (reasoningContent) {
assistantMsg.reasoning_content = reasoningContent;
}
if (toolCalls.length > 0) {
assistantMsg.tool_calls = toolCalls;
}
return [...toolResults, assistantMsg];
}
return toolResults;
}

View File

@@ -393,9 +393,12 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
reconcileOrphanedToolResults(history, currentMessage);
}
// API-key auth must never use the shared default ARN (403); OAuth/social fall back to it.
// api_key / idc / external_idp must never use the shared default ARN (belongs
// to another account → 403 "bearer token invalid"); OAuth/social fall back to it.
const authMethod = credentials?.providerSpecificData?.authMethod;
const profileArn = authMethod === "api_key"
const accountBoundAuth =
authMethod === "api_key" || authMethod === "idc" || authMethod === "external_idp";
const profileArn = accountBoundAuth
? (credentials?.providerSpecificData?.profileArn || "")
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));

View File

@@ -129,8 +129,24 @@ function fixMissingToolResponsesOpenAI(messages) {
}
}
// Wrap mid-conversation system text so it ends as a user turn (avoids Anthropic prefill 400)
function systemReminderText(content) {
const parts = Array.isArray(content)
? content.filter(c => c?.type === CLAUDE_BLOCK.TEXT).map(c => c.text || "")
: [typeof content === "string" ? content : ""];
const text = parts.filter(Boolean).join("\n");
if (!text.trim()) return "";
return `<system-reminder>\n${text}\n</system-reminder>`;
}
// Convert single Claude message - returns single message or array of messages
function convertClaudeMessage(msg) {
// Mid-conversation system message -> user (per Anthropic placement rules)
if (msg.role === ROLE.SYSTEM) {
const text = systemReminderText(msg.content);
return text ? { role: ROLE.USER, content: text } : null;
}
const role = msg.role === ROLE.USER || msg.role === ROLE.TOOL ? ROLE.USER : ROLE.ASSISTANT;
// Simple string content

View File

@@ -35,6 +35,17 @@ function sanitizeGeminiFunctionName(name) {
return sanitized.substring(0, 64);
}
function normalizeGeminiContents(contents) {
const out = [];
for (const c of contents || []) {
if (!c?.role || !Array.isArray(c.parts) || c.parts.length === 0) continue;
const last = out.at(-1);
if (last?.role === c.role) last.parts.push(...c.parts);
else out.push({ ...c, parts: [...c.parts] });
}
return out;
}
// Core: Convert OpenAI request to Gemini format (base for all variants)
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) {
const result = {
@@ -217,6 +228,7 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
}
}
result.contents = normalizeGeminiContents(result.contents);
return result;
}
@@ -299,7 +311,7 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra
}
// Wrap Claude format in Cloud Code envelope for Antigravity
function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = null) {
function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = null, signature = DEFAULT_THINKING_AG_SIGNATURE) {
const projectId = credentials?.projectId || generateProjectId();
const envelope = {
@@ -343,6 +355,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
parts.push({ text: block.text });
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
parts.push({
thoughtSignature: signature,
functionCall: {
id: block.id,
name: sanitizeGeminiFunctionName(block.name),
@@ -425,6 +438,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
envelope.request.systemInstruction = { role: GEMINI_ROLE.USER, parts: systemParts };
}
envelope.request.contents = normalizeGeminiContents(envelope.request.contents);
return envelope;
}

View File

@@ -530,8 +530,15 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
// (the ARN doesn't belong to the key's account). So for api_key, only send a
// profileArn that was actually resolved for this connection — never the default.
// OAuth/social keep the default fallback (their tokens accept it).
// api_key / idc / external_idp carry an account-specific (or token-bound)
// profile. The shared builder-id/social default ARN belongs to a different
// account and triggers 403 "bearer token invalid", so never fall back to it —
// send the resolved ARN, or an empty string so CodeWhisperer uses the token's
// own default profile. Only OAuth/social keep the shared placeholder.
const authMethod = credentials?.providerSpecificData?.authMethod;
const profileArn = authMethod === "api_key"
const accountBoundAuth =
authMethod === "api_key" || authMethod === "idc" || authMethod === "external_idp";
const profileArn = accountBoundAuth
? (credentials?.providerSpecificData?.profileArn || "")
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));

View File

@@ -27,6 +27,25 @@ export function claudeToOpenAIResponse(chunk, state) {
state.messageId = chunk.message?.id || `msg_${Date.now()}`;
state.model = chunk.message?.model;
state.toolCallIndex = 0;
// Claude sends input_tokens + cache_read + cache_creation here; message_delta
// later carries only the final output_tokens. Capture cache now so the
// delta (output-only) doesn't reset it to zero.
const startUsage = chunk.message?.usage;
if (startUsage && typeof startUsage === "object") {
const inputTokens = typeof startUsage.input_tokens === "number" ? startUsage.input_tokens : 0;
const cacheReadTokens = typeof startUsage.cache_read_input_tokens === "number" ? startUsage.cache_read_input_tokens : 0;
const cacheCreationTokens = typeof startUsage.cache_creation_input_tokens === "number" ? startUsage.cache_creation_input_tokens : 0;
const promptTokens = inputTokens + cacheReadTokens + cacheCreationTokens;
state.usage = {
prompt_tokens: promptTokens,
completion_tokens: 0,
total_tokens: promptTokens,
input_tokens: inputTokens,
output_tokens: 0
};
if (cacheReadTokens > 0) state.usage.cache_read_input_tokens = cacheReadTokens;
if (cacheCreationTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationTokens;
}
results.push(createChunk(state, { role: ROLE.ASSISTANT }));
break;
}
@@ -103,13 +122,15 @@ export function claudeToOpenAIResponse(chunk, state) {
}
case "message_delta": {
// Extract usage from message_delta event (Claude native format)
// Normalize to OpenAI format (prompt_tokens/completion_tokens) for consistent logging
// Extract usage from message_delta event (Claude native format).
// Anthropic sends input/cache in message_start and only output here, so
// fall back to cache captured in message_start when the delta omits it.
if (chunk.usage && typeof chunk.usage === "object") {
const inputTokens = typeof chunk.usage.input_tokens === "number" ? chunk.usage.input_tokens : 0;
const prev = state.usage || {};
const inputTokens = typeof chunk.usage.input_tokens === "number" ? chunk.usage.input_tokens : (prev.input_tokens || 0);
const outputTokens = typeof chunk.usage.output_tokens === "number" ? chunk.usage.output_tokens : 0;
const cacheReadTokens = typeof chunk.usage.cache_read_input_tokens === "number" ? chunk.usage.cache_read_input_tokens : 0;
const cacheCreationTokens = typeof chunk.usage.cache_creation_input_tokens === "number" ? chunk.usage.cache_creation_input_tokens : 0;
const cacheReadTokens = typeof chunk.usage.cache_read_input_tokens === "number" ? chunk.usage.cache_read_input_tokens : (prev.cache_read_input_tokens || 0);
const cacheCreationTokens = typeof chunk.usage.cache_creation_input_tokens === "number" ? chunk.usage.cache_creation_input_tokens : (prev.cache_creation_input_tokens || 0);
// prompt_tokens = input_tokens + cache_read + cache_creation (all prompt-side tokens)
const promptTokens = inputTokens + cacheReadTokens + cacheCreationTokens;
@@ -131,7 +152,14 @@ export function claudeToOpenAIResponse(chunk, state) {
const finalChunk = createChunk(state, {}, state.finishReason);
if (state.usage) {
finalChunk.usage = toOpenAIUsage(chunk.usage, "claude");
// Build OpenAI usage from the merged state (cache from message_start +
// output from message_delta), not the delta chunk alone.
finalChunk.usage = toOpenAIUsage({
input_tokens: state.usage.input_tokens || 0,
output_tokens: state.usage.output_tokens || 0,
cache_read_input_tokens: state.usage.cache_read_input_tokens,
cache_creation_input_tokens: state.usage.cache_creation_input_tokens
}, "claude");
}
results.push(finalChunk);

View File

@@ -25,7 +25,10 @@ function emitFunctionCall(functionCall, state) {
type: OPENAI_BLOCK.FUNCTION,
function: { name: fcName, arguments: JSON.stringify(fcArgs) },
};
state.toolCalls.set(toolCallIndex, toolCall);
// Keep Gemini bookkeeping separate from the shared translator state.toolCalls map.
// The downstream OpenAI→Claude translator uses state.toolCalls for Claude block
// metadata; pre-populating it here makes Anthropic tool deltas lose index.
state.geminiToolCallCount = (state.geminiToolCallCount || 0) + 1;
return buildChunk(chunkMeta(state), { tool_calls: [toolCall] }, null);
}
@@ -46,6 +49,7 @@ export function geminiToOpenAIResponse(chunk, state) {
state.messageId = response.responseId || `msg_${Date.now()}`;
state.model = response.modelVersion || "gemini";
state.functionIndex = 0;
state.geminiToolCallCount = 0;
results.push(buildChunk(chunkMeta(state), { role: ROLE.ASSISTANT }, null));
}
@@ -117,7 +121,7 @@ export function geminiToOpenAIResponse(chunk, state) {
// Finish reason - include usage in final chunk
if (candidate.finishReason) {
let finishReason = toOpenAIFinish(candidate.finishReason, "gemini");
if (finishReason === OPENAI_FINISH.STOP && state.toolCalls.size > 0) {
if (finishReason === OPENAI_FINISH.STOP && state.geminiToolCallCount > 0) {
finishReason = OPENAI_FINISH.TOOL_CALLS;
}

View File

@@ -184,7 +184,8 @@ export function openaiToClaudeResponse(chunk, state) {
for (const tc of delta.tool_calls) {
const idx = tc.index ?? 0;
if (tc.id) {
// GLM/fireworks repeats id+null-name on every arg chunk; open block once per idx
if (tc.id && !state.toolCalls.has(idx)) {
stopThinkingBlock(state, results);
stopTextBlock(state, results);

View File

@@ -247,9 +247,24 @@ function mergeChunksToResponse(chunks, sourceFormat) {
if (messageStart?.message) {
finalChunk = messageStart.message;
// Merge usage if available
if (messageDelta?.usage) {
finalChunk.usage = messageDelta.usage;
// message_start.usage has input + cache; message_delta.usage has the
// final output_tokens. Merge so cache survives (delta omits it).
const startUsage = messageStart.message.usage;
const deltaUsage = messageDelta?.usage;
if (startUsage || deltaUsage) {
finalChunk.usage = {
...(startUsage || {}),
...(deltaUsage || {}),
...(startUsage?.cache_read_input_tokens !== undefined
? { cache_read_input_tokens: startUsage.cache_read_input_tokens }
: {}),
...(startUsage?.cache_creation_input_tokens !== undefined
? { cache_creation_input_tokens: startUsage.cache_creation_input_tokens }
: {}),
...(startUsage?.input_tokens !== undefined
? { input_tokens: startUsage.input_tokens }
: {})
};
}
}
}

View File

@@ -5,6 +5,7 @@ import { formatSSE } from "./streamHelpers.js";
// Responses API events that signal the stream has reached a terminal state
const OPENAI_RESPONSES_TERMINAL_EVENTS = new Set([
"response.completed",
"response.done",
"response.failed",
"error"
]);

View File

@@ -1,8 +1,12 @@
import { translateResponse, initState } from "../translator/index.js";
import { FORMATS } from "../translator/formats.js";
import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb.js";
<<<<<<< HEAD
import { extractUsage, hasValidUsage, estimateUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js";
import { saveUsageStats } from "../handlers/chatCore/requestDetail.js";
=======
import { extractUsage, mergeUsage, hasValidUsage, estimateUsage, logUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js";
>>>>>>> 7f436e2792be4fa5a4d1c4d6b8e9bc85eaaa6a3d
import { parseSSELine, hasValuableContent, fixInvalidId, formatSSE } from "./streamHelpers.js";
import { getOpenAIResponsesEventName, isOpenAIResponsesTerminalEvent, formatIncompleteOpenAIResponsesStreamFailure } from "./responsesStreamHelpers.js";
import { dbg, isDebugEnabled } from "./debugLog.js";
@@ -131,6 +135,20 @@ export function createSSEStream(options = {}) {
}
}
// Strip empty tool_calls arrays that break AI SDK reasoning tracking.
// Some providers (e.g. CodeBuddy CN) include `"tool_calls": []` in
// every streaming delta. @ai-sdk/openai-compatible checks
// `delta.tool_calls != null` — an empty array passes this check,
// causing premature `reasoning-end` on every chunk.
if (parsed?.choices) {
for (const choice of parsed.choices) {
if (choice.delta?.tool_calls && Array.isArray(choice.delta.tool_calls) && choice.delta.tool_calls.length === 0) {
delete choice.delta.tool_calls;
fieldsInjected = true;
}
}
}
if (!hasValuableContent(parsed, FORMATS.OPENAI)) {
continue;
}
@@ -149,7 +167,7 @@ export function createSSEStream(options = {}) {
const extracted = extractUsage(parsed);
if (extracted) {
usage = extracted;
usage = mergeUsage(usage, extracted);
}
const isFinishChunk = parsed.choices?.[0]?.finish_reason;
@@ -218,9 +236,11 @@ export function createSSEStream(options = {}) {
sseEmittedCount++;
}
// [DONE] not emitted in translate mode — some clients' SSE decoders
// fail to parse the OpenAI sentinel on Claude-format translated streams.
// message_stop already signals end-of-response; stream close handles it.
if (keepsOpenAIResponsesFormat && !streamDoneSent) {
const doneOutput = "data: [DONE]\n\n";
reqLogger?.appendConvertedChunk?.(doneOutput);
controller.enqueue(sharedEncoder.encode(doneOutput));
}
streamDoneSent = true;
if (keepsOpenAIResponsesFormat) openAIResponsesDoneSent = true;
continue;
@@ -265,7 +285,7 @@ export function createSSEStream(options = {}) {
// Extract usage
const extracted = extractUsage(parsed);
if (extracted) state.usage = extracted; // Keep original usage for logging
if (extracted) state.usage = mergeUsage(state.usage, extracted); // Keep original usage for logging
// Responses same-format passthrough: re-emit with original event framing
if (keepsOpenAIResponsesFormat && openAIResponsesEventName) {
@@ -351,7 +371,9 @@ export function createSSEStream(options = {}) {
// Some clients (e.g. OpenClaw) expect the OpenAI-style sentinel:
// data: [DONE]\n\n
// Without it they can hang until timeout and trigger failover.
if (!streamDoneSent) {
// Gemini-family clients (Antigravity, Vertex, Gemini) reject this sentinel with 400 syntax errors.
const isGeminiFamily = provider === "antigravity" || provider === "gemini" || provider === "vertex";
if (!streamDoneSent && !isGeminiFamily) {
const doneOutput = "data: [DONE]\n\n";
reqLogger?.appendConvertedChunk?.(doneOutput);
controller.enqueue(sharedEncoder.encode(doneOutput));
@@ -416,8 +438,13 @@ export function createSSEStream(options = {}) {
openAIResponsesTerminalSeen = true;
}
// [DONE] not emitted in translate mode — see comment above.
// Passthrough mode still emits it for standard OpenAI clients.
if (keepsOpenAIResponsesFormat && !openAIResponsesDoneSent && !streamDoneSent) {
const doneOutput = "data: [DONE]\n\n";
reqLogger?.appendConvertedChunk?.(doneOutput);
controller.enqueue(sharedEncoder.encode(doneOutput));
openAIResponsesDoneSent = true;
streamDoneSent = true;
}
if (!hasValidUsage(state?.usage) && totalContentLength > 0) {
state.usage = estimateUsage(body, totalContentLength, sourceFormat);

View File

@@ -141,6 +141,68 @@ export function normalizeUsage(usage) {
return normalized;
}
/**
* Canonicalize usage into ONE storage/cost convention so token counts and cost
* are consistent across providers:
* prompt_tokens = total input INCLUDING cache read + cache creation
* cached_tokens = cache-read portion (subset of prompt_tokens)
* cache_creation_input_tokens = cache-write portion (subset of prompt_tokens)
* completion_tokens, reasoning_tokens, total_tokens
*
* Discriminator: Claude reports cache_read_input_tokens with a prompt that
* EXCLUDES cache, so we fold cache into prompt. OpenAI/Gemini report
* cached_tokens already counted inside prompt, so we pass through. Idempotent:
* once folded the output carries cached_tokens (not cache_read_input_tokens),
* so re-running takes the passthrough branch and does not double-add.
*
* @param {object} usage - a normalizeUsage()-shaped object
* @returns {object|null} canonical token object, or null for invalid input
*/
export function canonicalizeUsage(usage) {
if (!usage || typeof usage !== "object" || Array.isArray(usage)) return null;
const num = (v) => (Number.isFinite(Number(v)) ? Number(v) : 0);
const completion = num(usage.completion_tokens ?? usage.output_tokens);
const reasoning = num(usage.reasoning_tokens);
// Fall back to the nested prompt_tokens_details.cache_creation_tokens shape
// (buildUsage()'s OpenAI-forwarding format) when the top-level field is
// absent, so callers that pass a buildUsage() object through don't silently
// drop cache_creation.
const cacheCreation = num(usage.cache_creation_input_tokens ?? usage.prompt_tokens_details?.cache_creation_tokens);
let prompt = num(usage.prompt_tokens ?? usage.input_tokens);
let cached;
// Claude path: prompt excludes cache; cache_read_input_tokens and/or
// cache_creation_input_tokens are separate. A cache-miss "first write" only
// carries cache_creation_input_tokens (no cache_read_input_tokens yet), so
// check both fields — otherwise a first-write request falls through to the
// OpenAI passthrough branch below and cache_creation never gets folded in.
// Guard on the absence of `cached_tokens`: our own canonical output always
// sets that key (even to 0), so re-running canonicalizeUsage on an already-
// folded result takes the passthrough branch instead of folding again.
if (usage.cached_tokens === undefined &&
(usage.cache_read_input_tokens !== undefined || usage.cache_creation_input_tokens !== undefined)) {
cached = num(usage.cache_read_input_tokens);
prompt = prompt + cached + cacheCreation;
} else {
// OpenAI/Gemini path (or already-canonical input): prompt already includes cached_tokens.
cached = num(usage.cached_tokens);
}
const result = {
prompt_tokens: prompt,
completion_tokens: completion,
// Recompute rather than pass through: when the fold branch ran above,
// an upstream total_tokens (cache-exclusive) would otherwise be stale.
total_tokens: prompt + completion,
cached_tokens: cached,
cache_creation_input_tokens: cacheCreation,
};
if (reasoning > 0) result.reasoning_tokens = reasoning;
return result;
}
/**
* Check if usage has valid token data
* Valid = has at least one token field with value > 0
@@ -171,6 +233,19 @@ export function hasValidUsage(usage) {
export function extractUsage(chunk) {
if (!chunk || typeof chunk !== "object") return null;
// Claude format (message_start event): carries input_tokens + cache_read +
// cache_creation. message_delta later carries only the final output_tokens,
// so callers must MERGE (mergeUsage), not overwrite, to keep cache counts.
if (chunk.type === "message_start" && chunk.message?.usage && typeof chunk.message.usage === "object") {
const u = chunk.message.usage;
return normalizeUsage({
prompt_tokens: u.input_tokens || 0,
completion_tokens: u.output_tokens || 0,
cache_read_input_tokens: u.cache_read_input_tokens,
cache_creation_input_tokens: u.cache_creation_input_tokens
});
}
// Claude format (message_delta event)
if (chunk.type === "message_delta" && chunk.usage && typeof chunk.usage === "object") {
return normalizeUsage({
@@ -232,6 +307,27 @@ export function extractUsage(chunk) {
return null;
}
// Field-wise max-merge of two usage objects. Anthropic splits usage across
// events: message_start has real input+cache (output is a placeholder 1),
// message_delta has the real cumulative output (input/cache absent). Max keeps
// the meaningful value from each without clobbering. Idempotent for other
// providers that emit a single complete usage object.
export function mergeUsage(prev, next) {
if (!prev) return next || null;
if (!next) return prev;
const merged = { ...prev };
for (const [k, v] of Object.entries(next)) {
// typeof NaN === "number" — guard with Number.isFinite so one malformed
// chunk can't poison the whole accumulation (Math.max(x, NaN) is NaN).
if (typeof v === "number" && Number.isFinite(v)) {
merged[k] = Math.max(typeof merged[k] === "number" ? merged[k] : 0, v);
} else if (v && typeof v === "object") {
merged[k] = v; // nested details objects: take latest
}
}
return merged;
}
/**
* Estimate input tokens from request body
* Calculate total body size for more accurate estimation