end
This commit is contained in:
@@ -6,6 +6,7 @@ import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
|
||||
|
||||
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
|
||||
function sanitizeFunctionName(name) {
|
||||
@@ -177,8 +178,19 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
if (p.thoughtSignature && !p.functionCall && !p.text) return false;
|
||||
return true;
|
||||
});
|
||||
if (role !== c.role || parts?.length !== c.parts?.length) {
|
||||
return { ...c, role, parts };
|
||||
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
|
||||
// don't persist thoughtSignature in their history, so backfill the default signature on any
|
||||
// functionCall part that arrives without one.
|
||||
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
|
||||
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
|
||||
return {
|
||||
...c, role,
|
||||
parts: needsBackfill
|
||||
? parts.map(p => (p.functionCall && !p.thoughtSignature)
|
||||
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
|
||||
: p)
|
||||
: parts,
|
||||
};
|
||||
}
|
||||
return c;
|
||||
});
|
||||
|
||||
@@ -226,6 +226,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
|
||||
cline: () => this.refreshCline(credentials.refreshToken, proxyOptions),
|
||||
clinepass: () => this.refreshCline(credentials.refreshToken, proxyOptions),
|
||||
"kimi-coding": () => this.refreshKimiCoding(credentials.refreshToken, proxyOptions),
|
||||
kilocode: () => this.refreshKilocode(credentials.refreshToken, proxyOptions)
|
||||
};
|
||||
@@ -299,7 +300,11 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
const data = payload?.data || payload;
|
||||
const expiresAtIso = data?.expiresAt;
|
||||
const expiresIn = expiresAtIso ? Math.max(1, Math.floor((new Date(expiresAtIso).getTime() - Date.now()) / 1000)) : undefined;
|
||||
return { accessToken: data?.accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
|
||||
let accessToken = data?.accessToken;
|
||||
if (accessToken && !accessToken.startsWith("workos:")) {
|
||||
accessToken = `workos:${accessToken}`;
|
||||
}
|
||||
return { accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
|
||||
}
|
||||
|
||||
async refreshKimiCoding(refreshToken, proxyOptions = null) {
|
||||
|
||||
@@ -5,6 +5,7 @@ import { GithubExecutor } from "./github.js";
|
||||
import { IFlowExecutor } from "./iflow.js";
|
||||
import { QoderExecutor } from "./qoder.js";
|
||||
import { KiroExecutor } from "./kiro.js";
|
||||
import { KimchiExecutor } from "./kimchi.js";
|
||||
import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
@@ -28,6 +29,7 @@ const executors = {
|
||||
iflow: new IFlowExecutor(),
|
||||
qoder: new QoderExecutor(),
|
||||
kiro: new KiroExecutor(),
|
||||
kimchi: new KimchiExecutor(),
|
||||
codex: new CodexExecutor(),
|
||||
cursor: new CursorExecutor(),
|
||||
cu: new CursorExecutor(), // Alias for cursor
|
||||
@@ -66,6 +68,7 @@ export { GithubExecutor } from "./github.js";
|
||||
export { IFlowExecutor } from "./iflow.js";
|
||||
export { QoderExecutor } from "./qoder.js";
|
||||
export { KiroExecutor } from "./kiro.js";
|
||||
export { KimchiExecutor } from "./kimchi.js";
|
||||
export { CodexExecutor } from "./codex.js";
|
||||
export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
|
||||
123
open-sse/executors/kimchi.js
Normal file
123
open-sse/executors/kimchi.js
Normal file
@@ -0,0 +1,123 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { getCachedKimchiModelMetadata } from "../services/kimchiModels.js";
|
||||
|
||||
const TOP_LEVEL_OPENAI_GATEWAY_DROPS = [
|
||||
"anthropic_version",
|
||||
"anthropic_beta",
|
||||
"client_metadata",
|
||||
"mcp_servers",
|
||||
"stop_sequences",
|
||||
"thinking",
|
||||
"top_k",
|
||||
];
|
||||
|
||||
function systemToText(system) {
|
||||
if (typeof system === "string") return system;
|
||||
if (Array.isArray(system)) {
|
||||
return system
|
||||
.map((part) => {
|
||||
if (typeof part === "string") return part;
|
||||
if (typeof part?.text === "string") return part.text;
|
||||
return "";
|
||||
})
|
||||
.filter(Boolean)
|
||||
.join("\n");
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
function mergeTopLevelSystem(body) {
|
||||
if (!body?.system || !Array.isArray(body.messages)) return;
|
||||
const text = systemToText(body.system).trim();
|
||||
if (!text) return;
|
||||
|
||||
const existing = body.messages.find((msg) => msg?.role === "system");
|
||||
if (!existing) {
|
||||
body.messages.unshift({ role: "system", content: text });
|
||||
return;
|
||||
}
|
||||
|
||||
if (typeof existing.content === "string") {
|
||||
existing.content = `${text}\n\n${existing.content}`;
|
||||
} else if (Array.isArray(existing.content)) {
|
||||
existing.content.unshift({ type: "text", text });
|
||||
}
|
||||
}
|
||||
|
||||
function stripMessageArtifacts(body) {
|
||||
if (!Array.isArray(body?.messages)) return;
|
||||
for (const msg of body.messages) {
|
||||
if (!msg || typeof msg !== "object") continue;
|
||||
delete msg.cache_control;
|
||||
if (!Array.isArray(msg.content)) continue;
|
||||
msg.content = msg.content.map((part) => {
|
||||
if (!part || typeof part !== "object") return part;
|
||||
const { cache_control, signature, ...clean } = part;
|
||||
return clean;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
function stripToolArtifacts(body) {
|
||||
if (!Array.isArray(body?.tools)) return;
|
||||
body.tools = body.tools.map((tool) => {
|
||||
if (!tool || typeof tool !== "object") return tool;
|
||||
const { cache_control, ...clean } = tool;
|
||||
return clean;
|
||||
});
|
||||
}
|
||||
|
||||
// Strip `reasoning_content` echoed by clients on assistant messages — but
|
||||
// only when it's a real thinking block. `DefaultExecutor.transformRequest`
|
||||
// runs `injectReasoningContent` first and may inject a 1-char placeholder
|
||||
// (" ") for upstream validation; the placeholder is small (no token cost
|
||||
// worth stripping) and stripping it would re-trigger upstream to complain
|
||||
// about missing reasoning on the next turn. Threshold matches the
|
||||
// placeholder length with a safety margin.
|
||||
const REASONING_PLACEHOLDER_MAX_LEN = 8;
|
||||
|
||||
export function stripReasoningContent(body) {
|
||||
if (!Array.isArray(body?.messages)) return;
|
||||
for (const msg of body.messages) {
|
||||
if (msg && msg.role === "assistant" && typeof msg.reasoning_content === "string"
|
||||
&& msg.reasoning_content.length > REASONING_PLACEHOLDER_MAX_LEN) {
|
||||
delete msg.reasoning_content;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function isAnthropicBackedKimchiModel(model) {
|
||||
const meta = getCachedKimchiModelMetadata(model);
|
||||
if (meta?.provider === "anthropic" || meta?.upstreamProvider === "anthropic") return true;
|
||||
return /(^|[-_/])(?:claude|anthropic)(?:[-_/]|$)/i.test(String(model || ""));
|
||||
}
|
||||
|
||||
export class KimchiExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("kimchi");
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
const transformed = super.transformRequest(model, body, stream, credentials);
|
||||
if (!transformed || typeof transformed !== "object") return transformed;
|
||||
|
||||
mergeTopLevelSystem(transformed);
|
||||
for (const key of TOP_LEVEL_OPENAI_GATEWAY_DROPS) {
|
||||
if (transformed[key] !== undefined) delete transformed[key];
|
||||
}
|
||||
delete transformed.system;
|
||||
|
||||
if (isAnthropicBackedKimchiModel(model)) {
|
||||
delete transformed.reasoning_effort;
|
||||
delete transformed.reasoning;
|
||||
delete transformed.thinking;
|
||||
}
|
||||
|
||||
stripMessageArtifacts(transformed);
|
||||
stripToolArtifacts(transformed);
|
||||
stripReasoningContent(transformed);
|
||||
return transformed;
|
||||
}
|
||||
}
|
||||
|
||||
export default KimchiExecutor;
|
||||
@@ -64,9 +64,22 @@ export class KiroExecutor extends BaseExecutor {
|
||||
getOrderedBaseUrls(credentials) {
|
||||
const baseUrls = this.getBaseUrls();
|
||||
const authMethod = credentials?.providerSpecificData?.authMethod;
|
||||
const isCodeWhispererSurface = authMethod === "api_key" || authMethod === "external_idp";
|
||||
// IAM Identity Center (idc) tokens are AWS SSO access tokens — the same
|
||||
// family as external_idp/api_key. The kiro.dev gateway rejects them with
|
||||
// 403 "bearer token invalid", so they must hit the CodeWhisperer
|
||||
// *.amazonaws.com surface, and in the region the token was minted in
|
||||
// (the baseUrls are hardcoded us-east-1).
|
||||
const isCodeWhispererSurface =
|
||||
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
|
||||
if (!isCodeWhispererSurface) return baseUrls;
|
||||
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com"));
|
||||
|
||||
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
|
||||
const regionalize = (u) =>
|
||||
region && region !== "us-east-1" && u.includes("amazonaws.com")
|
||||
? u.replace(/([a-z]+)\.[a-z0-9-]+\.amazonaws\.com/, `$1.${region}.amazonaws.com`)
|
||||
: u;
|
||||
|
||||
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
|
||||
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
|
||||
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
|
||||
}
|
||||
@@ -122,7 +135,8 @@ export class KiroExecutor extends BaseExecutor {
|
||||
hasReasoningContent: false,
|
||||
reasoningChunkCount: 0,
|
||||
toolCallIndex: 0,
|
||||
seenToolIds: new Map()
|
||||
seenToolIds: new Map(),
|
||||
inThinking: false
|
||||
};
|
||||
|
||||
const transformStream = new TransformStream({
|
||||
@@ -159,7 +173,36 @@ export class KiroExecutor extends BaseExecutor {
|
||||
|
||||
// Handle assistantResponseEvent
|
||||
if (eventType === "assistantResponseEvent" && event.payload?.content) {
|
||||
const content = event.payload.content;
|
||||
let content = event.payload.content;
|
||||
|
||||
// Kiro Claude models can leak <thinking> blocks into the content stream.
|
||||
// We strip these literal tags to prevent duplication, as the reasoning
|
||||
// is already routed correctly via reasoningContentEvent.
|
||||
if (state.inThinking) {
|
||||
if (content.includes("</thinking>")) {
|
||||
state.inThinking = false;
|
||||
const after = content.split("</thinking>").slice(1).join("</thinking>");
|
||||
content = after.startsWith("\n") ? after.substring(1) : after;
|
||||
} else {
|
||||
content = ""; // Drop entirely while inside thinking block
|
||||
}
|
||||
} else if (content.includes("<thinking>")) {
|
||||
state.inThinking = true;
|
||||
if (content.includes("</thinking>")) {
|
||||
state.inThinking = false;
|
||||
const before = content.split("<thinking>")[0];
|
||||
const after = content.split("</thinking>").slice(1).join("</thinking>");
|
||||
content = before + (after.startsWith("\n") ? after.substring(1) : after);
|
||||
} else {
|
||||
content = content.split("<thinking>")[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (!content && state.hasReasoningContent) {
|
||||
// If we stripped everything, skip emitting an empty content chunk
|
||||
continue;
|
||||
}
|
||||
|
||||
state.totalContentLength += content.length;
|
||||
|
||||
const chunk = {
|
||||
@@ -348,6 +391,11 @@ export class KiroExecutor extends BaseExecutor {
|
||||
if (metrics && typeof metrics === 'object') {
|
||||
const inputTokens = metrics.inputTokens || 0;
|
||||
const outputTokens = metrics.outputTokens || 0;
|
||||
// ponytail: Amazon Q upstream does not expose cache fields today,
|
||||
// but pick up cache_read_input_tokens / cache_creation_input_tokens
|
||||
// if the event shape grows them so cost tracking stays accurate.
|
||||
const cachedTokens = metrics.cacheReadInputTokens || metrics.cache_read_input_tokens || 0;
|
||||
const cacheCreationInputTokens = metrics.cacheCreationInputTokens || metrics.cache_creation_input_tokens || 0;
|
||||
|
||||
if (inputTokens > 0 || outputTokens > 0) {
|
||||
state.usage = {
|
||||
@@ -355,6 +403,12 @@ export class KiroExecutor extends BaseExecutor {
|
||||
completion_tokens: outputTokens,
|
||||
total_tokens: inputTokens + outputTokens
|
||||
};
|
||||
// Kiro is Claude-backed: inputTokens EXCLUDES cache (Claude convention),
|
||||
// not inclusive like OpenAI's cached_tokens. Emit cache_read_input_tokens
|
||||
// (not cached_tokens) so canonicalizeUsage takes the Claude fold path and
|
||||
// correctly adds cache back into prompt_tokens instead of undercharging.
|
||||
if (cachedTokens > 0) state.usage.cache_read_input_tokens = cachedTokens;
|
||||
if (cacheCreationInputTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationInputTokens;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -326,8 +326,8 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
}
|
||||
|
||||
// Streaming response
|
||||
const { onStreamComplete } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, streamController, onStreamComplete });
|
||||
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId });
|
||||
}
|
||||
|
||||
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { needsTranslation } from "../../translator/index.js";
|
||||
import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
|
||||
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
|
||||
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
@@ -9,11 +10,65 @@ import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, sav
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
|
||||
function parseToolArguments(value) {
|
||||
if (!value) return {};
|
||||
if (typeof value === "object") return value;
|
||||
try {
|
||||
return JSON.parse(value);
|
||||
} catch {
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
function openAICompletionToClaudeMessage(responseBody) {
|
||||
if (!responseBody?.choices?.[0]) return responseBody;
|
||||
const choice = responseBody.choices[0];
|
||||
const message = choice.message || {};
|
||||
const content = [];
|
||||
|
||||
const reasoning = message.reasoning_content || message.provider_specific_fields?.reasoning_content || "";
|
||||
if (reasoning) {
|
||||
content.push({ type: "thinking", thinking: reasoning });
|
||||
}
|
||||
if (typeof message.content === "string" && message.content.length > 0) {
|
||||
content.push({ type: "text", text: message.content });
|
||||
}
|
||||
for (const toolCall of message.tool_calls || []) {
|
||||
const fn = toolCall.function || {};
|
||||
content.push({
|
||||
type: "tool_use",
|
||||
id: toolCall.id || `toolu_${Date.now()}_${content.length}`,
|
||||
name: fn.name || toolCall.name || "",
|
||||
input: parseToolArguments(fn.arguments || toolCall.arguments),
|
||||
});
|
||||
}
|
||||
if (content.length === 0) content.push({ type: "text", text: "" });
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
return {
|
||||
id: String(responseBody.id || `msg_${Date.now()}`).replace(/^chatcmpl-/, ""),
|
||||
type: "message",
|
||||
role: "assistant",
|
||||
model: responseBody.model || "unknown",
|
||||
content,
|
||||
stop_reason: fromOpenAIFinish(choice.finish_reason, FORMATS.CLAUDE),
|
||||
stop_sequence: null,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate non-streaming response body from provider format → OpenAI format.
|
||||
*/
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
|
||||
if (targetFormat === sourceFormat || targetFormat === FORMATS.OPENAI) return responseBody;
|
||||
if (targetFormat === sourceFormat) return responseBody;
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
||||
return openAICompletionToClaudeMessage(responseBody);
|
||||
}
|
||||
if (targetFormat === FORMATS.OPENAI) return responseBody;
|
||||
|
||||
// Gemini / Antigravity
|
||||
if (targetFormat === FORMATS.GEMINI || targetFormat === FORMATS.ANTIGRAVITY || targetFormat === FORMATS.GEMINI_CLI || targetFormat === FORMATS.VERTEX) {
|
||||
@@ -185,6 +240,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
: responseBody;
|
||||
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
||||
|
||||
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
||||
if (translatedResponse?.choices?.[0]) {
|
||||
@@ -197,13 +253,17 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
|
||||
// Ensure OpenAI-required fields
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
}
|
||||
|
||||
// Strip Azure-specific fields
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
if (!isClaudeMessageResponse) {
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
}
|
||||
}
|
||||
|
||||
if (translatedResponse?.usage) {
|
||||
@@ -213,7 +273,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
if (translatedResponse?.choices) {
|
||||
if (!isClaudeMessageResponse && translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { saveRequestUsage, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { COLORS } from "../../utils/stream.js";
|
||||
import { canonicalizeUsage } from "../../utils/usageTracking.js";
|
||||
|
||||
const OPTIONAL_PARAMS = [
|
||||
"temperature", "top_p", "top_k",
|
||||
@@ -48,7 +49,8 @@ export function extractUsageFromResponse(responseBody) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount
|
||||
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
|
||||
};
|
||||
}
|
||||
|
||||
@@ -96,8 +98,14 @@ export function saveUsageStats({ provider, model, tokens, connectionId, apiKey,
|
||||
msg += `${COLORS.reset}`;
|
||||
console.log(msg);
|
||||
|
||||
<<<<<<< HEAD
|
||||
// Normalize to OpenAI token shape for storage (include all token types)
|
||||
const normalized = {
|
||||
=======
|
||||
// Canonicalize to one storage convention (prompt_tokens cache-inclusive) so
|
||||
// cached/cache-creation tokens survive to cost calc + stats. See canonicalizeUsage.
|
||||
const normalized = canonicalizeUsage(tokens) || {
|
||||
>>>>>>> 7f436e2792be4fa5a4d1c4d6b8e9bc85eaaa6a3d
|
||||
prompt_tokens: tokens.prompt_tokens ?? tokens.input_tokens ?? 0,
|
||||
completion_tokens: tokens.completion_tokens ?? tokens.output_tokens ?? 0,
|
||||
cache_read_input_tokens: cacheRead,
|
||||
|
||||
@@ -43,7 +43,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete }) {
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
@@ -52,12 +52,30 @@ export function handleStreamingResponse({ providerResponse, provider, model, sou
|
||||
});
|
||||
}
|
||||
|
||||
// Warn when upstream returns unexpected Content-Type for a streaming response.
|
||||
// This often means the provider returned an HTML error page or plain-text error
|
||||
// that the SSE transform stream would forward as garbage to the client.
|
||||
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
|
||||
// page), piping it through the SSE transform stream causes Next.js
|
||||
// "failed to pipe response" and crashes the chat router. Read the body,
|
||||
// pull a short human-readable message from the <title>, sanitize it, and
|
||||
// return a clean JSON error instead. The message is stripped of HTML tags
|
||||
// and clamped so untrusted upstream text never reaches the client verbatim
|
||||
// (the UI may render error.message as HTML).
|
||||
const upstreamContentType = (providerResponse.headers.get('content-type') || '').toLowerCase();
|
||||
if (upstreamContentType && !upstreamContentType.includes('text/event-stream') && !upstreamContentType.includes('application/json')) {
|
||||
console.warn('[STREAM] ' + provider + ' | ' + model + ' | unexpected Content-Type: ' + upstreamContentType);
|
||||
const bodyText = await providerResponse.text().catch(() => '');
|
||||
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
|
||||
const sanitizedTitle = (titleMatch?.[1] || '').replace(/<[^>]*>/g, '').replace(/[\r\n]+/g, ' ').trim().slice(0, 160);
|
||||
const shortMsg = sanitizedTitle
|
||||
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
response: new Response(JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }), {
|
||||
status,
|
||||
headers: { 'Content-Type': 'application/json', 'Access-Control-Allow-Origin': '*' },
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey });
|
||||
@@ -68,7 +86,6 @@ export function handleStreamingResponse({ providerResponse, provider, model, sou
|
||||
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
|
||||
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
|
||||
@@ -71,7 +71,7 @@ export function capabilitiesFromServiceKind(kind) {
|
||||
* otherwise mis-match. Only declare deltas vs DEFAULT.
|
||||
*/
|
||||
export const MODEL_CAPABILITIES = {
|
||||
// Claude 4.6/4.7/4.8 have 1M context + adaptive thinking (override generic claude pattern)
|
||||
// Claude 4.6/4.7/4.8 and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
|
||||
"claude-opus-4.6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-4.7": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-4-7": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
@@ -82,6 +82,10 @@ export const MODEL_CAPABILITIES = {
|
||||
"claude-opus-4-8-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-sonnet-4.6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-sonnet-4-6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-sonnet-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-sonnet-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-sonnet-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-sonnet-5-thinking-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
|
||||
// Gemini image-gen / OpenAI image / xai image variants
|
||||
"gpt-image-1": { imageOutput: true, tools: false },
|
||||
@@ -98,6 +102,15 @@ export const MODEL_CAPABILITIES = {
|
||||
* Provider-specific capability overrides. Keyed by provider alias/id.
|
||||
*/
|
||||
export const PROVIDER_CAPABILITIES = {
|
||||
// NVIDIA NIM is OpenAI-compatible → rejects MiniMax/GLM native `thinking` field.
|
||||
// Force openai reasoning_effort format for its reasoning models. #issue
|
||||
"nvidia": {
|
||||
"minimaxai/minimax-m2.7": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 131072 },
|
||||
"minimaxai/minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 131072 },
|
||||
"z-ai/glm-5.2": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 },
|
||||
"deepseek-ai/deepseek-v4-pro": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
},
|
||||
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
|
||||
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
|
||||
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
|
||||
@@ -177,12 +190,16 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*grok-3*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 131072 } },
|
||||
{ pattern: "*grok*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
|
||||
// ── Qwen (enable_thinking + thinking_budget; QwQ = thinking-only) ─
|
||||
// ── Qwen (3.5+ = native vision/video; coder & max = text-only; QwQ = thinking-only) ─
|
||||
{ pattern: "*qwen*vl*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
|
||||
{ pattern: "*qwen*max*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen*omni*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen*coder*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 } },
|
||||
{ pattern: "*qwen*max*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen3.5*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen3.6*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen3.7*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen*plus*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen*235b*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
|
||||
{ pattern: "*qwen*coder*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 } },
|
||||
{ pattern: "*qwq*", caps: { reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 131072 } },
|
||||
{ pattern: "*qwen*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
|
||||
|
||||
|
||||
@@ -279,7 +279,10 @@ export function calculateCostFromTokens(tokens, pricing) {
|
||||
|
||||
const inputTokens = tokens.prompt_tokens || tokens.input_tokens || 0;
|
||||
const cachedTokens = tokens.cached_tokens || tokens.cache_read_input_tokens || 0;
|
||||
const nonCachedInput = Math.max(0, inputTokens - cachedTokens);
|
||||
const cacheCreationTokens = tokens.cache_creation_input_tokens || 0;
|
||||
// prompt_tokens is cache-inclusive (see canonicalizeUsage): cached + cache_creation
|
||||
// are subsets, so subtract both to avoid charging them at the full input rate.
|
||||
const nonCachedInput = Math.max(0, inputTokens - cachedTokens - cacheCreationTokens);
|
||||
|
||||
cost += nonCachedInput * (pricing.input / 1000000);
|
||||
|
||||
@@ -295,7 +298,6 @@ export function calculateCostFromTokens(tokens, pricing) {
|
||||
cost += reasoningTokens * ((pricing.reasoning || pricing.output) / 1000000);
|
||||
}
|
||||
|
||||
const cacheCreationTokens = tokens.cache_creation_input_tokens || 0;
|
||||
if (cacheCreationTokens > 0) {
|
||||
cost += cacheCreationTokens * ((pricing.cache_creation || pricing.input) / 1000000);
|
||||
}
|
||||
|
||||
@@ -16,6 +16,7 @@ export default {
|
||||
transport: {
|
||||
baseUrl: "https://coding-intl.dashscope.aliyuncs.com/v1/chat/completions",
|
||||
headers: {},
|
||||
quirks: { preserveCacheControl: true },
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3.5-plus", name: "Qwen3.5 Plus" },
|
||||
|
||||
@@ -16,6 +16,7 @@ export default {
|
||||
transport: {
|
||||
baseUrl: "https://coding.dashscope.aliyuncs.com/v1/chat/completions",
|
||||
headers: {},
|
||||
quirks: { preserveCacheControl: true },
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3.5-plus", name: "Qwen3.5 Plus" },
|
||||
|
||||
57
open-sse/providers/registry/clinepass.js
Normal file
57
open-sse/providers/registry/clinepass.js
Normal file
@@ -0,0 +1,57 @@
|
||||
export default {
|
||||
id: "clinepass",
|
||||
priority: 85,
|
||||
alias: "clinepass",
|
||||
uiAlias: "clinepass",
|
||||
display: {
|
||||
name: "ClinePass",
|
||||
icon: "vpn_key",
|
||||
color: "#5B9BD5",
|
||||
textIcon: "CP",
|
||||
website: "https://cline.bot",
|
||||
notice: {
|
||||
signupUrl: "https://app.cline.bot",
|
||||
},
|
||||
},
|
||||
category: "oauth",
|
||||
authModes: ["oauth", "apikey"],
|
||||
hasOAuth: true,
|
||||
transport: {
|
||||
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
|
||||
headers: {
|
||||
"HTTP-Referer": "https://cline.bot",
|
||||
"X-Title": "Cline",
|
||||
},
|
||||
auth: {
|
||||
combined: true,
|
||||
header: "Authorization",
|
||||
scheme: "bearer",
|
||||
hooks: [
|
||||
"clineHeaders",
|
||||
],
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "cline-pass/glm-5.2", name: "GLM-5.2 (ClinePass)" },
|
||||
{ id: "cline-pass/kimi-k2.7-code", name: "Kimi K2.7 Code (ClinePass)" },
|
||||
{ id: "cline-pass/kimi-k2.6", name: "Kimi K2.6 (ClinePass)" },
|
||||
{ id: "cline-pass/deepseek-v4-pro", name: "DeepSeek V4 Pro (ClinePass)" },
|
||||
{ id: "cline-pass/deepseek-v4-flash", name: "DeepSeek V4 Flash (ClinePass)" },
|
||||
{ id: "cline-pass/mimo-v2.5", name: "MiMo-V2.5 (ClinePass)" },
|
||||
{ id: "cline-pass/mimo-v2.5-pro", name: "MiMo-V2.5-Pro (ClinePass)" },
|
||||
{ id: "cline-pass/minimax-m3", name: "MiniMax M3 (ClinePass)" },
|
||||
{ id: "cline-pass/qwen3.7-max", name: "Qwen3.7 Max (ClinePass)" },
|
||||
{ id: "cline-pass/qwen3.7-plus", name: "Qwen3.7 Plus (ClinePass)" },
|
||||
],
|
||||
oauth: {
|
||||
appBaseUrl: "https://app.cline.bot",
|
||||
apiBaseUrl: "https://api.cline.bot",
|
||||
authorizeUrl: "https://api.cline.bot/api/v1/auth/authorize",
|
||||
tokenUrl: "https://api.cline.bot/api/v1/auth/token",
|
||||
refreshUrl: "https://api.cline.bot/api/v1/auth/refresh",
|
||||
},
|
||||
thinkingConfig: {
|
||||
options: ["auto", "on", "off"],
|
||||
defaultMode: "auto",
|
||||
},
|
||||
};
|
||||
@@ -40,6 +40,7 @@ export default {
|
||||
},
|
||||
usage: {
|
||||
url: "https://chatgpt.com/backend-api/wham/usage",
|
||||
resetCreditsUrl: "https://chatgpt.com/backend-api/wham/rate-limit-reset-credits",
|
||||
resetCreditsConsumeUrl: "https://chatgpt.com/backend-api/wham/rate-limit-reset-credits/consume",
|
||||
},
|
||||
},
|
||||
|
||||
@@ -15,85 +15,87 @@ import p12 from "./cerebras.js";
|
||||
import p13 from "./chutes.js";
|
||||
import p14 from "./claude.js";
|
||||
import p15 from "./cline.js";
|
||||
import p16 from "./cloudflare-ai.js";
|
||||
import p17 from "./codebuddy-cn.js";
|
||||
import p18 from "./codex.js";
|
||||
import p19 from "./cohere.js";
|
||||
import p20 from "./comfyui.js";
|
||||
import p21 from "./commandcode.js";
|
||||
import p22 from "./coqui.js";
|
||||
import p23 from "./cursor.js";
|
||||
import p24 from "./deepgram.js";
|
||||
import p25 from "./deepseek.js";
|
||||
import p26 from "./edge-tts.js";
|
||||
import p27 from "./elevenlabs.js";
|
||||
import p28 from "./exa.js";
|
||||
import p29 from "./fal-ai.js";
|
||||
import p30 from "./firecrawl.js";
|
||||
import p31 from "./fireworks.js";
|
||||
import p32 from "./gemini-cli.js";
|
||||
import p33 from "./gemini.js";
|
||||
import p34 from "./github.js";
|
||||
import p35 from "./gitlab.js";
|
||||
import p36 from "./glm-cn.js";
|
||||
import p37 from "./glm.js";
|
||||
import p38 from "./google-pse.js";
|
||||
import p39 from "./google-tts.js";
|
||||
import p40 from "./grok-web.js";
|
||||
import p41 from "./groq.js";
|
||||
import p42 from "./huggingface.js";
|
||||
import p43 from "./hyperbolic.js";
|
||||
import p44 from "./iflow.js";
|
||||
import p45 from "./inworld.js";
|
||||
import p46 from "./jina-ai.js";
|
||||
import p47 from "./jina-reader.js";
|
||||
import p48 from "./kilocode.js";
|
||||
import p49 from "./kimi-coding.js";
|
||||
import p50 from "./kimi.js";
|
||||
import p51 from "./kiro.js";
|
||||
import p52 from "./linkup.js";
|
||||
import p53 from "./local-device.js";
|
||||
import p54 from "./mimo-free.js";
|
||||
import p55 from "./minimax-cn.js";
|
||||
import p56 from "./minimax.js";
|
||||
import p57 from "./mistral.js";
|
||||
import p58 from "./mmf.js";
|
||||
import p59 from "./nanobanana.js";
|
||||
import p60 from "./nebius.js";
|
||||
import p61 from "./nvidia.js";
|
||||
import p62 from "./ollama-local.js";
|
||||
import p63 from "./ollama.js";
|
||||
import p64 from "./openai.js";
|
||||
import p65 from "./opencode-go.js";
|
||||
import p66 from "./opencode.js";
|
||||
import p67 from "./openrouter.js";
|
||||
import p68 from "./perplexity-web.js";
|
||||
import p69 from "./perplexity.js";
|
||||
import p70 from "./playht.js";
|
||||
import p71 from "./qoder.js";
|
||||
import p72 from "./qwen.js";
|
||||
import p73 from "./recraft.js";
|
||||
import p74 from "./runwayml.js";
|
||||
import p75 from "./sdwebui.js";
|
||||
import p76 from "./searchapi.js";
|
||||
import p77 from "./searxng.js";
|
||||
import p78 from "./serper.js";
|
||||
import p79 from "./siliconflow.js";
|
||||
import p80 from "./stability-ai.js";
|
||||
import p81 from "./tavily.js";
|
||||
import p82 from "./together.js";
|
||||
import p83 from "./topaz.js";
|
||||
import p84 from "./tortoise.js";
|
||||
import p85 from "./venice.js";
|
||||
import p86 from "./vercel-ai-gateway.js";
|
||||
import p87 from "./vertex-partner.js";
|
||||
import p88 from "./vertex.js";
|
||||
import p89 from "./volcengine-ark.js";
|
||||
import p90 from "./voyage-ai.js";
|
||||
import p91 from "./xai.js";
|
||||
import p92 from "./xiaomi-mimo.js";
|
||||
import p93 from "./xiaomi-tokenplan.js";
|
||||
import p94 from "./youcom.js";
|
||||
import p16 from "./clinepass.js";
|
||||
import p17 from "./cloudflare-ai.js";
|
||||
import p18 from "./codebuddy-cn.js";
|
||||
import p19 from "./codex.js";
|
||||
import p20 from "./cohere.js";
|
||||
import p21 from "./comfyui.js";
|
||||
import p22 from "./commandcode.js";
|
||||
import p23 from "./coqui.js";
|
||||
import p24 from "./cursor.js";
|
||||
import p25 from "./deepgram.js";
|
||||
import p26 from "./deepseek.js";
|
||||
import p27 from "./edge-tts.js";
|
||||
import p28 from "./elevenlabs.js";
|
||||
import p29 from "./exa.js";
|
||||
import p30 from "./fal-ai.js";
|
||||
import p31 from "./firecrawl.js";
|
||||
import p32 from "./fireworks.js";
|
||||
import p33 from "./gemini-cli.js";
|
||||
import p34 from "./gemini.js";
|
||||
import p35 from "./github.js";
|
||||
import p36 from "./gitlab.js";
|
||||
import p37 from "./glm-cn.js";
|
||||
import p38 from "./glm.js";
|
||||
import p39 from "./google-pse.js";
|
||||
import p40 from "./google-tts.js";
|
||||
import p41 from "./grok-web.js";
|
||||
import p42 from "./groq.js";
|
||||
import p43 from "./huggingface.js";
|
||||
import p44 from "./hyperbolic.js";
|
||||
import p45 from "./iflow.js";
|
||||
import p46 from "./inworld.js";
|
||||
import p47 from "./jina-ai.js";
|
||||
import p48 from "./jina-reader.js";
|
||||
import p49 from "./kilocode.js";
|
||||
import p50 from "./kimchi.js";
|
||||
import p51 from "./kimi-coding.js";
|
||||
import p52 from "./kimi.js";
|
||||
import p53 from "./kiro.js";
|
||||
import p54 from "./linkup.js";
|
||||
import p55 from "./local-device.js";
|
||||
import p56 from "./mimo-free.js";
|
||||
import p57 from "./minimax-cn.js";
|
||||
import p58 from "./minimax.js";
|
||||
import p59 from "./mistral.js";
|
||||
import p60 from "./mmf.js";
|
||||
import p61 from "./nanobanana.js";
|
||||
import p62 from "./nebius.js";
|
||||
import p63 from "./nvidia.js";
|
||||
import p64 from "./ollama-local.js";
|
||||
import p65 from "./ollama.js";
|
||||
import p66 from "./openai.js";
|
||||
import p67 from "./opencode-go.js";
|
||||
import p68 from "./opencode.js";
|
||||
import p69 from "./openrouter.js";
|
||||
import p70 from "./perplexity-web.js";
|
||||
import p71 from "./perplexity.js";
|
||||
import p72 from "./playht.js";
|
||||
import p73 from "./qoder.js";
|
||||
import p74 from "./qwen.js";
|
||||
import p75 from "./recraft.js";
|
||||
import p76 from "./runwayml.js";
|
||||
import p77 from "./sdwebui.js";
|
||||
import p78 from "./searchapi.js";
|
||||
import p79 from "./searxng.js";
|
||||
import p80 from "./serper.js";
|
||||
import p81 from "./siliconflow.js";
|
||||
import p82 from "./stability-ai.js";
|
||||
import p83 from "./tavily.js";
|
||||
import p84 from "./together.js";
|
||||
import p85 from "./topaz.js";
|
||||
import p86 from "./tortoise.js";
|
||||
import p87 from "./venice.js";
|
||||
import p88 from "./vercel-ai-gateway.js";
|
||||
import p89 from "./vertex-partner.js";
|
||||
import p90 from "./vertex.js";
|
||||
import p91 from "./volcengine-ark.js";
|
||||
import p92 from "./voyage-ai.js";
|
||||
import p93 from "./xai.js";
|
||||
import p94 from "./xiaomi-mimo.js";
|
||||
import p95 from "./xiaomi-tokenplan.js";
|
||||
import p96 from "./youcom.js";
|
||||
|
||||
export default [
|
||||
p0,
|
||||
@@ -190,5 +192,7 @@ export default [
|
||||
p91,
|
||||
p92,
|
||||
p93,
|
||||
p94
|
||||
p94,
|
||||
p95,
|
||||
p96
|
||||
];
|
||||
|
||||
@@ -36,6 +36,13 @@ export default {
|
||||
{ id: "deepseek/deepseek-chat", name: "DeepSeek Chat" },
|
||||
{ id: "deepseek/deepseek-reasoner", name: "DeepSeek Reasoner" },
|
||||
],
|
||||
// Kilo Code proxies the OpenRouter catalog (334 models at time of writing),
|
||||
// so the hardcoded list above is only a fallback. Surfacing the full catalog
|
||||
// requires a fetcher + passthroughModels, matching how openrouter.js is set up.
|
||||
// Without these, only the 8 hardcoded models appear in the combo model picker,
|
||||
// hiding dynamic models like cohere/north-mini-code:free and poolside/laguna-m.1:free.
|
||||
modelsFetcher: { url: "https://api.kilo.ai/api/gateway/models", type: "openrouter-free" },
|
||||
passthroughModels: true,
|
||||
oauth: {
|
||||
apiBaseUrl: "https://api.kilo.ai",
|
||||
initiateUrl: "https://api.kilo.ai/api/device-auth/codes",
|
||||
|
||||
49
open-sse/providers/registry/kimchi.js
Normal file
49
open-sse/providers/registry/kimchi.js
Normal file
@@ -0,0 +1,49 @@
|
||||
export default {
|
||||
id: "kimchi",
|
||||
priority: 95,
|
||||
alias: "kimchi",
|
||||
uiAlias: "kimchi",
|
||||
display: {
|
||||
name: "Kimchi",
|
||||
icon: "restaurant",
|
||||
color: "#FF521D",
|
||||
textIcon: "KC",
|
||||
website: "https://kimchi.dev",
|
||||
notice: {
|
||||
signupUrl: "https://app.kimchi.dev",
|
||||
},
|
||||
},
|
||||
category: "oauth",
|
||||
authModes: ["oauth"],
|
||||
hasOAuth: true,
|
||||
transport: {
|
||||
baseUrl: "https://llm.kimchi.dev/openai/v1/chat/completions",
|
||||
format: "openai",
|
||||
headers: {
|
||||
"User-Agent": "kimchi/0.1.50",
|
||||
},
|
||||
auth: {
|
||||
combined: true,
|
||||
header: "Authorization",
|
||||
scheme: "bearer",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "minimax-m3", name: "MiniMax-M3" },
|
||||
{ id: "kimi-k2.7", name: "Kimi-K2.7" },
|
||||
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
|
||||
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
|
||||
{ id: "nemotron-3-ultra-fp4", name: "Nemotron 3 Ultra FP4" },
|
||||
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
|
||||
{ id: "claude-opus-4-6", name: "Claude Opus 4.6" },
|
||||
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6" },
|
||||
],
|
||||
serviceKinds: ["llm", "imageToText"],
|
||||
oauth: {
|
||||
webAppUrl: "https://app.kimchi.dev",
|
||||
validationUrl: "https://api.cast.ai/v1/llm/openai/supported-providers",
|
||||
userInfoUrl: "https://app.kimchi.dev/api/v1/me",
|
||||
modelsUrl: "https://llm.kimchi.dev/v1/models/metadata?include_in_cli=true",
|
||||
},
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -42,16 +42,20 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
|
||||
{ id: "claude-haiku-4.5", name: "Claude Haiku 4.5" },
|
||||
{ id: "deepseek-3.2", name: "DeepSeek 3.2", strip: ["image","audio"] },
|
||||
{ id: "qwen3-coder-next", name: "Qwen3 Coder Next", strip: ["image","audio"] },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
{ id: "MiniMax-M2.5", name: "MiniMax M2.5" },
|
||||
{ id: "claude-sonnet-5-thinking", name: "Claude Sonnet 5 (Thinking)" },
|
||||
{ id: "claude-sonnet-4.5-thinking", name: "Claude Sonnet 4.5 (Thinking)" },
|
||||
{ id: "claude-haiku-4.5-thinking", name: "Claude Haiku 4.5 (Thinking)" },
|
||||
{ id: "claude-sonnet-5-agentic", name: "Claude Sonnet 5 (Agentic)" },
|
||||
{ id: "claude-sonnet-4.5-agentic", name: "Claude Sonnet 4.5 (Agentic)" },
|
||||
{ id: "claude-haiku-4.5-agentic", name: "Claude Haiku 4.5 (Agentic)" },
|
||||
{ id: "claude-sonnet-5-thinking-agentic", name: "Claude Sonnet 5 (Thinking + Agentic)" },
|
||||
{ id: "claude-sonnet-4.5-thinking-agentic", name: "Claude Sonnet 4.5 (Thinking + Agentic)" },
|
||||
{ id: "claude-haiku-4.5-thinking-agentic", name: "Claude Haiku 4.5 (Thinking + Agentic)" },
|
||||
],
|
||||
|
||||
@@ -20,8 +20,13 @@ export default {
|
||||
validateUrl: "https://integrate.api.nvidia.com/v1/models",
|
||||
},
|
||||
models: [
|
||||
{ id: "minimaxai/minimax-m2.7", name: "Minimax M2.7" },
|
||||
{ id: "z-ai/glm4.7", name: "GLM 4.7" },
|
||||
{ id: "minimaxai/minimax-m2.7", name: "MiniMax M2.7" },
|
||||
{ id: "minimaxai/minimax-m3", name: "MiniMax M3" },
|
||||
{ id: "z-ai/glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "deepseek-ai/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek-ai/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
|
||||
{ id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra" },
|
||||
{ id: "nvidia/nv-embedqa-e5-v5", name: "NV EmbedQA E5 v5", kind: "embedding" },
|
||||
{ id: "nvidia/parakeet-ctc-1.1b-asr", name: "Parakeet CTC 1.1B", params: ["language"], kind: "stt" },
|
||||
{ id: "fastpitch", name: "FastPitch", kind: "tts" },
|
||||
|
||||
@@ -21,6 +21,11 @@ export default {
|
||||
},
|
||||
category: "apikey",
|
||||
hasProviderSpecificData: true,
|
||||
regions: [
|
||||
{ id: "sgp", label: "Singapore (新加坡)" },
|
||||
{ id: "cn", label: "China (中国大陆)" },
|
||||
{ id: "ams", label: "Amsterdam (阿姆斯特丹)" },
|
||||
],
|
||||
defaultRegion: "sgp",
|
||||
transport: {
|
||||
baseUrl: "https://token-plan-sgp.xiaomimimo.com/v1/chat/completions",
|
||||
|
||||
@@ -73,6 +73,14 @@ function maskEndpoint(endpoint) {
|
||||
}
|
||||
}
|
||||
|
||||
function hasUnsafeResponsesInputForCompression(body) {
|
||||
if (!Array.isArray(body?.input)) return false;
|
||||
return body.input.some((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return false;
|
||||
return typeof item.type === "string" && item.type !== "message";
|
||||
});
|
||||
}
|
||||
|
||||
// POST messages to Headroom /v1/compress; returns compressed messages + stats or null.
|
||||
async function callCompress(url, messages, model, timeoutMs, compressUserMessages, diagnostics) {
|
||||
const endpoint = buildCompressEndpoint(url);
|
||||
@@ -143,6 +151,10 @@ export async function compressWithHeadroom(body, { enabled, url, model, format,
|
||||
// messages. Translate input -> OpenAI -> compress -> translate back to input so
|
||||
// body.input keeps the Responses contract (the proxy only understands OpenAI). (#1998)
|
||||
if (format === "openai-responses") {
|
||||
if (hasUnsafeResponsesInputForCompression(body)) {
|
||||
setDiagnostic(diagnostics, "skipped: openai-responses tool/reasoning input is not safe to compress");
|
||||
return null;
|
||||
}
|
||||
const oai = openaiResponsesToOpenAIRequest(model, body, false);
|
||||
if (!Array.isArray(oai?.messages)) return null;
|
||||
const data = await callCompress(url, oai.messages, model, timeoutMs, compressUserMessages, diagnostics || {});
|
||||
|
||||
63
open-sse/services/clinepassModels.js
Normal file
63
open-sse/services/clinepassModels.js
Normal file
@@ -0,0 +1,63 @@
|
||||
import { buildClineHeaders } from "../shared/clineAuth.js";
|
||||
|
||||
const CLINEPASS_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/models";
|
||||
const FETCH_TIMEOUT_MS = 5000;
|
||||
|
||||
/**
|
||||
* Build request headers for the ClinePass /models endpoint (Cline's upstream API).
|
||||
* - API keys are sent as plain Bearer tokens.
|
||||
* - OAuth access tokens must carry the WorkOS `workos:` prefix (handled by buildClineHeaders).
|
||||
*/
|
||||
function buildModelListHeaders(token, isApiKey) {
|
||||
if (isApiKey) {
|
||||
return {
|
||||
Accept: "application/json",
|
||||
Authorization: `Bearer ${token}`,
|
||||
};
|
||||
}
|
||||
return buildClineHeaders(token, { Accept: "application/json" });
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch ClinePass live model catalog from Cline's /models endpoint.
|
||||
*
|
||||
* @param {object} credentials - Connection credentials ({ accessToken, apiKey })
|
||||
* @returns {Promise<{ models: { id: string, name: string }[] } | null>}
|
||||
*/
|
||||
export async function resolveClinepassModels(credentials) {
|
||||
const isApiKey = Boolean(credentials?.apiKey);
|
||||
const token = isApiKey ? credentials.apiKey : credentials?.accessToken;
|
||||
if (!token) return null;
|
||||
|
||||
const controller = new AbortController();
|
||||
const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
|
||||
|
||||
try {
|
||||
const headers = buildModelListHeaders(token, isApiKey);
|
||||
|
||||
const response = await fetch(CLINEPASS_MODELS_ENDPOINT, {
|
||||
method: "GET",
|
||||
headers,
|
||||
signal: controller.signal,
|
||||
});
|
||||
|
||||
if (!response.ok) return null;
|
||||
|
||||
const json = await response.json();
|
||||
const rawList = Array.isArray(json) ? json : json?.data;
|
||||
if (!Array.isArray(rawList)) return null;
|
||||
|
||||
const models = rawList
|
||||
.filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/"))
|
||||
.map((m) => ({
|
||||
id: m.id,
|
||||
name: m.name || m.id,
|
||||
}));
|
||||
|
||||
return models.length ? { models } : null;
|
||||
} catch {
|
||||
return null;
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
176
open-sse/services/kimchiModels.js
Normal file
176
open-sse/services/kimchiModels.js
Normal file
@@ -0,0 +1,176 @@
|
||||
import { createHash } from "crypto";
|
||||
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
|
||||
export const KIMCHI_API = "https://llm.kimchi.dev";
|
||||
export const KIMCHI_USER_AGENT = "kimchi/0.1.40";
|
||||
|
||||
const FETCH_TIMEOUT_MS = 20_000;
|
||||
const CACHE_TTL_MS = 5 * 60 * 1000;
|
||||
const RETRYABLE_STATUSES = new Set([429, 500, 502, 503, 504]);
|
||||
|
||||
/** @type {Map<string, { expiresAt: number, models: object[], rawModels: object[] }>} */
|
||||
const catalogCache = new Map();
|
||||
/** @type {Map<string, object>} */
|
||||
const metadataByModelId = new Map();
|
||||
|
||||
function normalizeKimchiEndpoint(endpoint) {
|
||||
const raw = typeof endpoint === "string" ? endpoint.trim() : "";
|
||||
return (raw || KIMCHI_API).replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
export function buildKimchiModelsUrl(endpoint) {
|
||||
return `${normalizeKimchiEndpoint(endpoint)}/v1/models/metadata?include_in_cli=true`;
|
||||
}
|
||||
|
||||
function readToken(credentials) {
|
||||
return (
|
||||
credentials?.accessToken
|
||||
|| credentials?.apiKey
|
||||
|| credentials?.providerSpecificData?.apiKey
|
||||
|| null
|
||||
);
|
||||
}
|
||||
|
||||
function cacheKey(credentials, endpoint) {
|
||||
const psd = credentials?.providerSpecificData || {};
|
||||
const seed = psd.userId || psd.username || credentials?.refreshToken || readToken(credentials) || "anonymous";
|
||||
return createHash("sha256")
|
||||
.update(`kimchi:${normalizeKimchiEndpoint(endpoint)}:${seed}`)
|
||||
.digest("hex");
|
||||
}
|
||||
|
||||
function toModelKind(inputModalities) {
|
||||
return Array.isArray(inputModalities) && inputModalities.includes("image")
|
||||
? "imageToText"
|
||||
: "llm";
|
||||
}
|
||||
|
||||
export function normalizeKimchiModel(item) {
|
||||
if (!item || typeof item !== "object") return null;
|
||||
const id = item.slug || item.id || item.model || item.name;
|
||||
if (typeof id !== "string" || id.trim() === "") return null;
|
||||
|
||||
const inputModalities = Array.isArray(item.input_modalities)
|
||||
? item.input_modalities.filter((value) => value === "text" || value === "image")
|
||||
: [];
|
||||
const limits = item.limits && typeof item.limits === "object" ? item.limits : {};
|
||||
const contextLength = Number(limits.context_window || item.contextLength || item.context_length) || undefined;
|
||||
const maxOutputTokens = Number(limits.max_output_tokens || item.maxOutputTokens || item.max_output_tokens) || undefined;
|
||||
const upstreamProvider = typeof item.provider === "string" ? item.provider : "";
|
||||
const reasoning = item.reasoning === true;
|
||||
const kind = toModelKind(inputModalities);
|
||||
|
||||
const model = {
|
||||
...item,
|
||||
id: id.trim(),
|
||||
name: String(item.display_name || item.displayName || item.name || id).trim(),
|
||||
provider: upstreamProvider,
|
||||
upstreamProvider,
|
||||
reasoning,
|
||||
inputModalities,
|
||||
kind,
|
||||
type: kind,
|
||||
capabilities: {
|
||||
vision: inputModalities.includes("image"),
|
||||
reasoning,
|
||||
...(contextLength ? { contextWindow: contextLength } : {}),
|
||||
...(maxOutputTokens ? { maxOutput: maxOutputTokens } : {}),
|
||||
...(upstreamProvider ? { upstreamProvider } : {}),
|
||||
},
|
||||
...(contextLength ? { contextLength } : {}),
|
||||
...(maxOutputTokens ? { maxOutputTokens } : {}),
|
||||
};
|
||||
|
||||
if (upstreamProvider === "anthropic") {
|
||||
model.compat = { supportsReasoningEffort: false, cacheControlFormat: "anthropic" };
|
||||
}
|
||||
|
||||
return model;
|
||||
}
|
||||
|
||||
function rememberModels(models) {
|
||||
for (const model of models || []) {
|
||||
if (!model?.id) continue;
|
||||
metadataByModelId.set(model.id, model);
|
||||
metadataByModelId.set(model.id.toLowerCase(), model);
|
||||
}
|
||||
}
|
||||
|
||||
export function getCachedKimchiModelMetadata(modelId) {
|
||||
if (typeof modelId !== "string" || modelId.trim() === "") return null;
|
||||
const raw = modelId.includes("/") ? modelId.split("/").pop() : modelId;
|
||||
return metadataByModelId.get(raw) || metadataByModelId.get(raw.toLowerCase()) || null;
|
||||
}
|
||||
|
||||
async function fetchKimchiCatalogRaw(token, endpoint, options = {}) {
|
||||
const url = buildKimchiModelsUrl(endpoint);
|
||||
const controller = new AbortController();
|
||||
const timeout = setTimeout(() => controller.abort(new Error("Kimchi models fetch timeout")), FETCH_TIMEOUT_MS);
|
||||
const signal = options.signal
|
||||
? AbortSignal.any([options.signal, controller.signal])
|
||||
: controller.signal;
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(url, {
|
||||
method: "GET",
|
||||
headers: {
|
||||
"Accept": "application/json",
|
||||
"Authorization": `Bearer ${token}`,
|
||||
"User-Agent": KIMCHI_USER_AGENT,
|
||||
},
|
||||
cache: "no-store",
|
||||
signal,
|
||||
}, options.proxyOptions || null);
|
||||
|
||||
if (!response.ok) {
|
||||
const error = new Error(`Kimchi models ${response.status}: ${response.statusText}`);
|
||||
error.status = response.status;
|
||||
error.retryable = RETRYABLE_STATUSES.has(response.status);
|
||||
throw error;
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
return Array.isArray(data?.models) ? data.models : [];
|
||||
} finally {
|
||||
clearTimeout(timeout);
|
||||
}
|
||||
}
|
||||
|
||||
export async function resolveKimchiModels(credentials, options = {}) {
|
||||
const token = readToken(credentials);
|
||||
if (!token) return null;
|
||||
|
||||
const endpoint = credentials?.providerSpecificData?.kimchiEndpoint || options.endpoint || KIMCHI_API;
|
||||
const key = cacheKey(credentials, endpoint);
|
||||
const now = Date.now();
|
||||
if (!options.forceRefresh) {
|
||||
const cached = catalogCache.get(key);
|
||||
if (cached && cached.expiresAt > now) return cached;
|
||||
}
|
||||
|
||||
let rawModels;
|
||||
try {
|
||||
rawModels = await fetchKimchiCatalogRaw(token, endpoint, options);
|
||||
} catch (error) {
|
||||
options.log?.warn?.("KIMCHI_MODELS", error.message);
|
||||
return null;
|
||||
}
|
||||
|
||||
const models = rawModels.map(normalizeKimchiModel).filter(Boolean);
|
||||
if (models.length === 0) return null;
|
||||
|
||||
rememberModels(models);
|
||||
const entry = {
|
||||
expiresAt: Date.now() + CACHE_TTL_MS,
|
||||
models,
|
||||
rawModels,
|
||||
};
|
||||
catalogCache.set(key, entry);
|
||||
return entry;
|
||||
}
|
||||
|
||||
export function clearKimchiCatalog() {
|
||||
catalogCache.clear();
|
||||
metadataByModelId.clear();
|
||||
}
|
||||
@@ -5,9 +5,9 @@
|
||||
import { getGitHubUsage } from "./usage/github.js";
|
||||
import { getGeminiUsage, getAntigravityUsage } from "./usage/google.js";
|
||||
import { getClaudeUsage } from "./usage/claude.js";
|
||||
import { getCodexUsage, consumeCodexRateLimitResetCredit } from "./usage/codex.js";
|
||||
import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits } from "./usage/codex.js";
|
||||
|
||||
export { consumeCodexRateLimitResetCredit };
|
||||
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
|
||||
import { getKiroUsage } from "./usage/kiro.js";
|
||||
import { getMiniMaxUsage } from "./usage/minimax.js";
|
||||
import { getCodeBuddyCnUsage } from "./usage/codebuddy-cn.js";
|
||||
|
||||
@@ -109,15 +109,22 @@ export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificD
|
||||
total: num(acc.CycleCapacitySizePrecise, acc.CycleCapacitySize),
|
||||
resetAt: parseResetTime(acc.CycleEndTime),
|
||||
unlimited: false,
|
||||
// Recurring allowance: the CycleEndTime is the next refresh, not the
|
||||
// final expiry. The UI must show "Resets in", not "Expires in".
|
||||
recurring: true,
|
||||
};
|
||||
});
|
||||
// Bonus packs: use the lifetime Capacity balance; resetAt is the expiry.
|
||||
// These are one-shot credits (CycleEndTime == DeductionEndTime), so they
|
||||
// never replenish — mark recurring:false so the UI shows "Expires in"
|
||||
// instead of implying a monthly refill.
|
||||
bonuses.forEach((acc, i) => {
|
||||
quotas[`Bonus Pack ${i + 1}`] = {
|
||||
used: num(acc.CapacityUsedPrecise, acc.CapacityUsed),
|
||||
total: num(acc.CapacitySizePrecise, acc.CapacitySize),
|
||||
resetAt: parseResetTime(acc.CycleEndTime),
|
||||
unlimited: false,
|
||||
recurring: false,
|
||||
};
|
||||
});
|
||||
|
||||
|
||||
@@ -8,9 +8,23 @@ import { U, parseResetTime, toFiniteNumber } from "./shared.js";
|
||||
// Codex (OpenAI) API config
|
||||
const CODEX_CONFIG = {
|
||||
usageUrl: U("codex").url,
|
||||
resetCreditsUrl: U("codex").resetCreditsUrl,
|
||||
resetCreditsConsumeUrl: U("codex").resetCreditsConsumeUrl,
|
||||
};
|
||||
|
||||
function toIsoDate(value) {
|
||||
if (!value) return null;
|
||||
const date = value instanceof Date
|
||||
? value
|
||||
: new Date(typeof value === "number" && value < 1e12 ? value * 1000 : value);
|
||||
const time = date.getTime();
|
||||
return Number.isFinite(time) ? date.toISOString() : null;
|
||||
}
|
||||
|
||||
function getCodexAccountId(providerSpecificData) {
|
||||
return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null;
|
||||
}
|
||||
|
||||
function getCodexRateLimitBody(snapshot) {
|
||||
if (!snapshot || typeof snapshot !== "object" || Array.isArray(snapshot)) return null;
|
||||
return snapshot.rate_limit && typeof snapshot.rate_limit === "object"
|
||||
@@ -101,6 +115,48 @@ export async function getCodexUsage(accessToken, proxyOptions = null) {
|
||||
}
|
||||
}
|
||||
|
||||
export async function getCodexRateLimitResetCredits(accessToken, proxyOptions = null, providerSpecificData = null) {
|
||||
if (!accessToken) {
|
||||
throw new Error("No Codex access token available. Please re-authorize the connection.");
|
||||
}
|
||||
|
||||
const accountId = getCodexAccountId(providerSpecificData);
|
||||
const headers = {
|
||||
"Authorization": `Bearer ${accessToken}`,
|
||||
"Accept": "application/json",
|
||||
"OpenAI-Beta": "codex-1",
|
||||
"originator": "codex_cli_rs",
|
||||
};
|
||||
if (accountId) headers["ChatGPT-Account-ID"] = accountId;
|
||||
|
||||
const response = await proxyAwareFetch(CODEX_CONFIG.resetCreditsUrl, {
|
||||
method: "GET",
|
||||
headers,
|
||||
}, proxyOptions);
|
||||
|
||||
let data = null;
|
||||
try {
|
||||
data = await response.json();
|
||||
} catch {
|
||||
data = null;
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`;
|
||||
throw new Error(message);
|
||||
}
|
||||
|
||||
const credits = Array.isArray(data?.credits) ? data.credits : [];
|
||||
return {
|
||||
availableCount: Math.max(0, toFiniteNumber(data?.available_count ?? data?.availableCount, 0)),
|
||||
credits: credits.map((credit) => ({
|
||||
status: String(credit?.status || "unknown"),
|
||||
grantedAt: toIsoDate(credit?.granted_at ?? credit?.grantedAt),
|
||||
expiresAt: toIsoDate(credit?.expires_at ?? credit?.expiresAt),
|
||||
})),
|
||||
};
|
||||
}
|
||||
|
||||
// Consume one Codex rate-limit reset credit (irreversible, spends 1 credit)
|
||||
export async function consumeCodexRateLimitResetCredit(accessToken, redeemRequestId, proxyOptions = null) {
|
||||
if (!accessToken) {
|
||||
|
||||
@@ -39,7 +39,16 @@ const USAGE_EXTRACTORS = {
|
||||
},
|
||||
kiro(raw) {
|
||||
const input = n(raw.inputTokens), output = n(raw.outputTokens);
|
||||
return { promptTokens: input, completionTokens: output, totalTokens: input + output };
|
||||
// ponytail: Amazon Q (Kiro upstream) does not expose cache fields today,
|
||||
// but pass through any cache_read/cache_creation/cached_tokens if the
|
||||
// event shape grows them later so cost tracking keeps working without
|
||||
// a second pass.
|
||||
const cached = n(raw.cache_read_input_tokens) || n(raw.cachedTokens) || n(raw.cached_tokens);
|
||||
const cacheCreation = n(raw.cache_creation_input_tokens);
|
||||
const out = { promptTokens: input, completionTokens: output, totalTokens: input + output };
|
||||
if (cached > 0) out.cachedTokens = cached;
|
||||
if (cacheCreation > 0) out.cacheCreationTokens = cacheCreation;
|
||||
return out;
|
||||
},
|
||||
ollama(raw) {
|
||||
const input = n(raw.prompt_eval_count), output = n(raw.eval_count);
|
||||
|
||||
@@ -150,6 +150,33 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
|
||||
// so foreign signatures leak into history and Anthropic rejects them).
|
||||
const thinkingEnabled = body.thinking?.type === "enabled";
|
||||
if (Array.isArray(body.messages)) {
|
||||
for (const msg of body.messages) {
|
||||
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
|
||||
let hasToolUse = false;
|
||||
let hasKeptThinking = false;
|
||||
const kept = [];
|
||||
for (const block of msg.content) {
|
||||
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
|
||||
if (isValidClaudeSignature(block.signature)) {
|
||||
hasKeptThinking = true;
|
||||
kept.push(block);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
|
||||
kept.push(block);
|
||||
}
|
||||
msg.content = kept;
|
||||
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
|
||||
msg.content.unshift(buildThinkingPlaceholder("claude"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,8 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
|
||||
"default", "examples",
|
||||
// JSON Schema meta keywords
|
||||
"$schema", "$defs", "definitions", "const", "$ref", "$comment",
|
||||
// Annotation keywords (rejected by Gemini/Antigravity - e.g. MCP tool schemas set these)
|
||||
"deprecated", "readOnly", "writeOnly",
|
||||
// Object validation keywords (not supported)
|
||||
"additionalProperties", "propertyNames", "patternProperties", "enumDescriptions",
|
||||
// Complex schema keywords (handled by flattenAnyOfOneOf/mergeAllOf)
|
||||
@@ -19,7 +21,7 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
|
||||
// Dependency keywords (not supported)
|
||||
"dependencies", "dependentSchemas", "dependentRequired",
|
||||
// Other unsupported keywords
|
||||
"title", "optional", "if", "then", "else", "contentMediaType", "contentEncoding",
|
||||
"title", "optional", "deprecated", "if", "then", "else", "contentMediaType", "contentEncoding",
|
||||
// UI/Styling properties (from Cursor tools - NOT JSON Schema standard)
|
||||
"cornerRadius", "fillColor", "fontFamily", "fontSize", "fontWeight",
|
||||
"gap", "padding", "strokeColor", "strokeThickness", "textColor"
|
||||
|
||||
@@ -6,42 +6,46 @@ export { VALID_OPENAI_CONTENT_TYPES, VALID_OPENAI_MESSAGE_TYPES };
|
||||
|
||||
// Filter messages to OpenAI standard format
|
||||
// Remove: thinking, redacted_thinking, signature, and other non-OpenAI blocks
|
||||
export function filterToOpenAIFormat(body) {
|
||||
// opts.preserveCacheControl: keep cache_control on content blocks (e.g. for DashScope/alicode)
|
||||
export function filterToOpenAIFormat(body, opts = {}) {
|
||||
if (!body.messages || !Array.isArray(body.messages)) return body;
|
||||
|
||||
const keepCache = !!opts.preserveCacheControl;
|
||||
|
||||
function stripBlock(block) {
|
||||
const { signature, cache_control, ...rest } = block;
|
||||
return keepCache && cache_control ? { ...rest, cache_control } : rest;
|
||||
}
|
||||
|
||||
body.messages = body.messages.map(msg => {
|
||||
// Normalize developer role to system (many providers don't support developer)
|
||||
if (msg.role === ROLE.DEVELOPER) msg = { ...msg, role: ROLE.SYSTEM };
|
||||
|
||||
|
||||
// Keep tool messages as-is (OpenAI format)
|
||||
if (msg.role === ROLE.TOOL) return msg;
|
||||
|
||||
|
||||
// Keep assistant messages with tool_calls as-is
|
||||
if (msg.role === ROLE.ASSISTANT && msg.tool_calls) return msg;
|
||||
|
||||
|
||||
// Handle string content
|
||||
if (typeof msg.content === "string") return msg;
|
||||
|
||||
|
||||
// Handle array content
|
||||
if (Array.isArray(msg.content)) {
|
||||
const filteredContent = [];
|
||||
|
||||
|
||||
for (const block of msg.content) {
|
||||
// Skip thinking blocks
|
||||
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
|
||||
|
||||
|
||||
// Only keep valid OpenAI content types
|
||||
if (VALID_OPENAI_CONTENT_TYPES.includes(block.type)) {
|
||||
// Remove signature field if exists
|
||||
const { signature, cache_control, ...cleanBlock } = block;
|
||||
filteredContent.push(cleanBlock);
|
||||
filteredContent.push(stripBlock(block));
|
||||
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||
// Convert tool_use to tool_calls format (handled separately)
|
||||
continue;
|
||||
} else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) {
|
||||
// Keep tool_result but clean it
|
||||
const { signature, cache_control, ...cleanBlock } = block;
|
||||
filteredContent.push(cleanBlock);
|
||||
filteredContent.push(stripBlock(block));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -109,7 +109,9 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
|
||||
// Always normalize to clean OpenAI format when target is OpenAI
|
||||
// This handles hybrid requests (e.g., OpenAI messages + Claude tools)
|
||||
if (targetFormat === FORMATS.OPENAI) {
|
||||
result = filterToOpenAIFormat(result);
|
||||
result = filterToOpenAIFormat(result, {
|
||||
preserveCacheControl: !!PROVIDERS[provider]?.quirks?.preserveCacheControl,
|
||||
});
|
||||
}
|
||||
|
||||
// Final step: prepare request for Claude format endpoints
|
||||
|
||||
@@ -138,12 +138,14 @@ function convertContent(content) {
|
||||
|
||||
// Text with thoughtSignature = regular text after thinking
|
||||
if (part.thoughtSignature && part.text !== undefined) {
|
||||
textParts.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
if (part.text) {
|
||||
textParts.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Regular text
|
||||
if (part.text !== undefined) {
|
||||
if (part.text !== undefined && part.text !== "") {
|
||||
textParts.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
}
|
||||
|
||||
@@ -180,8 +182,22 @@ function convertContent(content) {
|
||||
}
|
||||
}
|
||||
|
||||
// Content with only functionResponses → return array of tool messages
|
||||
// Content with functionResponses — return array of tool result messages,
|
||||
// plus an assistant message for any co-located tool calls / text.
|
||||
if (toolResults.length > 0) {
|
||||
if (toolCalls.length > 0 || textParts.length > 0 || reasoningContent) {
|
||||
const assistantMsg = { role: ROLE.ASSISTANT };
|
||||
if (textParts.length > 0) {
|
||||
assistantMsg.content = collapseTextParts(textParts);
|
||||
}
|
||||
if (reasoningContent) {
|
||||
assistantMsg.reasoning_content = reasoningContent;
|
||||
}
|
||||
if (toolCalls.length > 0) {
|
||||
assistantMsg.tool_calls = toolCalls;
|
||||
}
|
||||
return [...toolResults, assistantMsg];
|
||||
}
|
||||
return toolResults;
|
||||
}
|
||||
|
||||
|
||||
@@ -393,9 +393,12 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
|
||||
reconcileOrphanedToolResults(history, currentMessage);
|
||||
}
|
||||
|
||||
// API-key auth must never use the shared default ARN (403); OAuth/social fall back to it.
|
||||
// api_key / idc / external_idp must never use the shared default ARN (belongs
|
||||
// to another account → 403 "bearer token invalid"); OAuth/social fall back to it.
|
||||
const authMethod = credentials?.providerSpecificData?.authMethod;
|
||||
const profileArn = authMethod === "api_key"
|
||||
const accountBoundAuth =
|
||||
authMethod === "api_key" || authMethod === "idc" || authMethod === "external_idp";
|
||||
const profileArn = accountBoundAuth
|
||||
? (credentials?.providerSpecificData?.profileArn || "")
|
||||
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));
|
||||
|
||||
|
||||
@@ -129,8 +129,24 @@ function fixMissingToolResponsesOpenAI(messages) {
|
||||
}
|
||||
}
|
||||
|
||||
// Wrap mid-conversation system text so it ends as a user turn (avoids Anthropic prefill 400)
|
||||
function systemReminderText(content) {
|
||||
const parts = Array.isArray(content)
|
||||
? content.filter(c => c?.type === CLAUDE_BLOCK.TEXT).map(c => c.text || "")
|
||||
: [typeof content === "string" ? content : ""];
|
||||
const text = parts.filter(Boolean).join("\n");
|
||||
if (!text.trim()) return "";
|
||||
return `<system-reminder>\n${text}\n</system-reminder>`;
|
||||
}
|
||||
|
||||
// Convert single Claude message - returns single message or array of messages
|
||||
function convertClaudeMessage(msg) {
|
||||
// Mid-conversation system message -> user (per Anthropic placement rules)
|
||||
if (msg.role === ROLE.SYSTEM) {
|
||||
const text = systemReminderText(msg.content);
|
||||
return text ? { role: ROLE.USER, content: text } : null;
|
||||
}
|
||||
|
||||
const role = msg.role === ROLE.USER || msg.role === ROLE.TOOL ? ROLE.USER : ROLE.ASSISTANT;
|
||||
|
||||
// Simple string content
|
||||
|
||||
@@ -35,6 +35,17 @@ function sanitizeGeminiFunctionName(name) {
|
||||
return sanitized.substring(0, 64);
|
||||
}
|
||||
|
||||
function normalizeGeminiContents(contents) {
|
||||
const out = [];
|
||||
for (const c of contents || []) {
|
||||
if (!c?.role || !Array.isArray(c.parts) || c.parts.length === 0) continue;
|
||||
const last = out.at(-1);
|
||||
if (last?.role === c.role) last.parts.push(...c.parts);
|
||||
else out.push({ ...c, parts: [...c.parts] });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// Core: Convert OpenAI request to Gemini format (base for all variants)
|
||||
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) {
|
||||
const result = {
|
||||
@@ -217,6 +228,7 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
|
||||
}
|
||||
}
|
||||
|
||||
result.contents = normalizeGeminiContents(result.contents);
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -299,7 +311,7 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra
|
||||
}
|
||||
|
||||
// Wrap Claude format in Cloud Code envelope for Antigravity
|
||||
function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = null) {
|
||||
function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = null, signature = DEFAULT_THINKING_AG_SIGNATURE) {
|
||||
const projectId = credentials?.projectId || generateProjectId();
|
||||
|
||||
const envelope = {
|
||||
@@ -343,6 +355,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
|
||||
parts.push({ text: block.text });
|
||||
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||
parts.push({
|
||||
thoughtSignature: signature,
|
||||
functionCall: {
|
||||
id: block.id,
|
||||
name: sanitizeGeminiFunctionName(block.name),
|
||||
@@ -425,6 +438,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
|
||||
envelope.request.systemInstruction = { role: GEMINI_ROLE.USER, parts: systemParts };
|
||||
}
|
||||
|
||||
envelope.request.contents = normalizeGeminiContents(envelope.request.contents);
|
||||
return envelope;
|
||||
}
|
||||
|
||||
|
||||
@@ -530,8 +530,15 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
|
||||
// (the ARN doesn't belong to the key's account). So for api_key, only send a
|
||||
// profileArn that was actually resolved for this connection — never the default.
|
||||
// OAuth/social keep the default fallback (their tokens accept it).
|
||||
// api_key / idc / external_idp carry an account-specific (or token-bound)
|
||||
// profile. The shared builder-id/social default ARN belongs to a different
|
||||
// account and triggers 403 "bearer token invalid", so never fall back to it —
|
||||
// send the resolved ARN, or an empty string so CodeWhisperer uses the token's
|
||||
// own default profile. Only OAuth/social keep the shared placeholder.
|
||||
const authMethod = credentials?.providerSpecificData?.authMethod;
|
||||
const profileArn = authMethod === "api_key"
|
||||
const accountBoundAuth =
|
||||
authMethod === "api_key" || authMethod === "idc" || authMethod === "external_idp";
|
||||
const profileArn = accountBoundAuth
|
||||
? (credentials?.providerSpecificData?.profileArn || "")
|
||||
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));
|
||||
|
||||
|
||||
@@ -27,6 +27,25 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
state.messageId = chunk.message?.id || `msg_${Date.now()}`;
|
||||
state.model = chunk.message?.model;
|
||||
state.toolCallIndex = 0;
|
||||
// Claude sends input_tokens + cache_read + cache_creation here; message_delta
|
||||
// later carries only the final output_tokens. Capture cache now so the
|
||||
// delta (output-only) doesn't reset it to zero.
|
||||
const startUsage = chunk.message?.usage;
|
||||
if (startUsage && typeof startUsage === "object") {
|
||||
const inputTokens = typeof startUsage.input_tokens === "number" ? startUsage.input_tokens : 0;
|
||||
const cacheReadTokens = typeof startUsage.cache_read_input_tokens === "number" ? startUsage.cache_read_input_tokens : 0;
|
||||
const cacheCreationTokens = typeof startUsage.cache_creation_input_tokens === "number" ? startUsage.cache_creation_input_tokens : 0;
|
||||
const promptTokens = inputTokens + cacheReadTokens + cacheCreationTokens;
|
||||
state.usage = {
|
||||
prompt_tokens: promptTokens,
|
||||
completion_tokens: 0,
|
||||
total_tokens: promptTokens,
|
||||
input_tokens: inputTokens,
|
||||
output_tokens: 0
|
||||
};
|
||||
if (cacheReadTokens > 0) state.usage.cache_read_input_tokens = cacheReadTokens;
|
||||
if (cacheCreationTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationTokens;
|
||||
}
|
||||
results.push(createChunk(state, { role: ROLE.ASSISTANT }));
|
||||
break;
|
||||
}
|
||||
@@ -103,13 +122,15 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
}
|
||||
|
||||
case "message_delta": {
|
||||
// Extract usage from message_delta event (Claude native format)
|
||||
// Normalize to OpenAI format (prompt_tokens/completion_tokens) for consistent logging
|
||||
// Extract usage from message_delta event (Claude native format).
|
||||
// Anthropic sends input/cache in message_start and only output here, so
|
||||
// fall back to cache captured in message_start when the delta omits it.
|
||||
if (chunk.usage && typeof chunk.usage === "object") {
|
||||
const inputTokens = typeof chunk.usage.input_tokens === "number" ? chunk.usage.input_tokens : 0;
|
||||
const prev = state.usage || {};
|
||||
const inputTokens = typeof chunk.usage.input_tokens === "number" ? chunk.usage.input_tokens : (prev.input_tokens || 0);
|
||||
const outputTokens = typeof chunk.usage.output_tokens === "number" ? chunk.usage.output_tokens : 0;
|
||||
const cacheReadTokens = typeof chunk.usage.cache_read_input_tokens === "number" ? chunk.usage.cache_read_input_tokens : 0;
|
||||
const cacheCreationTokens = typeof chunk.usage.cache_creation_input_tokens === "number" ? chunk.usage.cache_creation_input_tokens : 0;
|
||||
const cacheReadTokens = typeof chunk.usage.cache_read_input_tokens === "number" ? chunk.usage.cache_read_input_tokens : (prev.cache_read_input_tokens || 0);
|
||||
const cacheCreationTokens = typeof chunk.usage.cache_creation_input_tokens === "number" ? chunk.usage.cache_creation_input_tokens : (prev.cache_creation_input_tokens || 0);
|
||||
|
||||
// prompt_tokens = input_tokens + cache_read + cache_creation (all prompt-side tokens)
|
||||
const promptTokens = inputTokens + cacheReadTokens + cacheCreationTokens;
|
||||
@@ -131,7 +152,14 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
const finalChunk = createChunk(state, {}, state.finishReason);
|
||||
|
||||
if (state.usage) {
|
||||
finalChunk.usage = toOpenAIUsage(chunk.usage, "claude");
|
||||
// Build OpenAI usage from the merged state (cache from message_start +
|
||||
// output from message_delta), not the delta chunk alone.
|
||||
finalChunk.usage = toOpenAIUsage({
|
||||
input_tokens: state.usage.input_tokens || 0,
|
||||
output_tokens: state.usage.output_tokens || 0,
|
||||
cache_read_input_tokens: state.usage.cache_read_input_tokens,
|
||||
cache_creation_input_tokens: state.usage.cache_creation_input_tokens
|
||||
}, "claude");
|
||||
}
|
||||
|
||||
results.push(finalChunk);
|
||||
|
||||
@@ -25,7 +25,10 @@ function emitFunctionCall(functionCall, state) {
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: fcName, arguments: JSON.stringify(fcArgs) },
|
||||
};
|
||||
state.toolCalls.set(toolCallIndex, toolCall);
|
||||
// Keep Gemini bookkeeping separate from the shared translator state.toolCalls map.
|
||||
// The downstream OpenAI→Claude translator uses state.toolCalls for Claude block
|
||||
// metadata; pre-populating it here makes Anthropic tool deltas lose index.
|
||||
state.geminiToolCallCount = (state.geminiToolCallCount || 0) + 1;
|
||||
return buildChunk(chunkMeta(state), { tool_calls: [toolCall] }, null);
|
||||
}
|
||||
|
||||
@@ -46,6 +49,7 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
state.messageId = response.responseId || `msg_${Date.now()}`;
|
||||
state.model = response.modelVersion || "gemini";
|
||||
state.functionIndex = 0;
|
||||
state.geminiToolCallCount = 0;
|
||||
results.push(buildChunk(chunkMeta(state), { role: ROLE.ASSISTANT }, null));
|
||||
}
|
||||
|
||||
@@ -117,7 +121,7 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
// Finish reason - include usage in final chunk
|
||||
if (candidate.finishReason) {
|
||||
let finishReason = toOpenAIFinish(candidate.finishReason, "gemini");
|
||||
if (finishReason === OPENAI_FINISH.STOP && state.toolCalls.size > 0) {
|
||||
if (finishReason === OPENAI_FINISH.STOP && state.geminiToolCallCount > 0) {
|
||||
finishReason = OPENAI_FINISH.TOOL_CALLS;
|
||||
}
|
||||
|
||||
|
||||
@@ -184,7 +184,8 @@ export function openaiToClaudeResponse(chunk, state) {
|
||||
for (const tc of delta.tool_calls) {
|
||||
const idx = tc.index ?? 0;
|
||||
|
||||
if (tc.id) {
|
||||
// GLM/fireworks repeats id+null-name on every arg chunk; open block once per idx
|
||||
if (tc.id && !state.toolCalls.has(idx)) {
|
||||
stopThinkingBlock(state, results);
|
||||
stopTextBlock(state, results);
|
||||
|
||||
|
||||
@@ -247,9 +247,24 @@ function mergeChunksToResponse(chunks, sourceFormat) {
|
||||
|
||||
if (messageStart?.message) {
|
||||
finalChunk = messageStart.message;
|
||||
// Merge usage if available
|
||||
if (messageDelta?.usage) {
|
||||
finalChunk.usage = messageDelta.usage;
|
||||
// message_start.usage has input + cache; message_delta.usage has the
|
||||
// final output_tokens. Merge so cache survives (delta omits it).
|
||||
const startUsage = messageStart.message.usage;
|
||||
const deltaUsage = messageDelta?.usage;
|
||||
if (startUsage || deltaUsage) {
|
||||
finalChunk.usage = {
|
||||
...(startUsage || {}),
|
||||
...(deltaUsage || {}),
|
||||
...(startUsage?.cache_read_input_tokens !== undefined
|
||||
? { cache_read_input_tokens: startUsage.cache_read_input_tokens }
|
||||
: {}),
|
||||
...(startUsage?.cache_creation_input_tokens !== undefined
|
||||
? { cache_creation_input_tokens: startUsage.cache_creation_input_tokens }
|
||||
: {}),
|
||||
...(startUsage?.input_tokens !== undefined
|
||||
? { input_tokens: startUsage.input_tokens }
|
||||
: {})
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import { formatSSE } from "./streamHelpers.js";
|
||||
// Responses API events that signal the stream has reached a terminal state
|
||||
const OPENAI_RESPONSES_TERMINAL_EVENTS = new Set([
|
||||
"response.completed",
|
||||
"response.done",
|
||||
"response.failed",
|
||||
"error"
|
||||
]);
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
import { translateResponse, initState } from "../translator/index.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb.js";
|
||||
<<<<<<< HEAD
|
||||
import { extractUsage, hasValidUsage, estimateUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js";
|
||||
import { saveUsageStats } from "../handlers/chatCore/requestDetail.js";
|
||||
=======
|
||||
import { extractUsage, mergeUsage, hasValidUsage, estimateUsage, logUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js";
|
||||
>>>>>>> 7f436e2792be4fa5a4d1c4d6b8e9bc85eaaa6a3d
|
||||
import { parseSSELine, hasValuableContent, fixInvalidId, formatSSE } from "./streamHelpers.js";
|
||||
import { getOpenAIResponsesEventName, isOpenAIResponsesTerminalEvent, formatIncompleteOpenAIResponsesStreamFailure } from "./responsesStreamHelpers.js";
|
||||
import { dbg, isDebugEnabled } from "./debugLog.js";
|
||||
@@ -131,6 +135,20 @@ export function createSSEStream(options = {}) {
|
||||
}
|
||||
}
|
||||
|
||||
// Strip empty tool_calls arrays that break AI SDK reasoning tracking.
|
||||
// Some providers (e.g. CodeBuddy CN) include `"tool_calls": []` in
|
||||
// every streaming delta. @ai-sdk/openai-compatible checks
|
||||
// `delta.tool_calls != null` — an empty array passes this check,
|
||||
// causing premature `reasoning-end` on every chunk.
|
||||
if (parsed?.choices) {
|
||||
for (const choice of parsed.choices) {
|
||||
if (choice.delta?.tool_calls && Array.isArray(choice.delta.tool_calls) && choice.delta.tool_calls.length === 0) {
|
||||
delete choice.delta.tool_calls;
|
||||
fieldsInjected = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!hasValuableContent(parsed, FORMATS.OPENAI)) {
|
||||
continue;
|
||||
}
|
||||
@@ -149,7 +167,7 @@ export function createSSEStream(options = {}) {
|
||||
|
||||
const extracted = extractUsage(parsed);
|
||||
if (extracted) {
|
||||
usage = extracted;
|
||||
usage = mergeUsage(usage, extracted);
|
||||
}
|
||||
|
||||
const isFinishChunk = parsed.choices?.[0]?.finish_reason;
|
||||
@@ -218,9 +236,11 @@ export function createSSEStream(options = {}) {
|
||||
sseEmittedCount++;
|
||||
}
|
||||
|
||||
// [DONE] not emitted in translate mode — some clients' SSE decoders
|
||||
// fail to parse the OpenAI sentinel on Claude-format translated streams.
|
||||
// message_stop already signals end-of-response; stream close handles it.
|
||||
if (keepsOpenAIResponsesFormat && !streamDoneSent) {
|
||||
const doneOutput = "data: [DONE]\n\n";
|
||||
reqLogger?.appendConvertedChunk?.(doneOutput);
|
||||
controller.enqueue(sharedEncoder.encode(doneOutput));
|
||||
}
|
||||
streamDoneSent = true;
|
||||
if (keepsOpenAIResponsesFormat) openAIResponsesDoneSent = true;
|
||||
continue;
|
||||
@@ -265,7 +285,7 @@ export function createSSEStream(options = {}) {
|
||||
|
||||
// Extract usage
|
||||
const extracted = extractUsage(parsed);
|
||||
if (extracted) state.usage = extracted; // Keep original usage for logging
|
||||
if (extracted) state.usage = mergeUsage(state.usage, extracted); // Keep original usage for logging
|
||||
|
||||
// Responses same-format passthrough: re-emit with original event framing
|
||||
if (keepsOpenAIResponsesFormat && openAIResponsesEventName) {
|
||||
@@ -351,7 +371,9 @@ export function createSSEStream(options = {}) {
|
||||
// Some clients (e.g. OpenClaw) expect the OpenAI-style sentinel:
|
||||
// data: [DONE]\n\n
|
||||
// Without it they can hang until timeout and trigger failover.
|
||||
if (!streamDoneSent) {
|
||||
// Gemini-family clients (Antigravity, Vertex, Gemini) reject this sentinel with 400 syntax errors.
|
||||
const isGeminiFamily = provider === "antigravity" || provider === "gemini" || provider === "vertex";
|
||||
if (!streamDoneSent && !isGeminiFamily) {
|
||||
const doneOutput = "data: [DONE]\n\n";
|
||||
reqLogger?.appendConvertedChunk?.(doneOutput);
|
||||
controller.enqueue(sharedEncoder.encode(doneOutput));
|
||||
@@ -416,8 +438,13 @@ export function createSSEStream(options = {}) {
|
||||
openAIResponsesTerminalSeen = true;
|
||||
}
|
||||
|
||||
// [DONE] not emitted in translate mode — see comment above.
|
||||
// Passthrough mode still emits it for standard OpenAI clients.
|
||||
if (keepsOpenAIResponsesFormat && !openAIResponsesDoneSent && !streamDoneSent) {
|
||||
const doneOutput = "data: [DONE]\n\n";
|
||||
reqLogger?.appendConvertedChunk?.(doneOutput);
|
||||
controller.enqueue(sharedEncoder.encode(doneOutput));
|
||||
openAIResponsesDoneSent = true;
|
||||
streamDoneSent = true;
|
||||
}
|
||||
|
||||
if (!hasValidUsage(state?.usage) && totalContentLength > 0) {
|
||||
state.usage = estimateUsage(body, totalContentLength, sourceFormat);
|
||||
|
||||
@@ -141,6 +141,68 @@ export function normalizeUsage(usage) {
|
||||
return normalized;
|
||||
}
|
||||
|
||||
/**
|
||||
* Canonicalize usage into ONE storage/cost convention so token counts and cost
|
||||
* are consistent across providers:
|
||||
* prompt_tokens = total input INCLUDING cache read + cache creation
|
||||
* cached_tokens = cache-read portion (subset of prompt_tokens)
|
||||
* cache_creation_input_tokens = cache-write portion (subset of prompt_tokens)
|
||||
* completion_tokens, reasoning_tokens, total_tokens
|
||||
*
|
||||
* Discriminator: Claude reports cache_read_input_tokens with a prompt that
|
||||
* EXCLUDES cache, so we fold cache into prompt. OpenAI/Gemini report
|
||||
* cached_tokens already counted inside prompt, so we pass through. Idempotent:
|
||||
* once folded the output carries cached_tokens (not cache_read_input_tokens),
|
||||
* so re-running takes the passthrough branch and does not double-add.
|
||||
*
|
||||
* @param {object} usage - a normalizeUsage()-shaped object
|
||||
* @returns {object|null} canonical token object, or null for invalid input
|
||||
*/
|
||||
export function canonicalizeUsage(usage) {
|
||||
if (!usage || typeof usage !== "object" || Array.isArray(usage)) return null;
|
||||
|
||||
const num = (v) => (Number.isFinite(Number(v)) ? Number(v) : 0);
|
||||
const completion = num(usage.completion_tokens ?? usage.output_tokens);
|
||||
const reasoning = num(usage.reasoning_tokens);
|
||||
// Fall back to the nested prompt_tokens_details.cache_creation_tokens shape
|
||||
// (buildUsage()'s OpenAI-forwarding format) when the top-level field is
|
||||
// absent, so callers that pass a buildUsage() object through don't silently
|
||||
// drop cache_creation.
|
||||
const cacheCreation = num(usage.cache_creation_input_tokens ?? usage.prompt_tokens_details?.cache_creation_tokens);
|
||||
|
||||
let prompt = num(usage.prompt_tokens ?? usage.input_tokens);
|
||||
let cached;
|
||||
|
||||
// Claude path: prompt excludes cache; cache_read_input_tokens and/or
|
||||
// cache_creation_input_tokens are separate. A cache-miss "first write" only
|
||||
// carries cache_creation_input_tokens (no cache_read_input_tokens yet), so
|
||||
// check both fields — otherwise a first-write request falls through to the
|
||||
// OpenAI passthrough branch below and cache_creation never gets folded in.
|
||||
// Guard on the absence of `cached_tokens`: our own canonical output always
|
||||
// sets that key (even to 0), so re-running canonicalizeUsage on an already-
|
||||
// folded result takes the passthrough branch instead of folding again.
|
||||
if (usage.cached_tokens === undefined &&
|
||||
(usage.cache_read_input_tokens !== undefined || usage.cache_creation_input_tokens !== undefined)) {
|
||||
cached = num(usage.cache_read_input_tokens);
|
||||
prompt = prompt + cached + cacheCreation;
|
||||
} else {
|
||||
// OpenAI/Gemini path (or already-canonical input): prompt already includes cached_tokens.
|
||||
cached = num(usage.cached_tokens);
|
||||
}
|
||||
|
||||
const result = {
|
||||
prompt_tokens: prompt,
|
||||
completion_tokens: completion,
|
||||
// Recompute rather than pass through: when the fold branch ran above,
|
||||
// an upstream total_tokens (cache-exclusive) would otherwise be stale.
|
||||
total_tokens: prompt + completion,
|
||||
cached_tokens: cached,
|
||||
cache_creation_input_tokens: cacheCreation,
|
||||
};
|
||||
if (reasoning > 0) result.reasoning_tokens = reasoning;
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if usage has valid token data
|
||||
* Valid = has at least one token field with value > 0
|
||||
@@ -171,6 +233,19 @@ export function hasValidUsage(usage) {
|
||||
export function extractUsage(chunk) {
|
||||
if (!chunk || typeof chunk !== "object") return null;
|
||||
|
||||
// Claude format (message_start event): carries input_tokens + cache_read +
|
||||
// cache_creation. message_delta later carries only the final output_tokens,
|
||||
// so callers must MERGE (mergeUsage), not overwrite, to keep cache counts.
|
||||
if (chunk.type === "message_start" && chunk.message?.usage && typeof chunk.message.usage === "object") {
|
||||
const u = chunk.message.usage;
|
||||
return normalizeUsage({
|
||||
prompt_tokens: u.input_tokens || 0,
|
||||
completion_tokens: u.output_tokens || 0,
|
||||
cache_read_input_tokens: u.cache_read_input_tokens,
|
||||
cache_creation_input_tokens: u.cache_creation_input_tokens
|
||||
});
|
||||
}
|
||||
|
||||
// Claude format (message_delta event)
|
||||
if (chunk.type === "message_delta" && chunk.usage && typeof chunk.usage === "object") {
|
||||
return normalizeUsage({
|
||||
@@ -232,6 +307,27 @@ export function extractUsage(chunk) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Field-wise max-merge of two usage objects. Anthropic splits usage across
|
||||
// events: message_start has real input+cache (output is a placeholder 1),
|
||||
// message_delta has the real cumulative output (input/cache absent). Max keeps
|
||||
// the meaningful value from each without clobbering. Idempotent for other
|
||||
// providers that emit a single complete usage object.
|
||||
export function mergeUsage(prev, next) {
|
||||
if (!prev) return next || null;
|
||||
if (!next) return prev;
|
||||
const merged = { ...prev };
|
||||
for (const [k, v] of Object.entries(next)) {
|
||||
// typeof NaN === "number" — guard with Number.isFinite so one malformed
|
||||
// chunk can't poison the whole accumulation (Math.max(x, NaN) is NaN).
|
||||
if (typeof v === "number" && Number.isFinite(v)) {
|
||||
merged[k] = Math.max(typeof merged[k] === "number" ? merged[k] : 0, v);
|
||||
} else if (v && typeof v === "object") {
|
||||
merged[k] = v; // nested details objects: take latest
|
||||
}
|
||||
}
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* Estimate input tokens from request body
|
||||
* Calculate total body size for more accurate estimation
|
||||
|
||||
Reference in New Issue
Block a user