end
This commit is contained in:
@@ -6,6 +6,7 @@ import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
|
||||
|
||||
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
|
||||
function sanitizeFunctionName(name) {
|
||||
@@ -177,8 +178,19 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
if (p.thoughtSignature && !p.functionCall && !p.text) return false;
|
||||
return true;
|
||||
});
|
||||
if (role !== c.role || parts?.length !== c.parts?.length) {
|
||||
return { ...c, role, parts };
|
||||
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
|
||||
// don't persist thoughtSignature in their history, so backfill the default signature on any
|
||||
// functionCall part that arrives without one.
|
||||
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
|
||||
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
|
||||
return {
|
||||
...c, role,
|
||||
parts: needsBackfill
|
||||
? parts.map(p => (p.functionCall && !p.thoughtSignature)
|
||||
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
|
||||
: p)
|
||||
: parts,
|
||||
};
|
||||
}
|
||||
return c;
|
||||
});
|
||||
|
||||
@@ -226,6 +226,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
|
||||
cline: () => this.refreshCline(credentials.refreshToken, proxyOptions),
|
||||
clinepass: () => this.refreshCline(credentials.refreshToken, proxyOptions),
|
||||
"kimi-coding": () => this.refreshKimiCoding(credentials.refreshToken, proxyOptions),
|
||||
kilocode: () => this.refreshKilocode(credentials.refreshToken, proxyOptions)
|
||||
};
|
||||
@@ -299,7 +300,11 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
const data = payload?.data || payload;
|
||||
const expiresAtIso = data?.expiresAt;
|
||||
const expiresIn = expiresAtIso ? Math.max(1, Math.floor((new Date(expiresAtIso).getTime() - Date.now()) / 1000)) : undefined;
|
||||
return { accessToken: data?.accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
|
||||
let accessToken = data?.accessToken;
|
||||
if (accessToken && !accessToken.startsWith("workos:")) {
|
||||
accessToken = `workos:${accessToken}`;
|
||||
}
|
||||
return { accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
|
||||
}
|
||||
|
||||
async refreshKimiCoding(refreshToken, proxyOptions = null) {
|
||||
|
||||
@@ -5,6 +5,7 @@ import { GithubExecutor } from "./github.js";
|
||||
import { IFlowExecutor } from "./iflow.js";
|
||||
import { QoderExecutor } from "./qoder.js";
|
||||
import { KiroExecutor } from "./kiro.js";
|
||||
import { KimchiExecutor } from "./kimchi.js";
|
||||
import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
@@ -28,6 +29,7 @@ const executors = {
|
||||
iflow: new IFlowExecutor(),
|
||||
qoder: new QoderExecutor(),
|
||||
kiro: new KiroExecutor(),
|
||||
kimchi: new KimchiExecutor(),
|
||||
codex: new CodexExecutor(),
|
||||
cursor: new CursorExecutor(),
|
||||
cu: new CursorExecutor(), // Alias for cursor
|
||||
@@ -66,6 +68,7 @@ export { GithubExecutor } from "./github.js";
|
||||
export { IFlowExecutor } from "./iflow.js";
|
||||
export { QoderExecutor } from "./qoder.js";
|
||||
export { KiroExecutor } from "./kiro.js";
|
||||
export { KimchiExecutor } from "./kimchi.js";
|
||||
export { CodexExecutor } from "./codex.js";
|
||||
export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
|
||||
123
open-sse/executors/kimchi.js
Normal file
123
open-sse/executors/kimchi.js
Normal file
@@ -0,0 +1,123 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { getCachedKimchiModelMetadata } from "../services/kimchiModels.js";
|
||||
|
||||
const TOP_LEVEL_OPENAI_GATEWAY_DROPS = [
|
||||
"anthropic_version",
|
||||
"anthropic_beta",
|
||||
"client_metadata",
|
||||
"mcp_servers",
|
||||
"stop_sequences",
|
||||
"thinking",
|
||||
"top_k",
|
||||
];
|
||||
|
||||
function systemToText(system) {
|
||||
if (typeof system === "string") return system;
|
||||
if (Array.isArray(system)) {
|
||||
return system
|
||||
.map((part) => {
|
||||
if (typeof part === "string") return part;
|
||||
if (typeof part?.text === "string") return part.text;
|
||||
return "";
|
||||
})
|
||||
.filter(Boolean)
|
||||
.join("\n");
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
function mergeTopLevelSystem(body) {
|
||||
if (!body?.system || !Array.isArray(body.messages)) return;
|
||||
const text = systemToText(body.system).trim();
|
||||
if (!text) return;
|
||||
|
||||
const existing = body.messages.find((msg) => msg?.role === "system");
|
||||
if (!existing) {
|
||||
body.messages.unshift({ role: "system", content: text });
|
||||
return;
|
||||
}
|
||||
|
||||
if (typeof existing.content === "string") {
|
||||
existing.content = `${text}\n\n${existing.content}`;
|
||||
} else if (Array.isArray(existing.content)) {
|
||||
existing.content.unshift({ type: "text", text });
|
||||
}
|
||||
}
|
||||
|
||||
function stripMessageArtifacts(body) {
|
||||
if (!Array.isArray(body?.messages)) return;
|
||||
for (const msg of body.messages) {
|
||||
if (!msg || typeof msg !== "object") continue;
|
||||
delete msg.cache_control;
|
||||
if (!Array.isArray(msg.content)) continue;
|
||||
msg.content = msg.content.map((part) => {
|
||||
if (!part || typeof part !== "object") return part;
|
||||
const { cache_control, signature, ...clean } = part;
|
||||
return clean;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
function stripToolArtifacts(body) {
|
||||
if (!Array.isArray(body?.tools)) return;
|
||||
body.tools = body.tools.map((tool) => {
|
||||
if (!tool || typeof tool !== "object") return tool;
|
||||
const { cache_control, ...clean } = tool;
|
||||
return clean;
|
||||
});
|
||||
}
|
||||
|
||||
// Strip `reasoning_content` echoed by clients on assistant messages — but
|
||||
// only when it's a real thinking block. `DefaultExecutor.transformRequest`
|
||||
// runs `injectReasoningContent` first and may inject a 1-char placeholder
|
||||
// (" ") for upstream validation; the placeholder is small (no token cost
|
||||
// worth stripping) and stripping it would re-trigger upstream to complain
|
||||
// about missing reasoning on the next turn. Threshold matches the
|
||||
// placeholder length with a safety margin.
|
||||
const REASONING_PLACEHOLDER_MAX_LEN = 8;
|
||||
|
||||
export function stripReasoningContent(body) {
|
||||
if (!Array.isArray(body?.messages)) return;
|
||||
for (const msg of body.messages) {
|
||||
if (msg && msg.role === "assistant" && typeof msg.reasoning_content === "string"
|
||||
&& msg.reasoning_content.length > REASONING_PLACEHOLDER_MAX_LEN) {
|
||||
delete msg.reasoning_content;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function isAnthropicBackedKimchiModel(model) {
|
||||
const meta = getCachedKimchiModelMetadata(model);
|
||||
if (meta?.provider === "anthropic" || meta?.upstreamProvider === "anthropic") return true;
|
||||
return /(^|[-_/])(?:claude|anthropic)(?:[-_/]|$)/i.test(String(model || ""));
|
||||
}
|
||||
|
||||
export class KimchiExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("kimchi");
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
const transformed = super.transformRequest(model, body, stream, credentials);
|
||||
if (!transformed || typeof transformed !== "object") return transformed;
|
||||
|
||||
mergeTopLevelSystem(transformed);
|
||||
for (const key of TOP_LEVEL_OPENAI_GATEWAY_DROPS) {
|
||||
if (transformed[key] !== undefined) delete transformed[key];
|
||||
}
|
||||
delete transformed.system;
|
||||
|
||||
if (isAnthropicBackedKimchiModel(model)) {
|
||||
delete transformed.reasoning_effort;
|
||||
delete transformed.reasoning;
|
||||
delete transformed.thinking;
|
||||
}
|
||||
|
||||
stripMessageArtifacts(transformed);
|
||||
stripToolArtifacts(transformed);
|
||||
stripReasoningContent(transformed);
|
||||
return transformed;
|
||||
}
|
||||
}
|
||||
|
||||
export default KimchiExecutor;
|
||||
@@ -64,9 +64,22 @@ export class KiroExecutor extends BaseExecutor {
|
||||
getOrderedBaseUrls(credentials) {
|
||||
const baseUrls = this.getBaseUrls();
|
||||
const authMethod = credentials?.providerSpecificData?.authMethod;
|
||||
const isCodeWhispererSurface = authMethod === "api_key" || authMethod === "external_idp";
|
||||
// IAM Identity Center (idc) tokens are AWS SSO access tokens — the same
|
||||
// family as external_idp/api_key. The kiro.dev gateway rejects them with
|
||||
// 403 "bearer token invalid", so they must hit the CodeWhisperer
|
||||
// *.amazonaws.com surface, and in the region the token was minted in
|
||||
// (the baseUrls are hardcoded us-east-1).
|
||||
const isCodeWhispererSurface =
|
||||
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
|
||||
if (!isCodeWhispererSurface) return baseUrls;
|
||||
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com"));
|
||||
|
||||
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
|
||||
const regionalize = (u) =>
|
||||
region && region !== "us-east-1" && u.includes("amazonaws.com")
|
||||
? u.replace(/([a-z]+)\.[a-z0-9-]+\.amazonaws\.com/, `$1.${region}.amazonaws.com`)
|
||||
: u;
|
||||
|
||||
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
|
||||
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
|
||||
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
|
||||
}
|
||||
@@ -122,7 +135,8 @@ export class KiroExecutor extends BaseExecutor {
|
||||
hasReasoningContent: false,
|
||||
reasoningChunkCount: 0,
|
||||
toolCallIndex: 0,
|
||||
seenToolIds: new Map()
|
||||
seenToolIds: new Map(),
|
||||
inThinking: false
|
||||
};
|
||||
|
||||
const transformStream = new TransformStream({
|
||||
@@ -159,7 +173,36 @@ export class KiroExecutor extends BaseExecutor {
|
||||
|
||||
// Handle assistantResponseEvent
|
||||
if (eventType === "assistantResponseEvent" && event.payload?.content) {
|
||||
const content = event.payload.content;
|
||||
let content = event.payload.content;
|
||||
|
||||
// Kiro Claude models can leak <thinking> blocks into the content stream.
|
||||
// We strip these literal tags to prevent duplication, as the reasoning
|
||||
// is already routed correctly via reasoningContentEvent.
|
||||
if (state.inThinking) {
|
||||
if (content.includes("</thinking>")) {
|
||||
state.inThinking = false;
|
||||
const after = content.split("</thinking>").slice(1).join("</thinking>");
|
||||
content = after.startsWith("\n") ? after.substring(1) : after;
|
||||
} else {
|
||||
content = ""; // Drop entirely while inside thinking block
|
||||
}
|
||||
} else if (content.includes("<thinking>")) {
|
||||
state.inThinking = true;
|
||||
if (content.includes("</thinking>")) {
|
||||
state.inThinking = false;
|
||||
const before = content.split("<thinking>")[0];
|
||||
const after = content.split("</thinking>").slice(1).join("</thinking>");
|
||||
content = before + (after.startsWith("\n") ? after.substring(1) : after);
|
||||
} else {
|
||||
content = content.split("<thinking>")[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (!content && state.hasReasoningContent) {
|
||||
// If we stripped everything, skip emitting an empty content chunk
|
||||
continue;
|
||||
}
|
||||
|
||||
state.totalContentLength += content.length;
|
||||
|
||||
const chunk = {
|
||||
@@ -348,6 +391,11 @@ export class KiroExecutor extends BaseExecutor {
|
||||
if (metrics && typeof metrics === 'object') {
|
||||
const inputTokens = metrics.inputTokens || 0;
|
||||
const outputTokens = metrics.outputTokens || 0;
|
||||
// ponytail: Amazon Q upstream does not expose cache fields today,
|
||||
// but pick up cache_read_input_tokens / cache_creation_input_tokens
|
||||
// if the event shape grows them so cost tracking stays accurate.
|
||||
const cachedTokens = metrics.cacheReadInputTokens || metrics.cache_read_input_tokens || 0;
|
||||
const cacheCreationInputTokens = metrics.cacheCreationInputTokens || metrics.cache_creation_input_tokens || 0;
|
||||
|
||||
if (inputTokens > 0 || outputTokens > 0) {
|
||||
state.usage = {
|
||||
@@ -355,6 +403,12 @@ export class KiroExecutor extends BaseExecutor {
|
||||
completion_tokens: outputTokens,
|
||||
total_tokens: inputTokens + outputTokens
|
||||
};
|
||||
// Kiro is Claude-backed: inputTokens EXCLUDES cache (Claude convention),
|
||||
// not inclusive like OpenAI's cached_tokens. Emit cache_read_input_tokens
|
||||
// (not cached_tokens) so canonicalizeUsage takes the Claude fold path and
|
||||
// correctly adds cache back into prompt_tokens instead of undercharging.
|
||||
if (cachedTokens > 0) state.usage.cache_read_input_tokens = cachedTokens;
|
||||
if (cacheCreationInputTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationInputTokens;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user