This commit is contained in:
2026-07-06 00:01:34 +07:00
parent 2729408ef3
commit e7470e955e
108 changed files with 4131 additions and 529 deletions

View File

@@ -6,6 +6,7 @@ import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
function sanitizeFunctionName(name) {
@@ -177,8 +178,19 @@ export class AntigravityExecutor extends BaseExecutor {
if (p.thoughtSignature && !p.functionCall && !p.text) return false;
return true;
});
if (role !== c.role || parts?.length !== c.parts?.length) {
return { ...c, role, parts };
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
// don't persist thoughtSignature in their history, so backfill the default signature on any
// functionCall part that arrives without one.
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
return {
...c, role,
parts: needsBackfill
? parts.map(p => (p.functionCall && !p.thoughtSignature)
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
: p)
: parts,
};
}
return c;
});

View File

@@ -226,6 +226,7 @@ export class DefaultExecutor extends BaseExecutor {
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
cline: () => this.refreshCline(credentials.refreshToken, proxyOptions),
clinepass: () => this.refreshCline(credentials.refreshToken, proxyOptions),
"kimi-coding": () => this.refreshKimiCoding(credentials.refreshToken, proxyOptions),
kilocode: () => this.refreshKilocode(credentials.refreshToken, proxyOptions)
};
@@ -299,7 +300,11 @@ export class DefaultExecutor extends BaseExecutor {
const data = payload?.data || payload;
const expiresAtIso = data?.expiresAt;
const expiresIn = expiresAtIso ? Math.max(1, Math.floor((new Date(expiresAtIso).getTime() - Date.now()) / 1000)) : undefined;
return { accessToken: data?.accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
let accessToken = data?.accessToken;
if (accessToken && !accessToken.startsWith("workos:")) {
accessToken = `workos:${accessToken}`;
}
return { accessToken, refreshToken: data?.refreshToken || refreshToken, expiresIn };
}
async refreshKimiCoding(refreshToken, proxyOptions = null) {

View File

@@ -5,6 +5,7 @@ import { GithubExecutor } from "./github.js";
import { IFlowExecutor } from "./iflow.js";
import { QoderExecutor } from "./qoder.js";
import { KiroExecutor } from "./kiro.js";
import { KimchiExecutor } from "./kimchi.js";
import { CodexExecutor } from "./codex.js";
import { CursorExecutor } from "./cursor.js";
import { VertexExecutor } from "./vertex.js";
@@ -28,6 +29,7 @@ const executors = {
iflow: new IFlowExecutor(),
qoder: new QoderExecutor(),
kiro: new KiroExecutor(),
kimchi: new KimchiExecutor(),
codex: new CodexExecutor(),
cursor: new CursorExecutor(),
cu: new CursorExecutor(), // Alias for cursor
@@ -66,6 +68,7 @@ export { GithubExecutor } from "./github.js";
export { IFlowExecutor } from "./iflow.js";
export { QoderExecutor } from "./qoder.js";
export { KiroExecutor } from "./kiro.js";
export { KimchiExecutor } from "./kimchi.js";
export { CodexExecutor } from "./codex.js";
export { CursorExecutor } from "./cursor.js";
export { VertexExecutor } from "./vertex.js";

View File

@@ -0,0 +1,123 @@
import { DefaultExecutor } from "./default.js";
import { getCachedKimchiModelMetadata } from "../services/kimchiModels.js";
const TOP_LEVEL_OPENAI_GATEWAY_DROPS = [
"anthropic_version",
"anthropic_beta",
"client_metadata",
"mcp_servers",
"stop_sequences",
"thinking",
"top_k",
];
function systemToText(system) {
if (typeof system === "string") return system;
if (Array.isArray(system)) {
return system
.map((part) => {
if (typeof part === "string") return part;
if (typeof part?.text === "string") return part.text;
return "";
})
.filter(Boolean)
.join("\n");
}
return "";
}
function mergeTopLevelSystem(body) {
if (!body?.system || !Array.isArray(body.messages)) return;
const text = systemToText(body.system).trim();
if (!text) return;
const existing = body.messages.find((msg) => msg?.role === "system");
if (!existing) {
body.messages.unshift({ role: "system", content: text });
return;
}
if (typeof existing.content === "string") {
existing.content = `${text}\n\n${existing.content}`;
} else if (Array.isArray(existing.content)) {
existing.content.unshift({ type: "text", text });
}
}
function stripMessageArtifacts(body) {
if (!Array.isArray(body?.messages)) return;
for (const msg of body.messages) {
if (!msg || typeof msg !== "object") continue;
delete msg.cache_control;
if (!Array.isArray(msg.content)) continue;
msg.content = msg.content.map((part) => {
if (!part || typeof part !== "object") return part;
const { cache_control, signature, ...clean } = part;
return clean;
});
}
}
function stripToolArtifacts(body) {
if (!Array.isArray(body?.tools)) return;
body.tools = body.tools.map((tool) => {
if (!tool || typeof tool !== "object") return tool;
const { cache_control, ...clean } = tool;
return clean;
});
}
// Strip `reasoning_content` echoed by clients on assistant messages — but
// only when it's a real thinking block. `DefaultExecutor.transformRequest`
// runs `injectReasoningContent` first and may inject a 1-char placeholder
// (" ") for upstream validation; the placeholder is small (no token cost
// worth stripping) and stripping it would re-trigger upstream to complain
// about missing reasoning on the next turn. Threshold matches the
// placeholder length with a safety margin.
const REASONING_PLACEHOLDER_MAX_LEN = 8;
export function stripReasoningContent(body) {
if (!Array.isArray(body?.messages)) return;
for (const msg of body.messages) {
if (msg && msg.role === "assistant" && typeof msg.reasoning_content === "string"
&& msg.reasoning_content.length > REASONING_PLACEHOLDER_MAX_LEN) {
delete msg.reasoning_content;
}
}
}
function isAnthropicBackedKimchiModel(model) {
const meta = getCachedKimchiModelMetadata(model);
if (meta?.provider === "anthropic" || meta?.upstreamProvider === "anthropic") return true;
return /(^|[-_/])(?:claude|anthropic)(?:[-_/]|$)/i.test(String(model || ""));
}
export class KimchiExecutor extends DefaultExecutor {
constructor() {
super("kimchi");
}
transformRequest(model, body, stream, credentials) {
const transformed = super.transformRequest(model, body, stream, credentials);
if (!transformed || typeof transformed !== "object") return transformed;
mergeTopLevelSystem(transformed);
for (const key of TOP_LEVEL_OPENAI_GATEWAY_DROPS) {
if (transformed[key] !== undefined) delete transformed[key];
}
delete transformed.system;
if (isAnthropicBackedKimchiModel(model)) {
delete transformed.reasoning_effort;
delete transformed.reasoning;
delete transformed.thinking;
}
stripMessageArtifacts(transformed);
stripToolArtifacts(transformed);
stripReasoningContent(transformed);
return transformed;
}
}
export default KimchiExecutor;

View File

@@ -64,9 +64,22 @@ export class KiroExecutor extends BaseExecutor {
getOrderedBaseUrls(credentials) {
const baseUrls = this.getBaseUrls();
const authMethod = credentials?.providerSpecificData?.authMethod;
const isCodeWhispererSurface = authMethod === "api_key" || authMethod === "external_idp";
// IAM Identity Center (idc) tokens are AWS SSO access tokens — the same
// family as external_idp/api_key. The kiro.dev gateway rejects them with
// 403 "bearer token invalid", so they must hit the CodeWhisperer
// *.amazonaws.com surface, and in the region the token was minted in
// (the baseUrls are hardcoded us-east-1).
const isCodeWhispererSurface =
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
if (!isCodeWhispererSurface) return baseUrls;
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com"));
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
const regionalize = (u) =>
region && region !== "us-east-1" && u.includes("amazonaws.com")
? u.replace(/([a-z]+)\.[a-z0-9-]+\.amazonaws\.com/, `$1.${region}.amazonaws.com`)
: u;
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
}
@@ -122,7 +135,8 @@ export class KiroExecutor extends BaseExecutor {
hasReasoningContent: false,
reasoningChunkCount: 0,
toolCallIndex: 0,
seenToolIds: new Map()
seenToolIds: new Map(),
inThinking: false
};
const transformStream = new TransformStream({
@@ -159,7 +173,36 @@ export class KiroExecutor extends BaseExecutor {
// Handle assistantResponseEvent
if (eventType === "assistantResponseEvent" && event.payload?.content) {
const content = event.payload.content;
let content = event.payload.content;
// Kiro Claude models can leak <thinking> blocks into the content stream.
// We strip these literal tags to prevent duplication, as the reasoning
// is already routed correctly via reasoningContentEvent.
if (state.inThinking) {
if (content.includes("</thinking>")) {
state.inThinking = false;
const after = content.split("</thinking>").slice(1).join("</thinking>");
content = after.startsWith("\n") ? after.substring(1) : after;
} else {
content = ""; // Drop entirely while inside thinking block
}
} else if (content.includes("<thinking>")) {
state.inThinking = true;
if (content.includes("</thinking>")) {
state.inThinking = false;
const before = content.split("<thinking>")[0];
const after = content.split("</thinking>").slice(1).join("</thinking>");
content = before + (after.startsWith("\n") ? after.substring(1) : after);
} else {
content = content.split("<thinking>")[0];
}
}
if (!content && state.hasReasoningContent) {
// If we stripped everything, skip emitting an empty content chunk
continue;
}
state.totalContentLength += content.length;
const chunk = {
@@ -348,6 +391,11 @@ export class KiroExecutor extends BaseExecutor {
if (metrics && typeof metrics === 'object') {
const inputTokens = metrics.inputTokens || 0;
const outputTokens = metrics.outputTokens || 0;
// ponytail: Amazon Q upstream does not expose cache fields today,
// but pick up cache_read_input_tokens / cache_creation_input_tokens
// if the event shape grows them so cost tracking stays accurate.
const cachedTokens = metrics.cacheReadInputTokens || metrics.cache_read_input_tokens || 0;
const cacheCreationInputTokens = metrics.cacheCreationInputTokens || metrics.cache_creation_input_tokens || 0;
if (inputTokens > 0 || outputTokens > 0) {
state.usage = {
@@ -355,6 +403,12 @@ export class KiroExecutor extends BaseExecutor {
completion_tokens: outputTokens,
total_tokens: inputTokens + outputTokens
};
// Kiro is Claude-backed: inputTokens EXCLUDES cache (Claude convention),
// not inclusive like OpenAI's cached_tokens. Emit cache_read_input_tokens
// (not cached_tokens) so canonicalizeUsage takes the Claude fold path and
// correctly adds cache back into prompt_tokens instead of undercharging.
if (cachedTokens > 0) state.usage.cache_read_input_tokens = cachedTokens;
if (cacheCreationInputTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationInputTokens;
}
}
}