This commit is contained in:
2026-07-06 00:01:34 +07:00
parent 2729408ef3
commit e7470e955e
108 changed files with 4131 additions and 529 deletions

View File

@@ -64,9 +64,22 @@ export class KiroExecutor extends BaseExecutor {
getOrderedBaseUrls(credentials) {
const baseUrls = this.getBaseUrls();
const authMethod = credentials?.providerSpecificData?.authMethod;
const isCodeWhispererSurface = authMethod === "api_key" || authMethod === "external_idp";
// IAM Identity Center (idc) tokens are AWS SSO access tokens — the same
// family as external_idp/api_key. The kiro.dev gateway rejects them with
// 403 "bearer token invalid", so they must hit the CodeWhisperer
// *.amazonaws.com surface, and in the region the token was minted in
// (the baseUrls are hardcoded us-east-1).
const isCodeWhispererSurface =
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
if (!isCodeWhispererSurface) return baseUrls;
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com"));
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
const regionalize = (u) =>
region && region !== "us-east-1" && u.includes("amazonaws.com")
? u.replace(/([a-z]+)\.[a-z0-9-]+\.amazonaws\.com/, `$1.${region}.amazonaws.com`)
: u;
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
}
@@ -122,7 +135,8 @@ export class KiroExecutor extends BaseExecutor {
hasReasoningContent: false,
reasoningChunkCount: 0,
toolCallIndex: 0,
seenToolIds: new Map()
seenToolIds: new Map(),
inThinking: false
};
const transformStream = new TransformStream({
@@ -159,7 +173,36 @@ export class KiroExecutor extends BaseExecutor {
// Handle assistantResponseEvent
if (eventType === "assistantResponseEvent" && event.payload?.content) {
const content = event.payload.content;
let content = event.payload.content;
// Kiro Claude models can leak <thinking> blocks into the content stream.
// We strip these literal tags to prevent duplication, as the reasoning
// is already routed correctly via reasoningContentEvent.
if (state.inThinking) {
if (content.includes("</thinking>")) {
state.inThinking = false;
const after = content.split("</thinking>").slice(1).join("</thinking>");
content = after.startsWith("\n") ? after.substring(1) : after;
} else {
content = ""; // Drop entirely while inside thinking block
}
} else if (content.includes("<thinking>")) {
state.inThinking = true;
if (content.includes("</thinking>")) {
state.inThinking = false;
const before = content.split("<thinking>")[0];
const after = content.split("</thinking>").slice(1).join("</thinking>");
content = before + (after.startsWith("\n") ? after.substring(1) : after);
} else {
content = content.split("<thinking>")[0];
}
}
if (!content && state.hasReasoningContent) {
// If we stripped everything, skip emitting an empty content chunk
continue;
}
state.totalContentLength += content.length;
const chunk = {
@@ -348,6 +391,11 @@ export class KiroExecutor extends BaseExecutor {
if (metrics && typeof metrics === 'object') {
const inputTokens = metrics.inputTokens || 0;
const outputTokens = metrics.outputTokens || 0;
// ponytail: Amazon Q upstream does not expose cache fields today,
// but pick up cache_read_input_tokens / cache_creation_input_tokens
// if the event shape grows them so cost tracking stays accurate.
const cachedTokens = metrics.cacheReadInputTokens || metrics.cache_read_input_tokens || 0;
const cacheCreationInputTokens = metrics.cacheCreationInputTokens || metrics.cache_creation_input_tokens || 0;
if (inputTokens > 0 || outputTokens > 0) {
state.usage = {
@@ -355,6 +403,12 @@ export class KiroExecutor extends BaseExecutor {
completion_tokens: outputTokens,
total_tokens: inputTokens + outputTokens
};
// Kiro is Claude-backed: inputTokens EXCLUDES cache (Claude convention),
// not inclusive like OpenAI's cached_tokens. Emit cache_read_input_tokens
// (not cached_tokens) so canonicalizeUsage takes the Claude fold path and
// correctly adds cache back into prompt_tokens instead of undercharging.
if (cachedTokens > 0) state.usage.cache_read_input_tokens = cachedTokens;
if (cacheCreationInputTokens > 0) state.usage.cache_creation_input_tokens = cacheCreationInputTokens;
}
}
}