fix(kiro): improve direct session cache reuse

Reshape Kiro direct requests so resumed client sessions reuse Kiro's
cache-affinity fields instead of starting unrelated CodeWhisperer
conversations.

- keep conversationState.conversationId stable when the client sends an
  explicit session id (x-session-id, session_id, conversation_id, Claude
  Code session metadata)
- add a stable conversationState.agentContinuationId per Kiro session
- send conversationState.agentTaskType: "vibe" and agentMode: "vibe",
  matching the normal Kiro CLI/KAS chat path
- move Kiro thinking instructions into Kiro-compatible systemPrompt /
  additionalModelRequestFields instead of generic top-level thinking
- keep volatile timestamp context out of the top-level systemPrompt; it
  remains only in user content fallback
- suppress additionalModelRequestFields for legacy 4.5-era Claude/Kiro
  models that reject it, while defaulting future Claude/Kiro model ids
  to supported
- preserve Kiro meteringEvent credit usage internally for accounting
  without leaking provider-specific fields into OpenAI-compatible usage
- prevent unrelated headerless Kiro requests from sharing one
  connection-wide continuation
- cap/evict continuation sessions so long-running processes do not grow
  the continuation map unbounded
- treat generated headerless Kiro sessions as one-shot so they do not
  evict real explicit-session continuations
- keep credit-only Kiro metering valid for internal persistence when
  token metrics are unavailable
This commit is contained in:
Edison42
2026-07-16 15:13:28 +07:00
committed by decolua
parent 70e8dc4974
commit 9c58ba645e
10 changed files with 786 additions and 90 deletions

View File

@@ -11,6 +11,7 @@ import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to
const contentOf = (result) =>
result.conversationState.currentMessage.userInputMessage.content;
const systemPromptOf = (result) => result.systemPrompt || "";
describe("openaiToKiroRequest", () => {
describe("basic message conversion", () => {
@@ -293,7 +294,11 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6", body, true, {});
expect(contentOf(result)).toContain("<max_thinking_length>1024</max_thinking_length>");
expect(systemPromptOf(result)).toContain("<max_thinking_length>1024</max_thinking_length>");
expect(result.additionalModelRequestFields).toEqual({
thinking: { type: "adaptive", display: "summarized" },
output_config: { effort: "low" },
});
});
it("maps reasoning_effort high to max_thinking_length 24576", () => {
@@ -304,7 +309,97 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6", body, true, {});
expect(contentOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toEqual({
thinking: { type: "adaptive", display: "summarized" },
output_config: { effort: "high" },
});
});
it("does not send additionalModelRequestFields for legacy Kiro model ids", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Legacy model id should not get adaptive fields" }]
};
const result = openaiToKiroRequest("claude-sonnet-4.5", body, true, {});
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
it("does not send additionalModelRequestFields for date-suffixed Claude 4 model ids", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Date-suffixed Claude 4 should stay legacy" }]
};
const result = openaiToKiroRequest("claude-sonnet-4-20250514", body, true, {});
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
it("does not send additionalModelRequestFields for pre-4 legacy Kiro model ids", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Older model id should not get adaptive fields" }]
};
const result = openaiToKiroRequest("claude-sonnet-3.7", body, true, {});
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
it("does not send additionalModelRequestFields for prefixed pre-4 legacy Kiro model ids", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Prefixed older model id should not get adaptive fields" }]
};
const result = openaiToKiroRequest("kiro/claude-3-7-sonnet-20250219", body, true, {});
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
it("does not send Claude-specific additionalModelRequestFields for prefixed non-Claude aliases", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Prefixed non-Claude alias should not get adaptive fields" }]
};
const result = openaiToKiroRequest("kiro/gpt-4o", body, true, {});
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
it("does not send Claude-specific additionalModelRequestFields for non-Claude aliases", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Non-Claude aliases should not get Claude adaptive fields" }]
};
const result = openaiToKiroRequest("gpt-4o", body, true, {});
expect(systemPromptOf(result)).toContain("<max_thinking_length>24576</max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
it("defaults future Kiro model ids to additionalModelRequestFields support", () => {
const body = {
reasoning_effort: "high",
messages: [{ role: "user", content: "Future model id should get adaptive fields" }]
};
const result = openaiToKiroRequest("claude-sonnet-4.60", body, true, {});
expect(result.additionalModelRequestFields).toEqual({
thinking: { type: "adaptive", display: "summarized" },
output_config: { effort: "high" },
});
});
it("clamps reasoning_effort max to Kiro max_thinking_length 32000", () => {
@@ -315,7 +410,8 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6", body, true, {});
expect(contentOf(result)).toContain("<max_thinking_length>32000</max_thinking_length>");
expect(systemPromptOf(result)).toContain("<max_thinking_length>32000</max_thinking_length>");
expect(result.additionalModelRequestFields?.output_config?.effort).toBe("high");
});
it("clamps OpenAI Responses reasoning.effort xhigh to max_thinking_length 32000", () => {
@@ -326,7 +422,8 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6", body, true, {});
expect(contentOf(result)).toContain("<max_thinking_length>32000</max_thinking_length>");
expect(systemPromptOf(result)).toContain("<max_thinking_length>32000</max_thinking_length>");
expect(result.additionalModelRequestFields?.output_config?.effort).toBe("high");
});
it("uses Claude thinking.budget_tokens as max_thinking_length", () => {
@@ -337,7 +434,7 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6", body, true, {});
expect(contentOf(result)).toContain("<max_thinking_length>4096</max_thinking_length>");
expect(systemPromptOf(result)).toContain("<max_thinking_length>4096</max_thinking_length>");
});
it("uses the default budget for synthetic -thinking models with no explicit config", () => {
@@ -347,7 +444,54 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6-thinking", body, true, {});
expect(contentOf(result)).toContain("<max_thinking_length>16000</max_thinking_length>");
expect(systemPromptOf(result)).toContain("<max_thinking_length>16000</max_thinking_length>");
});
it("keeps top-level systemPrompt stable across turns", () => {
const first = openaiToKiroRequest(
"claude-sonnet-4.6-thinking",
{ messages: [{ role: "user", content: "first" }] },
true,
{}
);
const second = openaiToKiroRequest(
"claude-sonnet-4.6-thinking",
{ messages: [{ role: "user", content: "second" }] },
true,
{}
);
expect(first.systemPrompt).toBe(second.systemPrompt);
expect(first.systemPrompt).not.toContain("Current time");
expect(first.conversationState.currentMessage.userInputMessage.content).toContain("Current time");
});
it("replays frozen msg0 for explicit Kiro sessions while keeping current time fresh", () => {
const credentials = {
connectionId: "kiro-account-openai-replay",
rawHeaders: { "x-session-id": "hermes-session-openai-replay" },
};
const first = openaiToKiroRequest(
"claude-sonnet-4.6",
{ messages: [{ role: "user", content: "first turn" }] },
true,
credentials
);
const second = openaiToKiroRequest(
"claude-sonnet-4.6",
{ messages: [{ role: "user", content: "second turn" }] },
true,
credentials
);
expect(second.conversationState.conversationId).toBe("hermes-session-openai-replay");
expect(second.conversationState.agentContinuationId).toBe(first.conversationState.agentContinuationId);
expect(second.conversationState.history[0].userInputMessage.content).toBe(
first.conversationState.currentMessage.userInputMessage.content
);
expect(second.conversationState.history[0].userInputMessage.modelId).toBe("claude-sonnet-4.6");
expect(second.conversationState.currentMessage.userInputMessage.content).toContain("Current time");
expect(second.conversationState.currentMessage.userInputMessage.content).toContain("second turn");
});
it("does not inject thinking prefix for reasoning_effort none", () => {
@@ -358,8 +502,9 @@ describe("openaiToKiroRequest", () => {
const result = openaiToKiroRequest("claude-sonnet-4.6", body, true, {});
expect(contentOf(result)).not.toContain("<thinking_mode>enabled</thinking_mode>");
expect(contentOf(result)).not.toContain("<max_thinking_length>");
expect(systemPromptOf(result)).not.toContain("<thinking_mode>enabled</thinking_mode>");
expect(systemPromptOf(result)).not.toContain("<max_thinking_length>");
expect(result.additionalModelRequestFields).toBeUndefined();
});
});
});