Merge remote-tracking branch 'origin/master' into gitea/new_feature

# Conflicts:
#	open-sse/handlers/chatCore.js
#	open-sse/services/combo.js
#	src/app/(dashboard)/dashboard/profile/page.js
#	src/app/api/v1/models/route.js
#	src/lib/db/repos/settingsRepo.js
This commit is contained in:
2026-08-17 00:21:51 +07:00
103 changed files with 11250 additions and 3514 deletions

View File

@@ -62,6 +62,19 @@ function stripOpenAI(body, caps) {
if (!Array.isArray(body.messages)) return;
const last = body.messages.length - 1;
body.messages.forEach((msg, i) => {
if (caps.vision === false) {
if (Array.isArray(msg.images)) delete msg.images;
if (Array.isArray(msg.experimental_attachments)) {
msg.experimental_attachments = msg.experimental_attachments.filter(
(a) => !(a?.contentType?.startsWith("image/") || (typeof a?.url === "string" && a.url.startsWith("data:image/")))
);
}
if (Array.isArray(msg.attachments)) {
msg.attachments = msg.attachments.filter(
(a) => !(a?.contentType?.startsWith("image/") || (typeof a?.url === "string" && a.url.startsWith("data:image/")))
);
}
}
if (!Array.isArray(msg.content)) return;
const removed = new Set();
msg.content = filterBlocks(msg.content, capForOpenAIBlock, caps, removed, i === last);

View File

@@ -9,6 +9,9 @@ import { PROVIDERS } from "../../providers/index.js";
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
const CACHE_CONTROL_5M = { type: "ephemeral" };
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
// Check if message has valid non-empty content
export function hasValidContent(msg) {
if (typeof msg.content === "string" && msg.content.trim()) return true;
@@ -124,32 +127,38 @@ export function normalizeClaudePassthrough(body, model = "") {
if (Object.keys(body.output_config).length === 0) delete body.output_config;
}
// 2. Hoist mid-conversation system messages into the top-level system field
// 2. Fold mid-conversation system messages into the neighbouring turn.
// Hoisting them into body.system would insert volatile content (token counters,
// reminders) ahead of the whole conversation and invalidate the prefix cache on
// every request. Folding in place keeps the cached prefix stable.
if (Array.isArray(body.messages)) {
const systemBlocks = [];
const messages = [];
for (const msg of body.messages) {
if (msg.role === ROLE.SYSTEM) {
const text = typeof msg.content === "string"
? msg.content
: Array.isArray(msg.content)
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
: "";
if (text.trim()) systemBlocks.push({ type: CLAUDE_BLOCK.TEXT, text });
if (msg.role !== ROLE.SYSTEM) {
messages.push(msg);
continue;
}
messages.push(msg);
}
const text = typeof msg.content === "string"
? msg.content
: Array.isArray(msg.content)
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
: "";
if (!text.trim()) continue;
if (systemBlocks.length > 0) {
const existing = Array.isArray(body.system)
? body.system
: typeof body.system === "string" && body.system.trim()
? [{ type: "text", text: body.system }]
: [];
body.system = [...existing, ...systemBlocks];
body.messages = messages;
// Copy-on-write: the caller's body is reused across account-fallback
// attempts, so folding must never mutate the original message.
const block = { type: CLAUDE_BLOCK.TEXT, text };
const prev = messages[messages.length - 1];
if (prev?.role === ROLE.USER) {
const content = typeof prev.content === "string"
? [{ type: CLAUDE_BLOCK.TEXT, text: prev.content }]
: Array.isArray(prev.content) ? [...prev.content] : [];
messages[messages.length - 1] = { ...prev, content: [...content, block] };
continue;
}
messages.push({ role: ROLE.USER, content: [block] });
}
body.messages = messages;
}
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
@@ -182,6 +191,70 @@ export function normalizeClaudePassthrough(body, model = "") {
return body;
}
// Put a 5m breakpoint on the last cache-eligible block of a message.
// thinking/redacted_thinking blocks do not accept cache_control.
function markLastCacheableBlock(msg) {
if (!Array.isArray(msg?.content)) return false;
for (let i = msg.content.length - 1; i >= 0; i--) {
const block = msg.content[i];
if (typeof block !== "object" || block === null) continue;
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
block.cache_control = { ...CACHE_CONTROL_5M };
return true;
}
return false;
}
// Re-anchor cache breakpoints on a Claude passthrough body (same policy as
// prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m.
// The client's own markers point at pre-normalization offsets, so they are dropped.
// Must run LAST, after every step that can reshape system/tools/messages
// (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail.
export function anchorClaudeCache(body) {
if (!body || typeof body !== "object") return body;
if (Array.isArray(body.system)) {
const last = body.system.length - 1;
body.system.forEach((block, i) => {
if (typeof block !== "object" || block === null) return;
if (i === last) block.cache_control = { ...CACHE_CONTROL_1H };
else delete block.cache_control;
});
}
if (Array.isArray(body.tools)) {
const last = body.tools.length - 1;
body.tools.forEach((tool, i) => {
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
else delete tool.cache_control;
});
}
if (Array.isArray(body.messages)) {
let anchored = null;
for (let i = body.messages.length - 1; i >= 0; i--) {
const msg = body.messages[i];
if (!Array.isArray(msg.content)) continue;
for (const block of msg.content) delete block.cache_control;
// Prefer the last assistant turn: it ends a completed exchange, so the
// prefix up to it stays byte-stable across the following requests.
if (anchored || msg.role !== ROLE.ASSISTANT) continue;
anchored = markLastCacheableBlock(msg);
}
// First turn of a conversation has no assistant yet — anchor the final
// message instead, so the opening prompt is cached rather than paid twice.
if (!anchored) {
for (let i = body.messages.length - 1; i >= 0 && !anchored; i--) {
anchored = markLastCacheableBlock(body.messages[i]);
}
}
}
return body;
}
// Prepare request for Claude format endpoints
// - Cleanup cache_control
// - Filter empty messages

View File

@@ -287,6 +287,18 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
toolSpecs,
nameMap,
});
// canonicalizeKiroConversation() already ran its second-chance repair (flatten
// every structured tool turn to text, then re-validate). A body that is STILL
// invalid here cannot be made shippable, and Kiro answers it with
// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"}.
// Fail locally instead: chatCore turns a falsy return into a 400 without
// spending an upstream call or a per-account cooldown. The taxonomy
// (role:N | pair:N | id:N | spec:N | orphan:0 | current) names the offending
// turn so the shape can be diagnosed from the log alone.
if (!canonical.valid) {
console.error(`[Kiro] refusing invalid conversation (claude → kiro): ${(canonical.errors || []).join(", ") || "unknown"} | turns=${(canonical.history || []).length + 1}`);
return null;
}
const replayCurrent = canonical.currentMessage.userInputMessage;
const userInputMessage = {
content: replayCurrent.content || "",

View File

@@ -421,6 +421,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
if (body.reasoning !== undefined) result.reasoning = body.reasoning;
if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" };
if (body.service_tier !== undefined) result.service_tier = body.service_tier;
if (body.prompt_cache_key !== undefined) result.prompt_cache_key = body.prompt_cache_key;
return result;
}

View File

@@ -379,6 +379,18 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
toolSpecs,
nameMap,
});
// canonicalizeKiroConversation() already ran its second-chance repair (flatten
// every structured tool turn to text, then re-validate). A body that is STILL
// invalid here cannot be made shippable, and Kiro answers it with
// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"}.
// Fail locally instead: chatCore turns a falsy return into a 400 without
// spending an upstream call or a per-account cooldown. The taxonomy
// (role:N | pair:N | id:N | spec:N | orphan:0 | current) names the offending
// turn so the shape can be diagnosed from the log alone.
if (!canonical.valid) {
console.error(`[Kiro] refusing invalid conversation (openai → kiro): ${(canonical.errors || []).join(", ") || "unknown"} | turns=${(canonical.history || []).length + 1}`);
return null;
}
const replayCurrent = canonical.currentMessage.userInputMessage;
const payload = {

View File

@@ -75,6 +75,15 @@ export function kiroToClaudeResponse(chunk, state) {
? data.usage.completion_tokens
: 0;
state.usage = { input_tokens: promptTokens, output_tokens: outputTokens };
// Claude clients read cache_read/cache_creation to price a turn and to size
// their prompt cache. Both spellings are accepted because the Kiro executor
// emits the Chat shape and passthrough responses use the nested details form.
const cacheRead = data.usage.cache_read_input_tokens
?? data.usage.prompt_tokens_details?.cached_tokens;
const cacheCreation = data.usage.cache_creation_input_tokens
?? data.usage.prompt_tokens_details?.cache_creation_tokens;
if (typeof cacheRead === "number") state.usage.cache_read_input_tokens = cacheRead;
if (typeof cacheCreation === "number") state.usage.cache_creation_input_tokens = cacheCreation;
}
// First chunk → emit message_start.
@@ -254,6 +263,13 @@ export function kiroToClaudeNonStreaming(data) {
usage: {
input_tokens: usage.prompt_tokens || 0,
output_tokens: usage.completion_tokens || 0,
// Same cache preservation as the streaming path above.
...(typeof (usage.cache_read_input_tokens ?? usage.prompt_tokens_details?.cached_tokens) === "number"
? { cache_read_input_tokens: usage.cache_read_input_tokens ?? usage.prompt_tokens_details.cached_tokens }
: {}),
...(typeof (usage.cache_creation_input_tokens ?? usage.prompt_tokens_details?.cache_creation_tokens) === "number"
? { cache_creation_input_tokens: usage.cache_creation_input_tokens ?? usage.prompt_tokens_details.cache_creation_tokens }
: {}),
},
};
}

View File

@@ -99,8 +99,8 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
}
}
// Handle tool_calls
if (delta.tool_calls) {
// Handle tool_calls (empty array is truthy; require a real call)
if (delta.tool_calls && delta.tool_calls.length) {
closeMessage(state, emit, idx);
for (const tc of delta.tool_calls) {
emitToolCall(state, emit, tc);