Merge origin/master (v0.5.91) into gitea/new_feature

Resolve conflicts:
- streamingHandler.js: adopt upstreamResponseHeaders while keeping 0-token detail row avoidance
- capabilities.js: preserve user-asserted caps and globalThis slots without local caching of catalogSource
- AddCustomModelModal.js & providers/[id]/page.js: wire STT transport marker with custom model edits/assertions
- models/custom/route.js & aliasRepo.js: persist custom model transport and invalidate user caps
- usageRepo.js: key byApiKey live stats by full API key and keep tail in maskApiKey
- UsageStats.js: lazy load charts dynamically
This commit is contained in:
2026-09-28 21:17:33 +07:00
123 changed files with 5882 additions and 411 deletions

View File

@@ -102,9 +102,28 @@ export function extractThinking(body) {
return null;
}
// Capture thinking intent from a body. Alias of extractThinking, named for clarity
// at the call-site where intent is snapshotted before format translation.
export const captureThinking = extractThinking;
// Capture thinking intent from a body before format translation strips it.
// Besides the effort, records whether an OpenAI-shaped client wants the thinking
// text itself: Claude returns it only with thinking.display "summarized", a field
// OpenAI has no equivalent for, so the intent cannot survive translation on its own.
export function captureThinking(body) {
const cfg = extractThinking(body);
if (!cfg || cfg.mode === "none") return cfg;
const display = openAIThinkingDisplay(body);
return display ? { ...cfg, display } : cfg;
}
function openAIThinkingDisplay(body) {
// Responses API: reasoning.summary is the explicit request for reasoning text.
if (body.reasoning && typeof body.reasoning === "object") {
const summary = body.reasoning.summary;
return typeof summary === "string" && summary && summary !== "none" ? "summarized" : undefined;
}
// Chat Completions has no summary knob. A client setting reasoning_effort is
// asking for reasoning, and reasoning_content is how it would receive it.
if (typeof body.reasoning_effort === "string") return "summarized";
return undefined;
}
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
@@ -302,9 +321,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
case "deepseek": {
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
body.thinking = { type: "enabled" };
// DeepSeek: low/medium→high, xhigh/max→max.
// DeepSeek: low/medium→high, xhigh/max→max. Some backends (mimo v2.5-pro/v2.6
// on opencode-go, probed live) 400 on "max" — clamp to high when the declared
// levels exclude it.
const level = toLevel(eff);
body.reasoning_effort = level === "xhigh" || level === "max" ? "max" : "high";
const want = level === "xhigh" || level === "max" ? "max" : "high";
body.reasoning_effort = want === "max" && supportedLevels && !supportedLevels.includes("max") ? "high" : want;
break;
}
case "kimi": {
@@ -380,7 +402,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
const supportedLevels = getThinkingLevels(provider, cleanModel);
// Anthropic's `display` (summarized | omitted) decides whether thinking text
// comes back at all; keep what the client asked for instead of resetting it.
const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined;
// An OpenAI-shaped client's ask arrives via the captured intent instead.
const display = typeof body.thinking?.display === "string" ? body.thinking.display : intent?.display;
stripAll(body);
applyFormat(fmt, body, cfg, caps, supportedLevels, display);
return body;

View File

@@ -432,7 +432,7 @@ export function cleanJSONSchemaForAntigravity(schema) {
return cleaned;
}
// Merge adjacent same-role messages, strip empty parts, ensure initial user turn
// Merge adjacent same-role messages, strip empty parts, ensure initial and terminal user turns
export function normalizeGeminiContents(contents) {
const out = [];
for (const c of contents || []) {
@@ -446,6 +446,23 @@ export function normalizeGeminiContents(contents) {
if (out.length > 0 && out[0].role !== "user") {
out.unshift({ role: "user", parts: [{ text: "..." }] });
}
if (out.length > 0 && out.at(-1).role === "model") {
const fnCalls = (out.at(-1).parts || []).filter(p => p && p.functionCall);
if (fnCalls.length > 0) {
const responses = fnCalls.map(p => {
const call = p.functionCall || {};
const fr = {
name: call.name || "tool",
response: { result: "Continue." }
};
if (call.id) fr.id = call.id;
return { functionResponse: fr };
});
out.push({ role: "user", parts: responses });
} else {
out.push({ role: "user", parts: [{ text: "Continue." }] });
}
}
return out;
}

View File

@@ -59,10 +59,6 @@ export function claudeToOpenAIResponse(chunk, state) {
}
if (block?.type === CLAUDE_BLOCK.TEXT) {
state.textBlockStarted = true;
} else if (block?.type === CLAUDE_BLOCK.THINKING) {
state.inThinkingBlock = true;
state.currentBlockIndex = chunk.index;
results.push(createChunk(state, { content: "<think>" }));
} else if (block?.type === CLAUDE_BLOCK.TOOL_USE) {
const toolCallIndex = state.toolCallIndex++;
// Restore original tool name from mapping (Claude OAuth)
@@ -89,6 +85,8 @@ export function claudeToOpenAIResponse(chunk, state) {
if (delta?.type === "text_delta" && delta.text) {
results.push(createChunk(state, { content: delta.text }));
} else if (delta?.type === "thinking_delta" && delta.thinking) {
// Thinking travels only in reasoning_content. No "<think>" markers in
// content: OpenAI-format clients render them as literal text.
results.push(createChunk(state, reasoningDelta(delta.thinking)));
} else if (delta?.type === "input_json_delta" && delta.partial_json) {
const toolCall = state.toolCalls.get(chunk.index);
@@ -112,10 +110,6 @@ export function claudeToOpenAIResponse(chunk, state) {
state.serverToolBlockIndex = -1;
break;
}
if (state.inThinkingBlock && chunk.index === state.currentBlockIndex) {
results.push(createChunk(state, { content: "</think>" }));
state.inThinkingBlock = false;
}
state.textBlockStarted = false;
state.thinkingBlockStarted = false;
break;

View File

@@ -129,12 +129,16 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
}
if (content) {
// The answer starts, so thinking is over. Upstreams that send reasoning via
// reasoning_content never emit "</think>", so close it here rather than at finish.
closeReasoning(state, emit);
emitTextContent(state, emit, idx, content);
}
}
// Handle tool_calls (empty array is truthy; require a real call)
if (delta.tool_calls && delta.tool_calls.length) {
closeReasoning(state, emit);
closeMessage(state, emit, idx);
for (const tc of delta.tool_calls) {
emitToolCall(state, emit, tc);
@@ -219,15 +223,19 @@ function closeReasoning(state, emit) {
part: { type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }
});
const item = {
id: state.reasoningId,
type: RESPONSES_ITEM.REASONING,
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
};
emit("response.output_item.done", {
type: "response.output_item.done",
output_index: state.reasoningIndex,
item: {
id: state.reasoningId,
type: RESPONSES_ITEM.REASONING,
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
}
item
});
recordCompletedOutputItem(state, state.reasoningIndex, item);
}
}
@@ -291,16 +299,20 @@ function closeMessage(state, emit, idx) {
part: { type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }
});
const item = {
id: msgId,
type: RESPONSES_ITEM.MESSAGE,
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
role: ROLE.ASSISTANT
};
emit("response.output_item.done", {
type: "response.output_item.done",
output_index: parseInt(idx),
item: {
id: msgId,
type: RESPONSES_ITEM.MESSAGE,
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
role: ROLE.ASSISTANT
}
item
});
recordCompletedOutputItem(state, parseInt(idx), item);
}
}
@@ -394,23 +406,51 @@ function closeToolCall(state, emit, idx) {
});
}
const item = {
id: `${custom ? "ctc" : "fc"}_${callId}`,
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
call_id: callId,
name: state.funcNames[idx] || ""
};
emit("response.output_item.done", {
type: "response.output_item.done",
output_index: parseInt(idx),
item: {
id: `${custom ? "ctc" : "fc"}_${callId}`,
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
call_id: callId,
name: state.funcNames[idx] || ""
}
item
});
recordCompletedOutputItem(state, parseInt(idx), item);
state.funcItemDone[idx] = true;
state.funcArgsDone[idx] = true;
}
}
// response.completed carries the finished Response object, so response.output has
// to repeat the items already delivered in response.output_item.done. Clients that
// build their final result from the terminal event (GitHub Copilot CLI, the OpenAI
// SDK "final response" helpers) otherwise treat the turn as empty even though the
// text was streamed - see issue #4307.
//
// Keyed by output_index so a repeated close overwrites rather than duplicating the
// item, and ordered by output_index so response.output matches the order the items
// were emitted in. Lazily created because stream.js can hand us a state it built
// itself rather than one from initState().
function recordCompletedOutputItem(state, outputIndex, item) {
state.completedOutputItems ??= new Map();
const index = Number.isInteger(outputIndex) ? outputIndex : Number.parseInt(outputIndex, 10) || 0;
state.completedOutputItems.set(index, item);
}
function collectCompletedOutputItems(state) {
const recorded = state.completedOutputItems;
if (!(recorded instanceof Map) || recorded.size === 0) return [];
return [...recorded.entries()]
.sort((left, right) => left[0] - right[0])
.map(([, item]) => item);
}
function sendCompleted(state, emit) {
if (!state.completedSent) {
state.completedSent = true;
@@ -423,6 +463,7 @@ function sendCompleted(state, emit) {
status: "completed",
background: false,
error: null,
output: collectCompletedOutputItems(state),
...(state.responsesUsage ? { usage: state.responsesUsage } : {})
}
});