Merge origin/master (v0.5.91) into gitea/new_feature
Resolve conflicts: - streamingHandler.js: adopt upstreamResponseHeaders while keeping 0-token detail row avoidance - capabilities.js: preserve user-asserted caps and globalThis slots without local caching of catalogSource - AddCustomModelModal.js & providers/[id]/page.js: wire STT transport marker with custom model edits/assertions - models/custom/route.js & aliasRepo.js: persist custom model transport and invalidate user caps - usageRepo.js: key byApiKey live stats by full API key and keep tail in maskApiKey - UsageStats.js: lazy load charts dynamically
This commit is contained in:
@@ -102,9 +102,28 @@ export function extractThinking(body) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Capture thinking intent from a body. Alias of extractThinking, named for clarity
|
||||
// at the call-site where intent is snapshotted before format translation.
|
||||
export const captureThinking = extractThinking;
|
||||
// Capture thinking intent from a body before format translation strips it.
|
||||
// Besides the effort, records whether an OpenAI-shaped client wants the thinking
|
||||
// text itself: Claude returns it only with thinking.display "summarized", a field
|
||||
// OpenAI has no equivalent for, so the intent cannot survive translation on its own.
|
||||
export function captureThinking(body) {
|
||||
const cfg = extractThinking(body);
|
||||
if (!cfg || cfg.mode === "none") return cfg;
|
||||
const display = openAIThinkingDisplay(body);
|
||||
return display ? { ...cfg, display } : cfg;
|
||||
}
|
||||
|
||||
function openAIThinkingDisplay(body) {
|
||||
// Responses API: reasoning.summary is the explicit request for reasoning text.
|
||||
if (body.reasoning && typeof body.reasoning === "object") {
|
||||
const summary = body.reasoning.summary;
|
||||
return typeof summary === "string" && summary && summary !== "none" ? "summarized" : undefined;
|
||||
}
|
||||
// Chat Completions has no summary knob. A client setting reasoning_effort is
|
||||
// asking for reasoning, and reasoning_content is how it would receive it.
|
||||
if (typeof body.reasoning_effort === "string") return "summarized";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
|
||||
|
||||
@@ -302,9 +321,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
|
||||
case "deepseek": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
body.thinking = { type: "enabled" };
|
||||
// DeepSeek: low/medium→high, xhigh/max→max.
|
||||
// DeepSeek: low/medium→high, xhigh/max→max. Some backends (mimo v2.5-pro/v2.6
|
||||
// on opencode-go, probed live) 400 on "max" — clamp to high when the declared
|
||||
// levels exclude it.
|
||||
const level = toLevel(eff);
|
||||
body.reasoning_effort = level === "xhigh" || level === "max" ? "max" : "high";
|
||||
const want = level === "xhigh" || level === "max" ? "max" : "high";
|
||||
body.reasoning_effort = want === "max" && supportedLevels && !supportedLevels.includes("max") ? "high" : want;
|
||||
break;
|
||||
}
|
||||
case "kimi": {
|
||||
@@ -380,7 +402,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
|
||||
const supportedLevels = getThinkingLevels(provider, cleanModel);
|
||||
// Anthropic's `display` (summarized | omitted) decides whether thinking text
|
||||
// comes back at all; keep what the client asked for instead of resetting it.
|
||||
const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined;
|
||||
// An OpenAI-shaped client's ask arrives via the captured intent instead.
|
||||
const display = typeof body.thinking?.display === "string" ? body.thinking.display : intent?.display;
|
||||
stripAll(body);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels, display);
|
||||
return body;
|
||||
|
||||
@@ -432,7 +432,7 @@ export function cleanJSONSchemaForAntigravity(schema) {
|
||||
return cleaned;
|
||||
}
|
||||
|
||||
// Merge adjacent same-role messages, strip empty parts, ensure initial user turn
|
||||
// Merge adjacent same-role messages, strip empty parts, ensure initial and terminal user turns
|
||||
export function normalizeGeminiContents(contents) {
|
||||
const out = [];
|
||||
for (const c of contents || []) {
|
||||
@@ -446,6 +446,23 @@ export function normalizeGeminiContents(contents) {
|
||||
if (out.length > 0 && out[0].role !== "user") {
|
||||
out.unshift({ role: "user", parts: [{ text: "..." }] });
|
||||
}
|
||||
if (out.length > 0 && out.at(-1).role === "model") {
|
||||
const fnCalls = (out.at(-1).parts || []).filter(p => p && p.functionCall);
|
||||
if (fnCalls.length > 0) {
|
||||
const responses = fnCalls.map(p => {
|
||||
const call = p.functionCall || {};
|
||||
const fr = {
|
||||
name: call.name || "tool",
|
||||
response: { result: "Continue." }
|
||||
};
|
||||
if (call.id) fr.id = call.id;
|
||||
return { functionResponse: fr };
|
||||
});
|
||||
out.push({ role: "user", parts: responses });
|
||||
} else {
|
||||
out.push({ role: "user", parts: [{ text: "Continue." }] });
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
|
||||
@@ -59,10 +59,6 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
}
|
||||
if (block?.type === CLAUDE_BLOCK.TEXT) {
|
||||
state.textBlockStarted = true;
|
||||
} else if (block?.type === CLAUDE_BLOCK.THINKING) {
|
||||
state.inThinkingBlock = true;
|
||||
state.currentBlockIndex = chunk.index;
|
||||
results.push(createChunk(state, { content: "<think>" }));
|
||||
} else if (block?.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||
const toolCallIndex = state.toolCallIndex++;
|
||||
// Restore original tool name from mapping (Claude OAuth)
|
||||
@@ -89,6 +85,8 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
if (delta?.type === "text_delta" && delta.text) {
|
||||
results.push(createChunk(state, { content: delta.text }));
|
||||
} else if (delta?.type === "thinking_delta" && delta.thinking) {
|
||||
// Thinking travels only in reasoning_content. No "<think>" markers in
|
||||
// content: OpenAI-format clients render them as literal text.
|
||||
results.push(createChunk(state, reasoningDelta(delta.thinking)));
|
||||
} else if (delta?.type === "input_json_delta" && delta.partial_json) {
|
||||
const toolCall = state.toolCalls.get(chunk.index);
|
||||
@@ -112,10 +110,6 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
state.serverToolBlockIndex = -1;
|
||||
break;
|
||||
}
|
||||
if (state.inThinkingBlock && chunk.index === state.currentBlockIndex) {
|
||||
results.push(createChunk(state, { content: "</think>" }));
|
||||
state.inThinkingBlock = false;
|
||||
}
|
||||
state.textBlockStarted = false;
|
||||
state.thinkingBlockStarted = false;
|
||||
break;
|
||||
|
||||
@@ -129,12 +129,16 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
||||
}
|
||||
|
||||
if (content) {
|
||||
// The answer starts, so thinking is over. Upstreams that send reasoning via
|
||||
// reasoning_content never emit "</think>", so close it here rather than at finish.
|
||||
closeReasoning(state, emit);
|
||||
emitTextContent(state, emit, idx, content);
|
||||
}
|
||||
}
|
||||
|
||||
// Handle tool_calls (empty array is truthy; require a real call)
|
||||
if (delta.tool_calls && delta.tool_calls.length) {
|
||||
closeReasoning(state, emit);
|
||||
closeMessage(state, emit, idx);
|
||||
for (const tc of delta.tool_calls) {
|
||||
emitToolCall(state, emit, tc);
|
||||
@@ -219,15 +223,19 @@ function closeReasoning(state, emit) {
|
||||
part: { type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }
|
||||
});
|
||||
|
||||
const item = {
|
||||
id: state.reasoningId,
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
|
||||
};
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: state.reasoningIndex,
|
||||
item: {
|
||||
id: state.reasoningId,
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
|
||||
}
|
||||
item
|
||||
});
|
||||
|
||||
recordCompletedOutputItem(state, state.reasoningIndex, item);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -291,16 +299,20 @@ function closeMessage(state, emit, idx) {
|
||||
part: { type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }
|
||||
});
|
||||
|
||||
const item = {
|
||||
id: msgId,
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
|
||||
role: ROLE.ASSISTANT
|
||||
};
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: parseInt(idx),
|
||||
item: {
|
||||
id: msgId,
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
|
||||
role: ROLE.ASSISTANT
|
||||
}
|
||||
item
|
||||
});
|
||||
|
||||
recordCompletedOutputItem(state, parseInt(idx), item);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -394,23 +406,51 @@ function closeToolCall(state, emit, idx) {
|
||||
});
|
||||
}
|
||||
|
||||
const item = {
|
||||
id: `${custom ? "ctc" : "fc"}_${callId}`,
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
|
||||
call_id: callId,
|
||||
name: state.funcNames[idx] || ""
|
||||
};
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: parseInt(idx),
|
||||
item: {
|
||||
id: `${custom ? "ctc" : "fc"}_${callId}`,
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
|
||||
call_id: callId,
|
||||
name: state.funcNames[idx] || ""
|
||||
}
|
||||
item
|
||||
});
|
||||
|
||||
recordCompletedOutputItem(state, parseInt(idx), item);
|
||||
|
||||
state.funcItemDone[idx] = true;
|
||||
state.funcArgsDone[idx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
// response.completed carries the finished Response object, so response.output has
|
||||
// to repeat the items already delivered in response.output_item.done. Clients that
|
||||
// build their final result from the terminal event (GitHub Copilot CLI, the OpenAI
|
||||
// SDK "final response" helpers) otherwise treat the turn as empty even though the
|
||||
// text was streamed - see issue #4307.
|
||||
//
|
||||
// Keyed by output_index so a repeated close overwrites rather than duplicating the
|
||||
// item, and ordered by output_index so response.output matches the order the items
|
||||
// were emitted in. Lazily created because stream.js can hand us a state it built
|
||||
// itself rather than one from initState().
|
||||
function recordCompletedOutputItem(state, outputIndex, item) {
|
||||
state.completedOutputItems ??= new Map();
|
||||
const index = Number.isInteger(outputIndex) ? outputIndex : Number.parseInt(outputIndex, 10) || 0;
|
||||
state.completedOutputItems.set(index, item);
|
||||
}
|
||||
|
||||
function collectCompletedOutputItems(state) {
|
||||
const recorded = state.completedOutputItems;
|
||||
if (!(recorded instanceof Map) || recorded.size === 0) return [];
|
||||
return [...recorded.entries()]
|
||||
.sort((left, right) => left[0] - right[0])
|
||||
.map(([, item]) => item);
|
||||
}
|
||||
|
||||
function sendCompleted(state, emit) {
|
||||
if (!state.completedSent) {
|
||||
state.completedSent = true;
|
||||
@@ -423,6 +463,7 @@ function sendCompleted(state, emit) {
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null,
|
||||
output: collectCompletedOutputItems(state),
|
||||
...(state.responsesUsage ? { usage: state.responsesUsage } : {})
|
||||
}
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user