Merge remote-tracking branch 'origin/master' into gitea/feature/end

Resolved conflicts taking origin/master (v0.5.55) as canonical, with local
features re-applied:
- runtime log level (LOG_LEVEL env + dashboard Settings → Logging, applied
  immediately and persisted across restarts)
- free/noAuth provider enable/disable toggle via providerStrategies.enabled
- parallel model testing (Test All Models / Test Selected Keys)
This commit is contained in:
2026-08-17 00:38:31 +07:00
parent e7470e955e
commit 7e45ead2ac
557 changed files with 53396 additions and 6935 deletions

View File

@@ -9,6 +9,9 @@ import { PROVIDERS } from "../../providers/index.js";
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
const CACHE_CONTROL_5M = { type: "ephemeral" };
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
// Check if message has valid non-empty content
export function hasValidContent(msg) {
if (typeof msg.content === "string" && msg.content.trim()) return true;
@@ -16,7 +19,9 @@ export function hasValidContent(msg) {
return msg.content.some(block =>
(block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
block.type === CLAUDE_BLOCK.TOOL_USE ||
block.type === CLAUDE_BLOCK.TOOL_RESULT
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
block.type === CLAUDE_BLOCK.IMAGE ||
block.type === CLAUDE_BLOCK.DOCUMENT
);
}
return false;
@@ -122,31 +127,128 @@ export function normalizeClaudePassthrough(body, model = "") {
if (Object.keys(body.output_config).length === 0) delete body.output_config;
}
// 2. Hoist mid-conversation system messages into the top-level system field
// 2. Fold mid-conversation system messages into the neighbouring turn.
// Hoisting them into body.system would insert volatile content (token counters,
// reminders) ahead of the whole conversation and invalidate the prefix cache on
// every request. Folding in place keeps the cached prefix stable.
if (Array.isArray(body.messages)) {
const systemBlocks = [];
const messages = [];
for (const msg of body.messages) {
if (msg.role === ROLE.SYSTEM) {
const text = typeof msg.content === "string"
? msg.content
: Array.isArray(msg.content)
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
: "";
if (text.trim()) systemBlocks.push({ type: CLAUDE_BLOCK.TEXT, text });
if (msg.role !== ROLE.SYSTEM) {
messages.push(msg);
continue;
}
messages.push(msg);
const text = typeof msg.content === "string"
? msg.content
: Array.isArray(msg.content)
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
: "";
if (!text.trim()) continue;
// Copy-on-write: the caller's body is reused across account-fallback
// attempts, so folding must never mutate the original message.
const block = { type: CLAUDE_BLOCK.TEXT, text };
const prev = messages[messages.length - 1];
if (prev?.role === ROLE.USER) {
const content = typeof prev.content === "string"
? [{ type: CLAUDE_BLOCK.TEXT, text: prev.content }]
: Array.isArray(prev.content) ? [...prev.content] : [];
messages[messages.length - 1] = { ...prev, content: [...content, block] };
continue;
}
messages.push({ role: ROLE.USER, content: [block] });
}
body.messages = messages;
}
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// so foreign signatures leak into history and Anthropic rejects them).
const thinkingEnabled = body.thinking?.type === "enabled";
if (Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
let hasToolUse = false;
let hasKeptThinking = false;
const kept = [];
for (const block of msg.content) {
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
if (isValidClaudeSignature(block.signature)) {
hasKeptThinking = true;
kept.push(block);
}
continue;
}
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
kept.push(block);
}
msg.content = kept;
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
msg.content.unshift(buildThinkingPlaceholder("claude"));
}
}
}
return body;
}
// Put a 5m breakpoint on the last cache-eligible block of a message.
// thinking/redacted_thinking blocks do not accept cache_control.
function markLastCacheableBlock(msg) {
if (!Array.isArray(msg?.content)) return false;
for (let i = msg.content.length - 1; i >= 0; i--) {
const block = msg.content[i];
if (typeof block !== "object" || block === null) continue;
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
block.cache_control = { ...CACHE_CONTROL_5M };
return true;
}
return false;
}
// Re-anchor cache breakpoints on a Claude passthrough body (same policy as
// prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m.
// The client's own markers point at pre-normalization offsets, so they are dropped.
// Must run LAST, after every step that can reshape system/tools/messages
// (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail.
export function anchorClaudeCache(body) {
if (!body || typeof body !== "object") return body;
if (Array.isArray(body.system)) {
const last = body.system.length - 1;
body.system.forEach((block, i) => {
if (typeof block !== "object" || block === null) return;
if (i === last) block.cache_control = { ...CACHE_CONTROL_1H };
else delete block.cache_control;
});
}
if (Array.isArray(body.tools)) {
const last = body.tools.length - 1;
body.tools.forEach((tool, i) => {
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
else delete tool.cache_control;
});
}
if (Array.isArray(body.messages)) {
let anchored = null;
for (let i = body.messages.length - 1; i >= 0; i--) {
const msg = body.messages[i];
if (!Array.isArray(msg.content)) continue;
for (const block of msg.content) delete block.cache_control;
// Prefer the last assistant turn: it ends a completed exchange, so the
// prefix up to it stays byte-stable across the following requests.
if (anchored || msg.role !== ROLE.ASSISTANT) continue;
anchored = markLastCacheableBlock(msg);
}
if (systemBlocks.length > 0) {
const existing = Array.isArray(body.system)
? body.system
: typeof body.system === "string" && body.system.trim()
? [{ type: "text", text: body.system }]
: [];
body.system = [...existing, ...systemBlocks];
body.messages = messages;
// First turn of a conversation has no assistant yet — anchor the final
// message instead, so the opening prompt is cached rather than paid twice.
if (!anchored) {
for (let i = body.messages.length - 1; i >= 0 && !anchored; i--) {
anchored = markLastCacheableBlock(body.messages[i]);
}
}
}
@@ -192,10 +294,27 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
delete body.output_config;
}
// Clamp max_tokens to the model output ceiling (never above DEFAULT_MAX_TOKENS)
// Clamp max_tokens to the model's real output ceiling. Models whose caps
// declare a higher maxOutput (e.g. Opus 4.8 / Sonnet 4.6 = 128000) are allowed
// up to it, so max-effort thinking gets full budget; others fall back to the
// conservative 64000 default.
if (body.max_tokens) {
const ceiling = Math.min(getCapabilitiesForModel(provider, body.model).maxOutput, DEFAULT_MAX_TOKENS);
const ceiling = getCapabilitiesForModel(provider, body.model).maxOutput || DEFAULT_MAX_TOKENS;
if (body.max_tokens > ceiling) body.max_tokens = ceiling;
// Reconcile against thinking budget. applyThinking (thinkingUnified.js) runs
// AFTER adjustMaxTokens capped max_tokens, and the claude-budget format maps
// max effort → budget_tokens 128000 — larger than the clamped max_tokens.
// Anthropic requires max_tokens strictly greater than budget_tokens (else 400).
// Prefer raising max_tokens to preserve the requested thinking depth; if the
// budget alone meets/exceeds the ceiling, cap output and shrink the budget so
// some tokens remain for the answer.
if (body.thinking?.type === "enabled" && body.thinking.budget_tokens && body.thinking.budget_tokens >= body.max_tokens) {
body.max_tokens = Math.min(body.thinking.budget_tokens + 1024, ceiling);
if (body.thinking.budget_tokens >= body.max_tokens) {
body.thinking.budget_tokens = Math.max(1024, body.max_tokens - 1024);
}
}
}
// 1. System: remove all cache_control, add only to last block with ttl 1h

View File

@@ -7,7 +7,13 @@ import { OPENAI_BLOCK } from "../schema/index.js";
export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
// Basic constraints (not supported by Gemini API)
"minLength", "maxLength", "exclusiveMinimum", "exclusiveMaximum",
"minItems", "maxItems", "format",
"minItems", "maxItems", "format", "multipleOf",
// Array keywords the Gemini schema proto has no field for. Agent tool
// schemas set these routinely, and one occurrence rejects the whole request
// with "Unknown name ...: Cannot find field".
"uniqueItems", "contains",
// 2020-12 keywords with no Gemini equivalent
"unevaluatedProperties", "unevaluatedItems", "contentSchema",
// Claude rejects these in VALIDATED mode
"default", "examples",
// JSON Schema meta keywords
@@ -353,6 +359,19 @@ export function cleanJSONSchemaForAntigravity(schema) {
function addPlaceholders(obj) {
if (!obj || typeof obj !== "object") return;
// Empty schema {} (no type, no properties) after $ref removal — treat as object with placeholder
if (Object.keys(obj).length === 0) {
obj.type = "object";
obj.properties = {
reason: {
type: "string",
description: "Brief explanation of why you are calling this tool"
}
};
obj.required = ["reason"];
return;
}
if (obj.type === "object") {
if (!obj.properties || Object.keys(obj.properties).length === 0) {
obj.properties = {

View File

@@ -3,9 +3,13 @@ import { DEFAULT_MAX_TOKENS, DEFAULT_MIN_TOKENS } from "../../config/runtimeConf
/**
* Adjust max_tokens based on request context
* @param {object} body - Request body
* @param {number} [ceiling=DEFAULT_MAX_TOKENS] - Upper bound for max_tokens.
* Callers with model context (e.g. openai-to-claude) pass the model's real
* maxOutput so high-output models (Opus 4.8 = 128000) aren't pre-clamped to
* the conservative 64000 default before the model-aware step sees them.
* @returns {number} Adjusted max_tokens
*/
export function adjustMaxTokens(body) {
export function adjustMaxTokens(body, ceiling = DEFAULT_MAX_TOKENS) {
let maxTokens = body.max_tokens || DEFAULT_MAX_TOKENS;
// Auto-increase for tool calling to prevent truncated arguments (min never above max)
@@ -16,14 +20,14 @@ export function adjustMaxTokens(body) {
}
// Ensure max_tokens > thinking.budget_tokens (Claude API requirement)
// Claude API requires strictly greater, so add buffer instead of using DEFAULT_MAX_TOKENS
// which could equal budget_tokens when budget_tokens >= 64000
// Claude API requires strictly greater, so add buffer instead of using the
// ceiling which could equal budget_tokens when budget_tokens >= ceiling
if (body.thinking?.budget_tokens && maxTokens <= body.thinking.budget_tokens) {
maxTokens = body.thinking.budget_tokens + 1024;
}
// Never exceed the global ceiling
if (maxTokens > DEFAULT_MAX_TOKENS) maxTokens = DEFAULT_MAX_TOKENS;
// Never exceed the ceiling
if (maxTokens > ceiling) maxTokens = ceiling;
return maxTokens;
}