Merge remote-tracking branch 'origin/master' into gitea/feature/end
Resolved conflicts taking origin/master (v0.5.55) as canonical, with local features re-applied: - runtime log level (LOG_LEVEL env + dashboard Settings → Logging, applied immediately and persisted across restarts) - free/noAuth provider enable/disable toggle via providerStrategies.enabled - parallel model testing (Test All Models / Test Selected Keys)
This commit is contained in:
@@ -9,6 +9,9 @@ import { PROVIDERS } from "../../providers/index.js";
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
|
||||
|
||||
const CACHE_CONTROL_5M = { type: "ephemeral" };
|
||||
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
|
||||
|
||||
// Check if message has valid non-empty content
|
||||
export function hasValidContent(msg) {
|
||||
if (typeof msg.content === "string" && msg.content.trim()) return true;
|
||||
@@ -16,7 +19,9 @@ export function hasValidContent(msg) {
|
||||
return msg.content.some(block =>
|
||||
(block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
|
||||
block.type === CLAUDE_BLOCK.TOOL_USE ||
|
||||
block.type === CLAUDE_BLOCK.TOOL_RESULT
|
||||
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
|
||||
block.type === CLAUDE_BLOCK.IMAGE ||
|
||||
block.type === CLAUDE_BLOCK.DOCUMENT
|
||||
);
|
||||
}
|
||||
return false;
|
||||
@@ -122,31 +127,128 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
if (Object.keys(body.output_config).length === 0) delete body.output_config;
|
||||
}
|
||||
|
||||
// 2. Hoist mid-conversation system messages into the top-level system field
|
||||
// 2. Fold mid-conversation system messages into the neighbouring turn.
|
||||
// Hoisting them into body.system would insert volatile content (token counters,
|
||||
// reminders) ahead of the whole conversation and invalidate the prefix cache on
|
||||
// every request. Folding in place keeps the cached prefix stable.
|
||||
if (Array.isArray(body.messages)) {
|
||||
const systemBlocks = [];
|
||||
const messages = [];
|
||||
for (const msg of body.messages) {
|
||||
if (msg.role === ROLE.SYSTEM) {
|
||||
const text = typeof msg.content === "string"
|
||||
? msg.content
|
||||
: Array.isArray(msg.content)
|
||||
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
|
||||
: "";
|
||||
if (text.trim()) systemBlocks.push({ type: CLAUDE_BLOCK.TEXT, text });
|
||||
if (msg.role !== ROLE.SYSTEM) {
|
||||
messages.push(msg);
|
||||
continue;
|
||||
}
|
||||
messages.push(msg);
|
||||
const text = typeof msg.content === "string"
|
||||
? msg.content
|
||||
: Array.isArray(msg.content)
|
||||
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
|
||||
: "";
|
||||
if (!text.trim()) continue;
|
||||
|
||||
// Copy-on-write: the caller's body is reused across account-fallback
|
||||
// attempts, so folding must never mutate the original message.
|
||||
const block = { type: CLAUDE_BLOCK.TEXT, text };
|
||||
const prev = messages[messages.length - 1];
|
||||
if (prev?.role === ROLE.USER) {
|
||||
const content = typeof prev.content === "string"
|
||||
? [{ type: CLAUDE_BLOCK.TEXT, text: prev.content }]
|
||||
: Array.isArray(prev.content) ? [...prev.content] : [];
|
||||
messages[messages.length - 1] = { ...prev, content: [...content, block] };
|
||||
continue;
|
||||
}
|
||||
messages.push({ role: ROLE.USER, content: [block] });
|
||||
}
|
||||
body.messages = messages;
|
||||
}
|
||||
|
||||
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
|
||||
// so foreign signatures leak into history and Anthropic rejects them).
|
||||
const thinkingEnabled = body.thinking?.type === "enabled";
|
||||
if (Array.isArray(body.messages)) {
|
||||
for (const msg of body.messages) {
|
||||
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
|
||||
let hasToolUse = false;
|
||||
let hasKeptThinking = false;
|
||||
const kept = [];
|
||||
for (const block of msg.content) {
|
||||
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
|
||||
if (isValidClaudeSignature(block.signature)) {
|
||||
hasKeptThinking = true;
|
||||
kept.push(block);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
|
||||
kept.push(block);
|
||||
}
|
||||
msg.content = kept;
|
||||
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
|
||||
msg.content.unshift(buildThinkingPlaceholder("claude"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
// Put a 5m breakpoint on the last cache-eligible block of a message.
|
||||
// thinking/redacted_thinking blocks do not accept cache_control.
|
||||
function markLastCacheableBlock(msg) {
|
||||
if (!Array.isArray(msg?.content)) return false;
|
||||
for (let i = msg.content.length - 1; i >= 0; i--) {
|
||||
const block = msg.content[i];
|
||||
if (typeof block !== "object" || block === null) continue;
|
||||
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
|
||||
block.cache_control = { ...CACHE_CONTROL_5M };
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Re-anchor cache breakpoints on a Claude passthrough body (same policy as
|
||||
// prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m.
|
||||
// The client's own markers point at pre-normalization offsets, so they are dropped.
|
||||
// Must run LAST, after every step that can reshape system/tools/messages
|
||||
// (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail.
|
||||
export function anchorClaudeCache(body) {
|
||||
if (!body || typeof body !== "object") return body;
|
||||
|
||||
if (Array.isArray(body.system)) {
|
||||
const last = body.system.length - 1;
|
||||
body.system.forEach((block, i) => {
|
||||
if (typeof block !== "object" || block === null) return;
|
||||
if (i === last) block.cache_control = { ...CACHE_CONTROL_1H };
|
||||
else delete block.cache_control;
|
||||
});
|
||||
}
|
||||
|
||||
if (Array.isArray(body.tools)) {
|
||||
const last = body.tools.length - 1;
|
||||
body.tools.forEach((tool, i) => {
|
||||
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
|
||||
else delete tool.cache_control;
|
||||
});
|
||||
}
|
||||
|
||||
if (Array.isArray(body.messages)) {
|
||||
let anchored = null;
|
||||
for (let i = body.messages.length - 1; i >= 0; i--) {
|
||||
const msg = body.messages[i];
|
||||
if (!Array.isArray(msg.content)) continue;
|
||||
for (const block of msg.content) delete block.cache_control;
|
||||
|
||||
// Prefer the last assistant turn: it ends a completed exchange, so the
|
||||
// prefix up to it stays byte-stable across the following requests.
|
||||
if (anchored || msg.role !== ROLE.ASSISTANT) continue;
|
||||
anchored = markLastCacheableBlock(msg);
|
||||
}
|
||||
|
||||
if (systemBlocks.length > 0) {
|
||||
const existing = Array.isArray(body.system)
|
||||
? body.system
|
||||
: typeof body.system === "string" && body.system.trim()
|
||||
? [{ type: "text", text: body.system }]
|
||||
: [];
|
||||
body.system = [...existing, ...systemBlocks];
|
||||
body.messages = messages;
|
||||
// First turn of a conversation has no assistant yet — anchor the final
|
||||
// message instead, so the opening prompt is cached rather than paid twice.
|
||||
if (!anchored) {
|
||||
for (let i = body.messages.length - 1; i >= 0 && !anchored; i--) {
|
||||
anchored = markLastCacheableBlock(body.messages[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -192,10 +294,27 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
|
||||
delete body.output_config;
|
||||
}
|
||||
|
||||
// Clamp max_tokens to the model output ceiling (never above DEFAULT_MAX_TOKENS)
|
||||
// Clamp max_tokens to the model's real output ceiling. Models whose caps
|
||||
// declare a higher maxOutput (e.g. Opus 4.8 / Sonnet 4.6 = 128000) are allowed
|
||||
// up to it, so max-effort thinking gets full budget; others fall back to the
|
||||
// conservative 64000 default.
|
||||
if (body.max_tokens) {
|
||||
const ceiling = Math.min(getCapabilitiesForModel(provider, body.model).maxOutput, DEFAULT_MAX_TOKENS);
|
||||
const ceiling = getCapabilitiesForModel(provider, body.model).maxOutput || DEFAULT_MAX_TOKENS;
|
||||
if (body.max_tokens > ceiling) body.max_tokens = ceiling;
|
||||
|
||||
// Reconcile against thinking budget. applyThinking (thinkingUnified.js) runs
|
||||
// AFTER adjustMaxTokens capped max_tokens, and the claude-budget format maps
|
||||
// max effort → budget_tokens 128000 — larger than the clamped max_tokens.
|
||||
// Anthropic requires max_tokens strictly greater than budget_tokens (else 400).
|
||||
// Prefer raising max_tokens to preserve the requested thinking depth; if the
|
||||
// budget alone meets/exceeds the ceiling, cap output and shrink the budget so
|
||||
// some tokens remain for the answer.
|
||||
if (body.thinking?.type === "enabled" && body.thinking.budget_tokens && body.thinking.budget_tokens >= body.max_tokens) {
|
||||
body.max_tokens = Math.min(body.thinking.budget_tokens + 1024, ceiling);
|
||||
if (body.thinking.budget_tokens >= body.max_tokens) {
|
||||
body.thinking.budget_tokens = Math.max(1024, body.max_tokens - 1024);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 1. System: remove all cache_control, add only to last block with ttl 1h
|
||||
|
||||
@@ -7,7 +7,13 @@ import { OPENAI_BLOCK } from "../schema/index.js";
|
||||
export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
|
||||
// Basic constraints (not supported by Gemini API)
|
||||
"minLength", "maxLength", "exclusiveMinimum", "exclusiveMaximum",
|
||||
"minItems", "maxItems", "format",
|
||||
"minItems", "maxItems", "format", "multipleOf",
|
||||
// Array keywords the Gemini schema proto has no field for. Agent tool
|
||||
// schemas set these routinely, and one occurrence rejects the whole request
|
||||
// with "Unknown name ...: Cannot find field".
|
||||
"uniqueItems", "contains",
|
||||
// 2020-12 keywords with no Gemini equivalent
|
||||
"unevaluatedProperties", "unevaluatedItems", "contentSchema",
|
||||
// Claude rejects these in VALIDATED mode
|
||||
"default", "examples",
|
||||
// JSON Schema meta keywords
|
||||
@@ -353,6 +359,19 @@ export function cleanJSONSchemaForAntigravity(schema) {
|
||||
function addPlaceholders(obj) {
|
||||
if (!obj || typeof obj !== "object") return;
|
||||
|
||||
// Empty schema {} (no type, no properties) after $ref removal — treat as object with placeholder
|
||||
if (Object.keys(obj).length === 0) {
|
||||
obj.type = "object";
|
||||
obj.properties = {
|
||||
reason: {
|
||||
type: "string",
|
||||
description: "Brief explanation of why you are calling this tool"
|
||||
}
|
||||
};
|
||||
obj.required = ["reason"];
|
||||
return;
|
||||
}
|
||||
|
||||
if (obj.type === "object") {
|
||||
if (!obj.properties || Object.keys(obj.properties).length === 0) {
|
||||
obj.properties = {
|
||||
|
||||
@@ -3,9 +3,13 @@ import { DEFAULT_MAX_TOKENS, DEFAULT_MIN_TOKENS } from "../../config/runtimeConf
|
||||
/**
|
||||
* Adjust max_tokens based on request context
|
||||
* @param {object} body - Request body
|
||||
* @param {number} [ceiling=DEFAULT_MAX_TOKENS] - Upper bound for max_tokens.
|
||||
* Callers with model context (e.g. openai-to-claude) pass the model's real
|
||||
* maxOutput so high-output models (Opus 4.8 = 128000) aren't pre-clamped to
|
||||
* the conservative 64000 default before the model-aware step sees them.
|
||||
* @returns {number} Adjusted max_tokens
|
||||
*/
|
||||
export function adjustMaxTokens(body) {
|
||||
export function adjustMaxTokens(body, ceiling = DEFAULT_MAX_TOKENS) {
|
||||
let maxTokens = body.max_tokens || DEFAULT_MAX_TOKENS;
|
||||
|
||||
// Auto-increase for tool calling to prevent truncated arguments (min never above max)
|
||||
@@ -16,14 +20,14 @@ export function adjustMaxTokens(body) {
|
||||
}
|
||||
|
||||
// Ensure max_tokens > thinking.budget_tokens (Claude API requirement)
|
||||
// Claude API requires strictly greater, so add buffer instead of using DEFAULT_MAX_TOKENS
|
||||
// which could equal budget_tokens when budget_tokens >= 64000
|
||||
// Claude API requires strictly greater, so add buffer instead of using the
|
||||
// ceiling which could equal budget_tokens when budget_tokens >= ceiling
|
||||
if (body.thinking?.budget_tokens && maxTokens <= body.thinking.budget_tokens) {
|
||||
maxTokens = body.thinking.budget_tokens + 1024;
|
||||
}
|
||||
|
||||
// Never exceed the global ceiling
|
||||
if (maxTokens > DEFAULT_MAX_TOKENS) maxTokens = DEFAULT_MAX_TOKENS;
|
||||
// Never exceed the ceiling
|
||||
if (maxTokens > ceiling) maxTokens = ceiling;
|
||||
|
||||
return maxTokens;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user