Merge remote-tracking branch 'upstream/master'
# Conflicts: # .gitignore # open-sse/handlers/chatCore.js
This commit is contained in:
@@ -1,3 +1,5 @@
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
|
||||
// Strip request params a given provider/model rejects upstream (e.g. HTTP 400).
|
||||
// Config-driven: add a rule instead of scattering `delete body.x` across executors.
|
||||
|
||||
@@ -12,6 +14,13 @@ const STRIP_RULES = [
|
||||
{ provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] },
|
||||
// Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926)
|
||||
{ provider: "cloudflare-ai", flattenContent: true },
|
||||
{ provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true },
|
||||
// VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's
|
||||
// advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144),
|
||||
// so clampToModelMaxOutput alone leaves it uncapped and the request 400s with
|
||||
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
|
||||
// min() with the model ceiling still applies if a variant's own limit is lower.
|
||||
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
|
||||
];
|
||||
|
||||
// Test a rule's match (regex or predicate) against the model id.
|
||||
@@ -20,6 +29,12 @@ function matches(rule, model) {
|
||||
return typeof rule.match === "function" ? rule.match(model) : rule.match.test(model);
|
||||
}
|
||||
|
||||
function clampNumber(body, key, ceiling) {
|
||||
if (typeof body[key] === "number" && Number.isFinite(body[key]) && body[key] > ceiling) {
|
||||
body[key] = ceiling;
|
||||
}
|
||||
}
|
||||
|
||||
// Remove unsupported params from body in place; returns body.
|
||||
export function stripUnsupportedParams(provider, model, body) {
|
||||
if (!model || !body || typeof body !== "object") return body;
|
||||
@@ -39,6 +54,22 @@ export function stripUnsupportedParams(provider, model, body) {
|
||||
}
|
||||
}
|
||||
}
|
||||
if (rule.clampToModelMaxOutput || Number.isFinite(rule.maxOutputCap)) {
|
||||
const modelCeiling = getCapabilitiesForModel(provider, model).maxOutput;
|
||||
const candidates = [];
|
||||
if (rule.clampToModelMaxOutput && Number.isFinite(modelCeiling) && modelCeiling > 0) {
|
||||
candidates.push(modelCeiling);
|
||||
}
|
||||
if (Number.isFinite(rule.maxOutputCap) && rule.maxOutputCap > 0) {
|
||||
candidates.push(rule.maxOutputCap);
|
||||
}
|
||||
if (candidates.length > 0) {
|
||||
const ceiling = Math.min(...candidates);
|
||||
clampNumber(body, "max_tokens", ceiling);
|
||||
clampNumber(body, "max_completion_tokens", ceiling);
|
||||
clampNumber(body, "max_output_tokens", ceiling);
|
||||
}
|
||||
}
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -20,6 +20,13 @@ const FORMAT_TO_NATIVE = {
|
||||
kiro: "kiro",
|
||||
};
|
||||
|
||||
// Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent).
|
||||
export function stripThinkingSuffix(model) {
|
||||
if (typeof model !== "string") return model;
|
||||
const m = model.match(/^(.*)\([^()]+\)\s*$/);
|
||||
return m ? m[1].trim() : model;
|
||||
}
|
||||
|
||||
// Parse model-name suffix "model(value)" → { cleanModel, override }.
|
||||
// value: level name (high) | number (8192) | auto | none. null override when absent.
|
||||
export function parseSuffix(model) {
|
||||
@@ -132,18 +139,66 @@ function toGeminiThinkingLevel(cfg) {
|
||||
return effortToThinkingLevel(raw);
|
||||
}
|
||||
|
||||
function toKimiReasoningEffort(cfg) {
|
||||
const level = toLevel(cfg);
|
||||
if (level === "auto") return "high";
|
||||
if (level === "minimal") return "low";
|
||||
if (level === "xhigh") return "max";
|
||||
if (["low", "medium", "high", "max"].includes(level)) return level;
|
||||
return null;
|
||||
}
|
||||
|
||||
const GEMINI_LEVEL_OUTPUT_FLOOR = {
|
||||
minimal: 4096,
|
||||
low: 8192,
|
||||
medium: 16384,
|
||||
high: 65535,
|
||||
};
|
||||
|
||||
function geminiBudgetOutputFloor(budget) {
|
||||
if (budget === -1) return 32768;
|
||||
if (!Number.isFinite(budget)) return 32768;
|
||||
if (budget <= 1024) return 8192;
|
||||
if (budget <= 8192) return 16384;
|
||||
if (budget <= 24576) return 32768;
|
||||
return 65535;
|
||||
}
|
||||
|
||||
function geminiLevelOutputFloor(level) {
|
||||
return GEMINI_LEVEL_OUTPUT_FLOOR[level] || GEMINI_LEVEL_OUTPUT_FLOOR.high;
|
||||
}
|
||||
|
||||
// Gemini nests thinkingConfig under generationConfig. gemini-cli / antigravity wrap
|
||||
// the whole request in a { request: { generationConfig } } envelope — target the
|
||||
// envelope's generationConfig when present, else the top-level one.
|
||||
function getGeminiGenerationConfig(body) {
|
||||
if (body.request && typeof body.request === "object") {
|
||||
if (!body.request.generationConfig || typeof body.request.generationConfig !== "object") {
|
||||
body.request.generationConfig = {};
|
||||
}
|
||||
return body.request.generationConfig;
|
||||
}
|
||||
if (!body.generationConfig || typeof body.generationConfig !== "object") {
|
||||
body.generationConfig = {};
|
||||
}
|
||||
return body.generationConfig;
|
||||
}
|
||||
|
||||
function setGeminiThinking(body, tc) {
|
||||
const gc = body.request?.generationConfig
|
||||
? body.request.generationConfig
|
||||
: (body.generationConfig && typeof body.generationConfig === "object"
|
||||
? body.generationConfig
|
||||
: (body.generationConfig = {}));
|
||||
const gc = getGeminiGenerationConfig(body);
|
||||
gc.thinkingConfig = tc;
|
||||
}
|
||||
|
||||
function ensureGeminiOutputFloor(body, floor, caps) {
|
||||
const cap = Number.isFinite(caps?.maxOutput) ? caps.maxOutput : floor;
|
||||
const target = Math.min(floor, cap);
|
||||
const gc = getGeminiGenerationConfig(body);
|
||||
const current = Number(gc.maxOutputTokens);
|
||||
if (!Number.isFinite(current) || current < target) {
|
||||
gc.maxOutputTokens = target;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip every known thinking field from a body (used before re-applying / when unsupported).
|
||||
function stripAll(body) {
|
||||
delete body.thinking;
|
||||
@@ -168,7 +223,8 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "openai": {
|
||||
if (none && canDisable) { body.reasoning_effort = "none"; break; }
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level;
|
||||
// OpenAI reasoning_effort enum caps at "xhigh" (no "max"); clamp Claude Code's "max".
|
||||
if (level) body.reasoning_effort = level === "max" ? "xhigh" : level;
|
||||
break;
|
||||
}
|
||||
case "claude-adaptive": {
|
||||
@@ -192,12 +248,14 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "gemini-level": {
|
||||
const level = none ? "minimal" : toGeminiThinkingLevel(eff);
|
||||
setGeminiThinking(body, { thinkingLevel: level, includeThoughts: level !== "minimal" });
|
||||
ensureGeminiOutputFloor(body, geminiLevelOutputFloor(level), caps);
|
||||
break;
|
||||
}
|
||||
case "gemini-budget": {
|
||||
if (none && canDisable) { setGeminiThinking(body, { thinkingBudget: 0, includeThoughts: false }); break; }
|
||||
const budget = toBudget(eff, caps.thinkingRange);
|
||||
setGeminiThinking(body, { thinkingBudget: budget ?? -1, includeThoughts: true });
|
||||
ensureGeminiOutputFloor(body, geminiBudgetOutputFloor(budget ?? -1), caps);
|
||||
break;
|
||||
}
|
||||
case "zai": {
|
||||
@@ -223,8 +281,8 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
}
|
||||
case "kimi": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level === "max" ? "high" : level;
|
||||
const effort = toKimiReasoningEffort(eff);
|
||||
if (effort) body.reasoning_effort = effort;
|
||||
break;
|
||||
}
|
||||
case "minimax": {
|
||||
|
||||
Reference in New Issue
Block a user