Merge remote-tracking branch 'origin/master' into gitea/feature/end
Resolved conflicts taking origin/master (v0.5.55) as canonical, with local features re-applied: - runtime log level (LOG_LEVEL env + dashboard Settings → Logging, applied immediately and persisted across restarts) - free/noAuth provider enable/disable toggle via providerStrategies.enabled - parallel model testing (Test All Models / Test Selected Keys)
This commit is contained in:
435
open-sse/translator/concerns/kiroConversation.js
Normal file
435
open-sse/translator/concerns/kiroConversation.js
Normal file
@@ -0,0 +1,435 @@
|
||||
import {
|
||||
KIRO_TOOL_DESCRIPTION_MAX_LENGTH,
|
||||
KIRO_TOOL_ID_MAX_LENGTH,
|
||||
KIRO_TOOL_NAME_MAX_LENGTH,
|
||||
} from "../../config/kiroConstants.js";
|
||||
|
||||
const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/;
|
||||
const TOOL_NAME_PATTERN = /[^a-zA-Z0-9_-]/g;
|
||||
|
||||
function clone(value) {
|
||||
return value == null ? value : JSON.parse(JSON.stringify(value));
|
||||
}
|
||||
|
||||
function text(value) {
|
||||
if (typeof value === "string") return value;
|
||||
if (value == null) return "";
|
||||
try {
|
||||
return JSON.stringify(value);
|
||||
} catch {
|
||||
return String(value);
|
||||
}
|
||||
}
|
||||
|
||||
function appendText(target, extra) {
|
||||
if (!extra) return;
|
||||
target.content = target.content ? `${target.content}\n\n${extra}` : extra;
|
||||
}
|
||||
|
||||
function trimCodePoints(value, limit) {
|
||||
return [...String(value || "")].slice(0, limit).join("");
|
||||
}
|
||||
|
||||
function uniqueName(rawName, index, usedNames) {
|
||||
const cleaned = String(rawName || "")
|
||||
.trim()
|
||||
.replace(TOOL_NAME_PATTERN, "_")
|
||||
.replace(/_+/g, "_")
|
||||
.replace(/^_+|_+$/g, "");
|
||||
const base = trimCodePoints(cleaned || `tool_${index + 1}`, KIRO_TOOL_NAME_MAX_LENGTH);
|
||||
let candidate = base;
|
||||
let suffix = 2;
|
||||
while (usedNames.has(candidate)) {
|
||||
const tail = `_${suffix++}`;
|
||||
candidate = `${base.slice(0, KIRO_TOOL_NAME_MAX_LENGTH - tail.length)}${tail}`;
|
||||
}
|
||||
usedNames.add(candidate);
|
||||
return candidate;
|
||||
}
|
||||
|
||||
function cleanSchemaValue(value) {
|
||||
if (Array.isArray(value)) return value.map(cleanSchemaValue);
|
||||
if (!value || typeof value !== "object") return value;
|
||||
|
||||
const cleaned = {};
|
||||
for (const [key, child] of Object.entries(value)) {
|
||||
if (key === "additionalProperties") continue;
|
||||
if (key === "required" && Array.isArray(child) && child.length === 0) continue;
|
||||
cleaned[key] = cleanSchemaValue(child);
|
||||
}
|
||||
return cleaned;
|
||||
}
|
||||
|
||||
function normalizeRootSchema(schema) {
|
||||
const cleaned = cleanSchemaValue(schema && typeof schema === "object" ? clone(schema) : {});
|
||||
cleaned.type = "object";
|
||||
if (!cleaned.properties || typeof cleaned.properties !== "object" || Array.isArray(cleaned.properties)) {
|
||||
cleaned.properties = {};
|
||||
}
|
||||
if (Array.isArray(cleaned.required)) {
|
||||
cleaned.required = [...new Set(cleaned.required.filter(
|
||||
(name) => typeof name === "string" && Object.hasOwn(cleaned.properties, name)
|
||||
))];
|
||||
if (cleaned.required.length === 0) delete cleaned.required;
|
||||
}
|
||||
return cleaned;
|
||||
}
|
||||
|
||||
/** Normalize OpenAI- or Claude-shaped tool definitions into Kiro tool specs. */
|
||||
export function normalizeKiroToolSpecs(tools) {
|
||||
const specs = [];
|
||||
const nameMap = new Map();
|
||||
const usedNames = new Set();
|
||||
|
||||
for (const [index, tool] of (Array.isArray(tools) ? tools : []).entries()) {
|
||||
if (!tool || typeof tool !== "object") continue;
|
||||
const rawName = tool.function?.name ?? tool.name;
|
||||
if (typeof rawName !== "string" || !rawName.trim()) continue;
|
||||
|
||||
// A repeated definition with the same source name describes the same tool.
|
||||
if (nameMap.has(rawName)) continue;
|
||||
const name = uniqueName(rawName, index, usedNames);
|
||||
nameMap.set(rawName, name);
|
||||
|
||||
const rawDescription = tool.function?.description ?? tool.description ?? `Tool: ${rawName}`;
|
||||
const description = trimCodePoints(
|
||||
String(rawDescription || `Tool: ${rawName}`),
|
||||
KIRO_TOOL_DESCRIPTION_MAX_LENGTH
|
||||
);
|
||||
const schema = tool.function?.parameters ?? tool.parameters ?? tool.input_schema ?? {};
|
||||
specs.push({
|
||||
toolSpecification: {
|
||||
name,
|
||||
description,
|
||||
inputSchema: { json: normalizeRootSchema(schema) },
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
return { specs, nameMap };
|
||||
}
|
||||
|
||||
function toolCallText(toolUse) {
|
||||
return `[Tool call: ${toolUse?.name || "unknown"}(${text(toolUse?.input || {})})]`;
|
||||
}
|
||||
|
||||
function toolResultText(toolResult) {
|
||||
const content = Array.isArray(toolResult?.content)
|
||||
? toolResult.content.map((part) => text(part?.text ?? part)).filter(Boolean).join("\n")
|
||||
: text(toolResult?.content);
|
||||
return `[Tool result${toolResult?.status === "error" ? " (error)" : ""}: ${content}]`;
|
||||
}
|
||||
|
||||
function mergeUser(target, source) {
|
||||
appendText(target, source.content);
|
||||
if (Array.isArray(source.images) && source.images.length > 0) {
|
||||
target.images = [...(target.images || []), ...source.images];
|
||||
}
|
||||
const results = source.userInputMessageContext?.toolResults;
|
||||
if (Array.isArray(results) && results.length > 0) {
|
||||
target.userInputMessageContext ||= {};
|
||||
target.userInputMessageContext.toolResults = [
|
||||
...(target.userInputMessageContext.toolResults || []),
|
||||
...results,
|
||||
];
|
||||
}
|
||||
}
|
||||
|
||||
function mergeAssistant(target, source) {
|
||||
appendText(target, source.content);
|
||||
if (Array.isArray(source.toolUses) && source.toolUses.length > 0) {
|
||||
target.toolUses = [...(target.toolUses || []), ...source.toolUses];
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeTurns(history, currentMessage, modelId) {
|
||||
const rawTurns = [...(Array.isArray(history) ? history : [])];
|
||||
if (currentMessage) rawTurns.push(currentMessage);
|
||||
const turns = [];
|
||||
|
||||
for (const raw of rawTurns) {
|
||||
const isUser = !!raw?.userInputMessage;
|
||||
const isAssistant = !!raw?.assistantResponseMessage;
|
||||
if (isUser === isAssistant) continue;
|
||||
|
||||
const turn = isUser
|
||||
? { userInputMessage: clone(raw.userInputMessage) }
|
||||
: { assistantResponseMessage: clone(raw.assistantResponseMessage) };
|
||||
const previous = turns[turns.length - 1];
|
||||
if (turn.userInputMessage && previous?.userInputMessage) {
|
||||
mergeUser(previous.userInputMessage, turn.userInputMessage);
|
||||
} else if (turn.assistantResponseMessage && previous?.assistantResponseMessage) {
|
||||
mergeAssistant(previous.assistantResponseMessage, turn.assistantResponseMessage);
|
||||
} else {
|
||||
turns.push(turn);
|
||||
}
|
||||
}
|
||||
|
||||
if (turns[0]?.assistantResponseMessage) {
|
||||
turns.unshift({ userInputMessage: { content: "continue", modelId } });
|
||||
}
|
||||
if (turns.length === 0 || turns[turns.length - 1]?.assistantResponseMessage) {
|
||||
turns.push({ userInputMessage: { content: "continue", modelId } });
|
||||
}
|
||||
|
||||
for (const turn of turns) {
|
||||
if (turn.userInputMessage) {
|
||||
turn.userInputMessage.content = text(turn.userInputMessage.content).trim() || "continue";
|
||||
turn.userInputMessage.modelId ||= modelId;
|
||||
if (turn.userInputMessage.userInputMessageContext?.tools) {
|
||||
delete turn.userInputMessage.userInputMessageContext.tools;
|
||||
}
|
||||
} else {
|
||||
turn.assistantResponseMessage.content =
|
||||
text(turn.assistantResponseMessage.content).trim() || "...";
|
||||
}
|
||||
}
|
||||
return turns;
|
||||
}
|
||||
|
||||
function rawId(value) {
|
||||
return typeof value === "string" ? value : "";
|
||||
}
|
||||
|
||||
function reserveToolId(value, turnIndex, callIndex, name, usedIds) {
|
||||
const sanitized = rawId(value).replace(/[^a-zA-Z0-9_-]/g, "");
|
||||
const generated = `call_msg${turnIndex}_tc${callIndex}_${name || "tool"}`;
|
||||
const base = trimCodePoints(
|
||||
TOOL_ID_PATTERN.test(sanitized) && sanitized ? sanitized : generated,
|
||||
KIRO_TOOL_ID_MAX_LENGTH
|
||||
);
|
||||
let candidate = base;
|
||||
let suffix = 2;
|
||||
while (usedIds.has(candidate)) {
|
||||
const tail = `_${suffix++}`;
|
||||
candidate = `${base.slice(0, KIRO_TOOL_ID_MAX_LENGTH - tail.length)}${tail}`;
|
||||
}
|
||||
usedIds.add(candidate);
|
||||
return candidate;
|
||||
}
|
||||
|
||||
function normalizeToolInput(input) {
|
||||
if (input && typeof input === "object" && !Array.isArray(input)) return clone(input);
|
||||
if (typeof input === "string") {
|
||||
try {
|
||||
const parsed = JSON.parse(input);
|
||||
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return parsed;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
return input == null ? {} : null;
|
||||
}
|
||||
|
||||
function normalizeToolResult(result) {
|
||||
const content = Array.isArray(result?.content)
|
||||
? result.content.map((part) => ({ text: text(part?.text ?? part) }))
|
||||
: [{ text: text(result?.content) }];
|
||||
return {
|
||||
toolUseId: rawId(result?.toolUseId),
|
||||
status: result?.status === "error" ? "error" : "success",
|
||||
content: content.length > 0 ? content : [{ text: "" }],
|
||||
};
|
||||
}
|
||||
|
||||
function flattenResults(userMessage, results) {
|
||||
for (const result of results) appendText(userMessage, toolResultText(result));
|
||||
}
|
||||
|
||||
function cleanUserContext(userMessage) {
|
||||
const context = userMessage.userInputMessageContext;
|
||||
if (!context) return;
|
||||
if (!context.toolResults?.length) delete context.toolResults;
|
||||
if (!context.tools?.length) delete context.tools;
|
||||
if (Object.keys(context).length === 0) delete userMessage.userInputMessageContext;
|
||||
}
|
||||
|
||||
function reconcileToolPair(assistant, user, turnIndex, nameMap, specNames, usedIds, repairs) {
|
||||
const calls = Array.isArray(assistant.toolUses) ? assistant.toolUses : [];
|
||||
const results = Array.isArray(user.userInputMessageContext?.toolResults)
|
||||
? user.userInputMessageContext.toolResults.map(normalizeToolResult)
|
||||
: [];
|
||||
if (calls.length === 0) {
|
||||
if (results.length > 0) {
|
||||
flattenResults(user, results);
|
||||
repairs.orphanResults += results.length;
|
||||
}
|
||||
if (user.userInputMessageContext) delete user.userInputMessageContext.toolResults;
|
||||
cleanUserContext(user);
|
||||
return;
|
||||
}
|
||||
|
||||
const callQueues = new Map();
|
||||
const callRecords = calls.map((call, callIndex) => {
|
||||
const key = rawId(call?.toolUseId);
|
||||
const mappedName = nameMap.get(call?.name) || call?.name;
|
||||
const input = normalizeToolInput(call?.input);
|
||||
const record = { call, callIndex, key, mappedName, input, result: null };
|
||||
const queue = callQueues.get(key) || [];
|
||||
queue.push(record);
|
||||
callQueues.set(key, queue);
|
||||
return record;
|
||||
});
|
||||
|
||||
const orphanResults = [];
|
||||
for (const result of results) {
|
||||
const queue = callQueues.get(rawId(result.toolUseId));
|
||||
const record = queue?.find((candidate) => !candidate.result);
|
||||
if (record) record.result = result;
|
||||
else orphanResults.push(result);
|
||||
}
|
||||
|
||||
const keptCalls = [];
|
||||
const keptResults = [];
|
||||
for (const record of callRecords) {
|
||||
const hasSpec = typeof record.mappedName === "string" && specNames.has(record.mappedName);
|
||||
const valid = !!record.result && hasSpec && record.input !== null;
|
||||
if (!valid) {
|
||||
appendText(assistant, toolCallText({ name: record.mappedName, input: record.call?.input }));
|
||||
repairs.missingResults += record.result ? 0 : 1;
|
||||
repairs.invalidToolUses += hasSpec && record.input !== null ? 0 : 1;
|
||||
if (record.result) {
|
||||
flattenResults(user, [record.result]);
|
||||
repairs.orphanResults++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
const toolUseId = reserveToolId(
|
||||
record.key,
|
||||
turnIndex,
|
||||
record.callIndex,
|
||||
record.mappedName,
|
||||
usedIds
|
||||
);
|
||||
keptCalls.push({
|
||||
toolUseId,
|
||||
name: record.mappedName,
|
||||
input: record.input,
|
||||
});
|
||||
keptResults.push({ ...record.result, toolUseId });
|
||||
}
|
||||
|
||||
if (orphanResults.length > 0) {
|
||||
flattenResults(user, orphanResults);
|
||||
repairs.orphanResults += orphanResults.length;
|
||||
}
|
||||
|
||||
if (keptCalls.length > 0) assistant.toolUses = keptCalls;
|
||||
else delete assistant.toolUses;
|
||||
user.userInputMessageContext ||= {};
|
||||
if (keptResults.length > 0) user.userInputMessageContext.toolResults = keptResults;
|
||||
else delete user.userInputMessageContext.toolResults;
|
||||
cleanUserContext(user);
|
||||
}
|
||||
|
||||
/** Validate the final Kiro wire conversation without mutating it. */
|
||||
export function validateKiroConversation(history, currentMessage, toolSpecs = []) {
|
||||
const errors = [];
|
||||
const turns = [...(history || []), currentMessage].filter(Boolean);
|
||||
const specNames = new Set(toolSpecs.map((spec) => spec?.toolSpecification?.name).filter(Boolean));
|
||||
const usedIds = new Set();
|
||||
|
||||
for (let index = 0; index < turns.length; index++) {
|
||||
const expectedUser = index % 2 === 0;
|
||||
const isUser = !!turns[index]?.userInputMessage;
|
||||
if (isUser !== expectedUser) errors.push(`role:${index}`);
|
||||
if (!isUser) {
|
||||
const calls = turns[index].assistantResponseMessage?.toolUses || [];
|
||||
const results = turns[index + 1]?.userInputMessage?.userInputMessageContext?.toolResults || [];
|
||||
const callIds = calls.map((call) => call.toolUseId);
|
||||
const resultIds = results.map((result) => result.toolUseId);
|
||||
if (calls.length !== results.length || callIds.some((id) => !resultIds.includes(id))) {
|
||||
errors.push(`pair:${index}`);
|
||||
}
|
||||
for (const call of calls) {
|
||||
if (!call.toolUseId || usedIds.has(call.toolUseId)) errors.push(`id:${index}`);
|
||||
usedIds.add(call.toolUseId);
|
||||
if (!specNames.has(call.name)) errors.push(`spec:${index}`);
|
||||
}
|
||||
} else if (index === 0) {
|
||||
const results = turns[index].userInputMessage?.userInputMessageContext?.toolResults;
|
||||
if (results?.length) errors.push("orphan:0");
|
||||
}
|
||||
}
|
||||
if (!currentMessage?.userInputMessage?.content) errors.push("current");
|
||||
return { valid: errors.length === 0, errors };
|
||||
}
|
||||
|
||||
function flattenAllStructuredTools(turns, repairs) {
|
||||
for (const turn of turns) {
|
||||
if (turn.assistantResponseMessage?.toolUses?.length) {
|
||||
for (const call of turn.assistantResponseMessage.toolUses) {
|
||||
appendText(turn.assistantResponseMessage, toolCallText(call));
|
||||
}
|
||||
repairs.invalidToolUses += turn.assistantResponseMessage.toolUses.length;
|
||||
delete turn.assistantResponseMessage.toolUses;
|
||||
}
|
||||
const user = turn.userInputMessage;
|
||||
const results = user?.userInputMessageContext?.toolResults;
|
||||
if (results?.length) {
|
||||
flattenResults(user, results);
|
||||
repairs.orphanResults += results.length;
|
||||
delete user.userInputMessageContext.toolResults;
|
||||
cleanUserContext(user);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Produce a strict Kiro conversation: alternating turns, current user message,
|
||||
* adjacent one-to-one tool use/result pairs, and tool specs only on currentMessage.
|
||||
*/
|
||||
export function canonicalizeKiroConversation({
|
||||
history,
|
||||
currentMessage,
|
||||
modelId,
|
||||
toolSpecs = [],
|
||||
nameMap = new Map(),
|
||||
} = {}) {
|
||||
const turns = normalizeTurns(history, currentMessage, modelId);
|
||||
const repairs = { missingResults: 0, orphanResults: 0, invalidToolUses: 0 };
|
||||
const specNames = new Set(toolSpecs.map((spec) => spec?.toolSpecification?.name).filter(Boolean));
|
||||
const usedIds = new Set();
|
||||
|
||||
for (let index = 0; index < turns.length; index += 2) {
|
||||
const user = turns[index].userInputMessage;
|
||||
if (index === 0) {
|
||||
const leadingResults = user.userInputMessageContext?.toolResults || [];
|
||||
if (leadingResults.length > 0) {
|
||||
flattenResults(user, leadingResults);
|
||||
repairs.orphanResults += leadingResults.length;
|
||||
delete user.userInputMessageContext.toolResults;
|
||||
cleanUserContext(user);
|
||||
}
|
||||
}
|
||||
const assistant = turns[index + 1]?.assistantResponseMessage;
|
||||
const nextUser = turns[index + 2]?.userInputMessage;
|
||||
if (assistant && nextUser) {
|
||||
reconcileToolPair(assistant, nextUser, index + 1, nameMap, specNames, usedIds, repairs);
|
||||
}
|
||||
}
|
||||
|
||||
const finalCurrent = turns[turns.length - 1];
|
||||
finalCurrent.userInputMessage.userInputMessageContext ||= {};
|
||||
if (toolSpecs.length > 0) {
|
||||
finalCurrent.userInputMessage.userInputMessageContext.tools = clone(toolSpecs);
|
||||
}
|
||||
cleanUserContext(finalCurrent.userInputMessage);
|
||||
|
||||
let finalHistory = turns.slice(0, -1);
|
||||
let validation = validateKiroConversation(finalHistory, finalCurrent, toolSpecs);
|
||||
if (!validation.valid) {
|
||||
flattenAllStructuredTools(turns, repairs);
|
||||
finalHistory = turns.slice(0, -1);
|
||||
validation = validateKiroConversation(finalHistory, finalCurrent, toolSpecs);
|
||||
}
|
||||
|
||||
return {
|
||||
history: finalHistory,
|
||||
currentMessage: finalCurrent,
|
||||
repairs,
|
||||
valid: validation.valid,
|
||||
errors: validation.errors,
|
||||
};
|
||||
}
|
||||
@@ -62,6 +62,19 @@ function stripOpenAI(body, caps) {
|
||||
if (!Array.isArray(body.messages)) return;
|
||||
const last = body.messages.length - 1;
|
||||
body.messages.forEach((msg, i) => {
|
||||
if (caps.vision === false) {
|
||||
if (Array.isArray(msg.images)) delete msg.images;
|
||||
if (Array.isArray(msg.experimental_attachments)) {
|
||||
msg.experimental_attachments = msg.experimental_attachments.filter(
|
||||
(a) => !(a?.contentType?.startsWith("image/") || (typeof a?.url === "string" && a.url.startsWith("data:image/")))
|
||||
);
|
||||
}
|
||||
if (Array.isArray(msg.attachments)) {
|
||||
msg.attachments = msg.attachments.filter(
|
||||
(a) => !(a?.contentType?.startsWith("image/") || (typeof a?.url === "string" && a.url.startsWith("data:image/")))
|
||||
);
|
||||
}
|
||||
}
|
||||
if (!Array.isArray(msg.content)) return;
|
||||
const removed = new Set();
|
||||
msg.content = filterBlocks(msg.content, capForOpenAIBlock, caps, removed, i === last);
|
||||
|
||||
@@ -1,17 +1,26 @@
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
|
||||
// Strip request params a given provider/model rejects upstream (e.g. HTTP 400).
|
||||
// Config-driven: add a rule instead of scattering `delete body.x` across executors.
|
||||
|
||||
// Each rule: optional provider, regex match on model, list of params to drop.
|
||||
// A param is removed only when it is present (!== undefined).
|
||||
const STRIP_RULES = [
|
||||
// claude-opus-4 series: temperature is deprecated (Anthropic 400). #1748
|
||||
{ match: /claude-opus-4/i, drop: ["temperature"] },
|
||||
// All Claude models: temperature deprecated/rejected upstream (Anthropic 400). #1748
|
||||
{ match: /claude/i, drop: ["temperature"] },
|
||||
// GitHub Copilot gpt-5.4: temperature unsupported.
|
||||
{ provider: "github", match: /gpt-5\.4/i, drop: ["temperature"] },
|
||||
// GitHub Copilot Claude (except opus/sonnet 4.6): thinking + reasoning_effort rejected. #713
|
||||
{ provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] },
|
||||
// Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926)
|
||||
{ provider: "cloudflare-ai", flattenContent: true },
|
||||
{ provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true },
|
||||
// VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's
|
||||
// advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144),
|
||||
// so clampToModelMaxOutput alone leaves it uncapped and the request 400s with
|
||||
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
|
||||
// min() with the model ceiling still applies if a variant's own limit is lower.
|
||||
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
|
||||
];
|
||||
|
||||
// Test a rule's match (regex or predicate) against the model id.
|
||||
@@ -20,6 +29,12 @@ function matches(rule, model) {
|
||||
return typeof rule.match === "function" ? rule.match(model) : rule.match.test(model);
|
||||
}
|
||||
|
||||
function clampNumber(body, key, ceiling) {
|
||||
if (typeof body[key] === "number" && Number.isFinite(body[key]) && body[key] > ceiling) {
|
||||
body[key] = ceiling;
|
||||
}
|
||||
}
|
||||
|
||||
// Remove unsupported params from body in place; returns body.
|
||||
export function stripUnsupportedParams(provider, model, body) {
|
||||
if (!model || !body || typeof body !== "object") return body;
|
||||
@@ -39,6 +54,22 @@ export function stripUnsupportedParams(provider, model, body) {
|
||||
}
|
||||
}
|
||||
}
|
||||
if (rule.clampToModelMaxOutput || Number.isFinite(rule.maxOutputCap)) {
|
||||
const modelCeiling = getCapabilitiesForModel(provider, model).maxOutput;
|
||||
const candidates = [];
|
||||
if (rule.clampToModelMaxOutput && Number.isFinite(modelCeiling) && modelCeiling > 0) {
|
||||
candidates.push(modelCeiling);
|
||||
}
|
||||
if (Number.isFinite(rule.maxOutputCap) && rule.maxOutputCap > 0) {
|
||||
candidates.push(rule.maxOutputCap);
|
||||
}
|
||||
if (candidates.length > 0) {
|
||||
const ceiling = Math.min(...candidates);
|
||||
clampNumber(body, "max_tokens", ceiling);
|
||||
clampNumber(body, "max_completion_tokens", ceiling);
|
||||
clampNumber(body, "max_output_tokens", ceiling);
|
||||
}
|
||||
}
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
// never hardcoded per-model here. See .docs/thinking/plan.md MATRIX VI-A.
|
||||
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
import { getThinkingLevels } from "../../providers/thinkingLevels.js";
|
||||
import { PROVIDERS } from "../../providers/index.js";
|
||||
import { LEVEL_TO_BUDGET, budgetToLevel, effortToBudget, effortToThinkingLevel } from "./thinking.js";
|
||||
|
||||
@@ -20,6 +21,13 @@ const FORMAT_TO_NATIVE = {
|
||||
kiro: "kiro",
|
||||
};
|
||||
|
||||
// Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent).
|
||||
export function stripThinkingSuffix(model) {
|
||||
if (typeof model !== "string") return model;
|
||||
const m = model.match(/^(.*)\([^()]+\)\s*$/);
|
||||
return m ? m[1].trim() : model;
|
||||
}
|
||||
|
||||
// Parse model-name suffix "model(value)" → { cleanModel, override }.
|
||||
// value: level name (high) | number (8192) | auto | none. null override when absent.
|
||||
export function parseSuffix(model) {
|
||||
@@ -30,6 +38,7 @@ export function parseSuffix(model) {
|
||||
const raw = m[2].trim().toLowerCase();
|
||||
if (raw === "none" || raw === "off") return { cleanModel, override: { mode: "none" } };
|
||||
if (raw === "auto") return { cleanModel, override: { mode: "auto" } };
|
||||
if (raw === "ultra") return { cleanModel, override: { mode: "level", level: raw } };
|
||||
if (/^\d+$/.test(raw)) return { cleanModel, override: { mode: "budget", budget: Number(raw) } };
|
||||
if (LEVEL_TO_BUDGET[raw] !== undefined) return { cleanModel, override: { mode: "level", level: raw } };
|
||||
return { cleanModel, override: null };
|
||||
@@ -127,23 +136,78 @@ function toLevel(cfg) {
|
||||
return null;
|
||||
}
|
||||
|
||||
function normalizeOpenAILevel(level, supportedLevels) {
|
||||
if (level !== "max" && level !== "ultra") return level;
|
||||
if (supportedLevels?.includes(level)) return level;
|
||||
if (level === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
return "xhigh";
|
||||
}
|
||||
|
||||
function toGeminiThinkingLevel(cfg) {
|
||||
const raw = cfg.mode === "auto" ? "high" : (toLevel(cfg) || "high");
|
||||
return effortToThinkingLevel(raw);
|
||||
}
|
||||
|
||||
function toKimiReasoningEffort(cfg) {
|
||||
const level = toLevel(cfg);
|
||||
if (level === "auto") return "high";
|
||||
if (level === "minimal") return "low";
|
||||
if (level === "xhigh") return "max";
|
||||
if (["low", "medium", "high", "max"].includes(level)) return level;
|
||||
return null;
|
||||
}
|
||||
|
||||
const GEMINI_LEVEL_OUTPUT_FLOOR = {
|
||||
minimal: 4096,
|
||||
low: 8192,
|
||||
medium: 16384,
|
||||
high: 65535,
|
||||
};
|
||||
|
||||
function geminiBudgetOutputFloor(budget) {
|
||||
if (budget === -1) return 32768;
|
||||
if (!Number.isFinite(budget)) return 32768;
|
||||
if (budget <= 1024) return 8192;
|
||||
if (budget <= 8192) return 16384;
|
||||
if (budget <= 24576) return 32768;
|
||||
return 65535;
|
||||
}
|
||||
|
||||
function geminiLevelOutputFloor(level) {
|
||||
return GEMINI_LEVEL_OUTPUT_FLOOR[level] || GEMINI_LEVEL_OUTPUT_FLOOR.high;
|
||||
}
|
||||
|
||||
// Gemini nests thinkingConfig under generationConfig. gemini-cli / antigravity wrap
|
||||
// the whole request in a { request: { generationConfig } } envelope — target the
|
||||
// envelope's generationConfig when present, else the top-level one.
|
||||
function getGeminiGenerationConfig(body) {
|
||||
if (body.request && typeof body.request === "object") {
|
||||
if (!body.request.generationConfig || typeof body.request.generationConfig !== "object") {
|
||||
body.request.generationConfig = {};
|
||||
}
|
||||
return body.request.generationConfig;
|
||||
}
|
||||
if (!body.generationConfig || typeof body.generationConfig !== "object") {
|
||||
body.generationConfig = {};
|
||||
}
|
||||
return body.generationConfig;
|
||||
}
|
||||
|
||||
function setGeminiThinking(body, tc) {
|
||||
const gc = body.request?.generationConfig
|
||||
? body.request.generationConfig
|
||||
: (body.generationConfig && typeof body.generationConfig === "object"
|
||||
? body.generationConfig
|
||||
: (body.generationConfig = {}));
|
||||
const gc = getGeminiGenerationConfig(body);
|
||||
gc.thinkingConfig = tc;
|
||||
}
|
||||
|
||||
function ensureGeminiOutputFloor(body, floor, caps) {
|
||||
const cap = Number.isFinite(caps?.maxOutput) ? caps.maxOutput : floor;
|
||||
const target = Math.min(floor, cap);
|
||||
const gc = getGeminiGenerationConfig(body);
|
||||
const current = Number(gc.maxOutputTokens);
|
||||
if (!Number.isFinite(current) || current < target) {
|
||||
gc.maxOutputTokens = target;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip every known thinking field from a body (used before re-applying / when unsupported).
|
||||
function stripAll(body) {
|
||||
delete body.thinking;
|
||||
@@ -158,7 +222,7 @@ function stripAll(body) {
|
||||
}
|
||||
|
||||
// Apply unified thinking config to body in the resolved provider-native format.
|
||||
function applyFormat(fmt, body, cfg, caps) {
|
||||
function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
const none = cfg.mode === "none";
|
||||
const canDisable = caps.thinkingCanDisable !== false;
|
||||
// Model cannot disable thinking → clamp "none" to minimal effort instead.
|
||||
@@ -168,11 +232,17 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "openai": {
|
||||
if (none && canDisable) { body.reasoning_effort = "none"; break; }
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level;
|
||||
if (level) body.reasoning_effort = normalizeOpenAILevel(level, supportedLevels);
|
||||
break;
|
||||
}
|
||||
case "claude-adaptive": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
// output_config.effort alone does NOT turn thinking on: Anthropic requires
|
||||
// an explicit thinking:{type:"adaptive"} on Opus 4.6/4.7/4.8 and Sonnet 4.6
|
||||
// ("thinking is off unless you explicitly set it"), and Anthropic-compatible
|
||||
// shims (e.g. GitHub Copilot /v1/messages) default thinking off even for
|
||||
// Sonnet 5. Send both fields — the documented adaptive-thinking shape.
|
||||
body.thinking = { type: "adaptive" };
|
||||
const level = toLevel(eff);
|
||||
body.output_config = { effort: level === "xhigh" ? "high" : level };
|
||||
break;
|
||||
@@ -186,12 +256,14 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "gemini-level": {
|
||||
const level = none ? "minimal" : toGeminiThinkingLevel(eff);
|
||||
setGeminiThinking(body, { thinkingLevel: level, includeThoughts: level !== "minimal" });
|
||||
ensureGeminiOutputFloor(body, geminiLevelOutputFloor(level), caps);
|
||||
break;
|
||||
}
|
||||
case "gemini-budget": {
|
||||
if (none && canDisable) { setGeminiThinking(body, { thinkingBudget: 0, includeThoughts: false }); break; }
|
||||
const budget = toBudget(eff, caps.thinkingRange);
|
||||
setGeminiThinking(body, { thinkingBudget: budget ?? -1, includeThoughts: true });
|
||||
ensureGeminiOutputFloor(body, geminiBudgetOutputFloor(budget ?? -1), caps);
|
||||
break;
|
||||
}
|
||||
case "zai": {
|
||||
@@ -217,8 +289,8 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
}
|
||||
case "kimi": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level === "max" ? "high" : level;
|
||||
const effort = toKimiReasoningEffort(eff);
|
||||
if (effort) body.reasoning_effort = effort;
|
||||
break;
|
||||
}
|
||||
case "minimax": {
|
||||
@@ -238,6 +310,15 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
if (level) body.reasoning_effort = level === "xhigh" || level === "max" ? "high" : level;
|
||||
break;
|
||||
}
|
||||
case "tokenrouter": {
|
||||
// TokenRouter's reasoning_effort enum is low/medium/high/xhigh/max — it rejects
|
||||
// "none"/"auto" with a 400 and supports "max" natively (no clamp like openai).
|
||||
// "none" → omit the field so the upstream default applies; pass levels through.
|
||||
if (none || eff.mode === "auto") break;
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level;
|
||||
break;
|
||||
}
|
||||
case "kiro":
|
||||
// Kiro thinking handled via system-tag injection in openai-to-kiro.js; no body field here.
|
||||
break;
|
||||
@@ -265,7 +346,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
|
||||
if (!cfg) return body;
|
||||
|
||||
const fmt = resolveFormat(targetFormat, cleanModel, provider);
|
||||
const supportedLevels = getThinkingLevels(provider, cleanModel);
|
||||
stripAll(body);
|
||||
applyFormat(fmt, body, cfg, caps);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels);
|
||||
return body;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user