fix(translator): zai thinkingFormat sends reasoning.effort object
Z.ai / GLM-5.2+ require a top-level reasoning_effort (low/high/max)
alongside thinking:{type:"enabled"} to control reasoning depth; the zai
branch previously only set thinking and dropped reasoning_effort, so every
GLM-5.x request ran at the model default (max). Gate the field behind
GLM-5.2+ (thinkingEffortSupported in capabilities.js) since older GLM
(4.x, 5.0, 5.1, 5-turbo, 5v-turbo) do not read it, and map client levels
to the exact low/high/max values z.ai accepts.
extractThinking now checks reasoning_effort/reasoning.effort before the
thinking object so a client-supplied effort is not overwritten by
thinking:{type:"enabled"} mapping to mode:auto.
Fixes #2721
This commit is contained in:
@@ -58,6 +58,15 @@ export function extractThinking(body) {
|
||||
return { mode: "level", level: e };
|
||||
}
|
||||
|
||||
// OpenAI chat / Responses shape — check effort first (zai sends both thinking object and reasoning.effort)
|
||||
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
|
||||
if (typeof effort === "string" && effort) {
|
||||
const e = effort.toLowerCase();
|
||||
if (e === "none" || e === "off") return { mode: "none" };
|
||||
if (e === "auto") return { mode: "auto" };
|
||||
return { mode: "level", level: e };
|
||||
}
|
||||
|
||||
// Claude shape
|
||||
const t = body.thinking;
|
||||
if (t && typeof t === "object") {
|
||||
@@ -69,15 +78,6 @@ export function extractThinking(body) {
|
||||
}
|
||||
}
|
||||
|
||||
// OpenAI chat / Responses shape
|
||||
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
|
||||
if (typeof effort === "string" && effort) {
|
||||
const e = effort.toLowerCase();
|
||||
if (e === "none" || e === "off") return { mode: "none" };
|
||||
if (e === "auto") return { mode: "auto" };
|
||||
return { mode: "level", level: e };
|
||||
}
|
||||
|
||||
// Gemini shape (top-level, generationConfig, or request envelope)
|
||||
const tc = body.thinkingConfig || body.generationConfig?.thinkingConfig || body.request?.generationConfig?.thinkingConfig;
|
||||
if (tc && typeof tc === "object") {
|
||||
@@ -270,6 +270,18 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
// Z.ai ignores thinking.disabled → must use enable_thinking:false to turn off.
|
||||
if (none && canDisable) { body.enable_thinking = false; delete body.thinking; break; }
|
||||
body.thinking = { type: "enabled" };
|
||||
// reasoning_effort is only read by z.ai from GLM-5.2 onward — older GLM ignores it
|
||||
// (see thinkingEffortSupported in capabilities.js). Skip on unsupported models so we
|
||||
// don't send a field the API doesn't recognize.
|
||||
if (caps.thinkingEffortSupported) {
|
||||
const zaiLvl = toLevel(eff);
|
||||
// GLM-5.3 only accepts exactly low|high|max (anything else errors); GLM-5.2 accepts
|
||||
// a wider set but z.ai maps low/medium->high and xhigh->max server-side anyway, so
|
||||
// this 3-value mapping matches both.
|
||||
body.reasoning_effort = (zaiLvl === "low" || zaiLvl === "minimal") ? "low"
|
||||
: (zaiLvl === "high" || zaiLvl === "medium") ? "high"
|
||||
: "max";
|
||||
}
|
||||
break;
|
||||
}
|
||||
case "qwen": {
|
||||
|
||||
Reference in New Issue
Block a user