fix(translator): zai thinkingFormat sends reasoning.effort object

Z.ai / GLM-5.2+ require a top-level reasoning_effort (low/high/max)
alongside thinking:{type:"enabled"} to control reasoning depth; the zai
branch previously only set thinking and dropped reasoning_effort, so every
GLM-5.x request ran at the model default (max). Gate the field behind
GLM-5.2+ (thinkingEffortSupported in capabilities.js) since older GLM
(4.x, 5.0, 5.1, 5-turbo, 5v-turbo) do not read it, and map client levels
to the exact low/high/max values z.ai accepts.

extractThinking now checks reasoning_effort/reasoning.effort before the
thinking object so a client-supplied effort is not overwritten by
thinking:{type:"enabled"} mapping to mode:auto.

Fixes #2721
This commit is contained in:
qingyong
2026-08-28 12:32:26 +07:00
committed by decolua
parent cadef6c4ff
commit 56a40765e9
3 changed files with 59 additions and 9 deletions

View File

@@ -57,6 +57,7 @@ export const DEFAULT_CAPABILITIES = {
thinkingFormat: null,
thinkingCanDisable: true, // false → model cannot turn thinking off (clamp to min instead of disable)
thinkingRange: null, // { min, max } for budget formats; null = no clamp
thinkingEffortSupported: false, // zai format only: model accepts a reasoning_effort level (GLM-5.2+; older GLM ignores it)
// limits (tokens)
contextWindow: 200000,
maxOutput: 64000,
@@ -282,6 +283,10 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*kimi*", caps: { reasoning: true, thinkingFormat: "kimi", contextWindow: 262144 } },
// ── GLM / Z.ai (thinking.enabled; disable via enable_thinking:false) ─
// reasoning_effort is only read by z.ai from GLM-5.2 onward (docs.z.ai/guides/capabilities/thinking) —
// older GLM (4.x, 5.0, 5.1, 5-turbo, 5v-turbo) ignore it, so gate it per exact version, not the "*glm-5*" catch-all.
{ pattern: "*glm-5.3*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-5.2*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-5*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-4.7*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-4*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },

View File

@@ -58,6 +58,15 @@ export function extractThinking(body) {
return { mode: "level", level: e };
}
// OpenAI chat / Responses shape — check effort first (zai sends both thinking object and reasoning.effort)
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
if (typeof effort === "string" && effort) {
const e = effort.toLowerCase();
if (e === "none" || e === "off") return { mode: "none" };
if (e === "auto") return { mode: "auto" };
return { mode: "level", level: e };
}
// Claude shape
const t = body.thinking;
if (t && typeof t === "object") {
@@ -69,15 +78,6 @@ export function extractThinking(body) {
}
}
// OpenAI chat / Responses shape
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
if (typeof effort === "string" && effort) {
const e = effort.toLowerCase();
if (e === "none" || e === "off") return { mode: "none" };
if (e === "auto") return { mode: "auto" };
return { mode: "level", level: e };
}
// Gemini shape (top-level, generationConfig, or request envelope)
const tc = body.thinkingConfig || body.generationConfig?.thinkingConfig || body.request?.generationConfig?.thinkingConfig;
if (tc && typeof tc === "object") {
@@ -270,6 +270,18 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
// Z.ai ignores thinking.disabled → must use enable_thinking:false to turn off.
if (none && canDisable) { body.enable_thinking = false; delete body.thinking; break; }
body.thinking = { type: "enabled" };
// reasoning_effort is only read by z.ai from GLM-5.2 onward — older GLM ignores it
// (see thinkingEffortSupported in capabilities.js). Skip on unsupported models so we
// don't send a field the API doesn't recognize.
if (caps.thinkingEffortSupported) {
const zaiLvl = toLevel(eff);
// GLM-5.3 only accepts exactly low|high|max (anything else errors); GLM-5.2 accepts
// a wider set but z.ai maps low/medium->high and xhigh->max server-side anyway, so
// this 3-value mapping matches both.
body.reasoning_effort = (zaiLvl === "low" || zaiLvl === "minimal") ? "low"
: (zaiLvl === "high" || zaiLvl === "medium") ? "high"
: "max";
}
break;
}
case "qwen": {

View File

@@ -58,6 +58,18 @@ describe("extractThinking", () => {
it("no intent → null", () => {
expect(extractThinking({ messages: [] })).toBeNull();
});
it("reasoning_effort wins over thinking:{type:enabled} (no budget)", () => {
expect(extractThinking({
thinking: { type: "enabled" },
reasoning_effort: "high",
})).toEqual({ mode: "level", level: "high" });
});
it("reasoning.effort wins over thinking:{type:enabled} (no budget)", () => {
expect(extractThinking({
thinking: { type: "enabled" },
reasoning: { effort: "medium" },
})).toEqual({ mode: "level", level: "medium" });
});
});
describe("applyThinking per provider format", () => {
@@ -114,6 +126,27 @@ describe("applyThinking per provider format", () => {
expect(out.enable_thinking).toBe(false);
expect(out.thinking).toBeUndefined();
});
it.each([
["high", "high"],
["max", "max"],
["xhigh", "max"],
["low", "low"],
["medium", "high"],
["minimal", "low"],
])("GLM-5.3 %s → reasoning_effort=%s (low|high|max only, per z.ai docs)", (input, expected) => {
const out = apply("openai", "glm-5.3", { reasoning_effort: input }, "glm-cn");
expect(out.thinking).toEqual({ type: "enabled" });
expect(out.reasoning_effort).toBe(expected);
});
it("GLM-5.2 also gets reasoning_effort (supported from 5.2 onward)", () => {
const out = apply("openai", "glm-5.2", { reasoning_effort: "low" }, "glm-cn");
expect(out.reasoning_effort).toBe("low");
});
it("GLM-4.7 (pre-5.2) does not get reasoning_effort — z.ai ignores it", () => {
const out = apply("openai", "glm-4.7", { reasoning_effort: "low" }, "glm-cn");
expect(out.thinking).toEqual({ type: "enabled" });
expect(out.reasoning_effort).toBeUndefined();
});
it("Qwen on → enable_thinking + thinking_budget", () => {
const out = apply("openai", "qwen3-max", { reasoning_effort: "medium" }, "qwen");
expect(out.enable_thinking).toBe(true);