fix(translator): drop replayed reasoning fields for Groq/Mistral/Cerebras (#4220)
Strict OpenAI-compatible validators reject unknown assistant-message
fields: Groq 400 ("property 'reasoning_content' is unsupported"),
Mistral 422 ("extra_forbidden"), Cerebras 400 ("wrong_api_format").
Clients driving reasoning models (Hermes Agent, and anything following
the DeepSeek/Kimi convention) echo the previous turn's reasoning_content
on every assistant message, so from the second turn on every request to
these providers fails and a fallback combo silently skips them.
Add a dropMessageFields rule to paramSupport.js that strips
reasoning_content / reasoning / reasoning_details from assistant turns
for groq, mistral, and cerebras.
This commit is contained in:
@@ -24,6 +24,15 @@ const STRIP_RULES = [
|
|||||||
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
|
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
|
||||||
// min() with the model ceiling still applies if a variant's own limit is lower.
|
// min() with the model ceiling still applies if a variant's own limit is lower.
|
||||||
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
|
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
|
||||||
|
// Strict OpenAI-compatible validators reject unknown assistant-message fields.
|
||||||
|
// Clients that talk to reasoning models (e.g. Hermes) echo the prior turn's
|
||||||
|
// reasoning back on every assistant message; Groq answers 400 and Mistral 422
|
||||||
|
// ("extra_forbidden") on it, which knocks these providers out of every
|
||||||
|
// multi-turn combo. Providers that *require* the field (DeepSeek, Kimi) are
|
||||||
|
// handled by reasoningContentInjector and are not listed here.
|
||||||
|
{ provider: "groq", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
|
||||||
|
{ provider: "mistral", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
|
||||||
|
{ provider: "cerebras", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
|
||||||
];
|
];
|
||||||
|
|
||||||
// Test a rule's match (regex or predicate) against the model id.
|
// Test a rule's match (regex or predicate) against the model id.
|
||||||
@@ -47,6 +56,15 @@ export function stripUnsupportedParams(provider, model, body) {
|
|||||||
for (const key of rule.drop || []) {
|
for (const key of rule.drop || []) {
|
||||||
if (body[key] !== undefined) delete body[key];
|
if (body[key] !== undefined) delete body[key];
|
||||||
}
|
}
|
||||||
|
// Per-message field drop (assistant turns only — that is where clients replay reasoning).
|
||||||
|
if (Array.isArray(rule.dropMessageFields) && Array.isArray(body.messages)) {
|
||||||
|
for (const msg of body.messages) {
|
||||||
|
if (!msg || msg.role !== "assistant") continue;
|
||||||
|
for (const key of rule.dropMessageFields) {
|
||||||
|
if (msg[key] !== undefined) delete msg[key];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
// CF Workers AI oneOf root schema only accepts content as plain string (#1926)
|
// CF Workers AI oneOf root schema only accepts content as plain string (#1926)
|
||||||
if (rule.flattenContent && Array.isArray(body.messages)) {
|
if (rule.flattenContent && Array.isArray(body.messages)) {
|
||||||
for (const msg of body.messages) {
|
for (const msg of body.messages) {
|
||||||
|
|||||||
@@ -52,4 +52,48 @@ describe("stripUnsupportedParams", () => {
|
|||||||
|
|
||||||
expect(body.max_tokens).toBe(64000);
|
expect(body.max_tokens).toBe(64000);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("drops replayed reasoning fields from assistant messages for strict providers", () => {
|
||||||
|
const makeBody = () => ({
|
||||||
|
messages: [
|
||||||
|
{ role: "user", content: "hi" },
|
||||||
|
{
|
||||||
|
role: "assistant",
|
||||||
|
content: "hello",
|
||||||
|
reasoning_content: "thinking...",
|
||||||
|
reasoning: "thinking...",
|
||||||
|
reasoning_details: [{ text: "thinking..." }],
|
||||||
|
tool_calls: [{ id: "c1", type: "function", function: { name: "f", arguments: "{}" } }],
|
||||||
|
},
|
||||||
|
{ role: "user", content: "again", reasoning_content: "user-side field stays" },
|
||||||
|
],
|
||||||
|
});
|
||||||
|
|
||||||
|
for (const [provider, model] of [
|
||||||
|
["groq", "openai/gpt-oss-120b"],
|
||||||
|
["mistral", "codestral-latest"],
|
||||||
|
["cerebras", "gpt-oss-120b"],
|
||||||
|
]) {
|
||||||
|
const body = makeBody();
|
||||||
|
stripUnsupportedParams(provider, model, body);
|
||||||
|
expect(body.messages[1]).toEqual({
|
||||||
|
role: "assistant",
|
||||||
|
content: "hello",
|
||||||
|
tool_calls: [{ id: "c1", type: "function", function: { name: "f", arguments: "{}" } }],
|
||||||
|
});
|
||||||
|
// only assistant turns are touched
|
||||||
|
expect(body.messages[2].reasoning_content).toBe("user-side field stays");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("leaves reasoning fields alone for providers that accept or require them", () => {
|
||||||
|
const body = {
|
||||||
|
messages: [{ role: "assistant", content: "hello", reasoning_content: "thinking..." }],
|
||||||
|
};
|
||||||
|
|
||||||
|
stripUnsupportedParams("deepseek", "deepseek-reasoner", body);
|
||||||
|
stripUnsupportedParams("openrouter", "nvidia/nemotron-3-ultra-550b-a55b:free", body);
|
||||||
|
|
||||||
|
expect(body.messages[0].reasoning_content).toBe("thinking...");
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|||||||
Reference in New Issue
Block a user