fix(translator): stop emitting empty <think> markers into OpenAI content
This commit is contained in:
@@ -333,6 +333,9 @@ export function createResponsesApiTransformStream(logger = null) {
|
|||||||
|
|
||||||
// Regular text content
|
// Regular text content
|
||||||
if (content) {
|
if (content) {
|
||||||
|
// The answer starts, so thinking is over. Upstreams that send reasoning via
|
||||||
|
// reasoning_content never emit "</think>", so close it here rather than at finish.
|
||||||
|
closeReasoning(controller);
|
||||||
if (!state.msgItemAdded[idx]) {
|
if (!state.msgItemAdded[idx]) {
|
||||||
state.msgItemAdded[idx] = true;
|
state.msgItemAdded[idx] = true;
|
||||||
const msgId = `msg_${state.responseId}_${idx}`;
|
const msgId = `msg_${state.responseId}_${idx}`;
|
||||||
@@ -372,6 +375,7 @@ export function createResponsesApiTransformStream(logger = null) {
|
|||||||
|
|
||||||
// Handle tool_calls
|
// Handle tool_calls
|
||||||
if (delta.tool_calls) {
|
if (delta.tool_calls) {
|
||||||
|
closeReasoning(controller);
|
||||||
closeMessage(controller, idx);
|
closeMessage(controller, idx);
|
||||||
|
|
||||||
for (const tc of delta.tool_calls) {
|
for (const tc of delta.tool_calls) {
|
||||||
|
|||||||
@@ -59,10 +59,6 @@ export function claudeToOpenAIResponse(chunk, state) {
|
|||||||
}
|
}
|
||||||
if (block?.type === CLAUDE_BLOCK.TEXT) {
|
if (block?.type === CLAUDE_BLOCK.TEXT) {
|
||||||
state.textBlockStarted = true;
|
state.textBlockStarted = true;
|
||||||
} else if (block?.type === CLAUDE_BLOCK.THINKING) {
|
|
||||||
state.inThinkingBlock = true;
|
|
||||||
state.currentBlockIndex = chunk.index;
|
|
||||||
results.push(createChunk(state, { content: "<think>" }));
|
|
||||||
} else if (block?.type === CLAUDE_BLOCK.TOOL_USE) {
|
} else if (block?.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||||
const toolCallIndex = state.toolCallIndex++;
|
const toolCallIndex = state.toolCallIndex++;
|
||||||
// Restore original tool name from mapping (Claude OAuth)
|
// Restore original tool name from mapping (Claude OAuth)
|
||||||
@@ -89,6 +85,8 @@ export function claudeToOpenAIResponse(chunk, state) {
|
|||||||
if (delta?.type === "text_delta" && delta.text) {
|
if (delta?.type === "text_delta" && delta.text) {
|
||||||
results.push(createChunk(state, { content: delta.text }));
|
results.push(createChunk(state, { content: delta.text }));
|
||||||
} else if (delta?.type === "thinking_delta" && delta.thinking) {
|
} else if (delta?.type === "thinking_delta" && delta.thinking) {
|
||||||
|
// Thinking travels only in reasoning_content. No "<think>" markers in
|
||||||
|
// content: OpenAI-format clients render them as literal text.
|
||||||
results.push(createChunk(state, reasoningDelta(delta.thinking)));
|
results.push(createChunk(state, reasoningDelta(delta.thinking)));
|
||||||
} else if (delta?.type === "input_json_delta" && delta.partial_json) {
|
} else if (delta?.type === "input_json_delta" && delta.partial_json) {
|
||||||
const toolCall = state.toolCalls.get(chunk.index);
|
const toolCall = state.toolCalls.get(chunk.index);
|
||||||
@@ -112,10 +110,6 @@ export function claudeToOpenAIResponse(chunk, state) {
|
|||||||
state.serverToolBlockIndex = -1;
|
state.serverToolBlockIndex = -1;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (state.inThinkingBlock && chunk.index === state.currentBlockIndex) {
|
|
||||||
results.push(createChunk(state, { content: "</think>" }));
|
|
||||||
state.inThinkingBlock = false;
|
|
||||||
}
|
|
||||||
state.textBlockStarted = false;
|
state.textBlockStarted = false;
|
||||||
state.thinkingBlockStarted = false;
|
state.thinkingBlockStarted = false;
|
||||||
break;
|
break;
|
||||||
|
|||||||
@@ -129,12 +129,16 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (content) {
|
if (content) {
|
||||||
|
// The answer starts, so thinking is over. Upstreams that send reasoning via
|
||||||
|
// reasoning_content never emit "</think>", so close it here rather than at finish.
|
||||||
|
closeReasoning(state, emit);
|
||||||
emitTextContent(state, emit, idx, content);
|
emitTextContent(state, emit, idx, content);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Handle tool_calls (empty array is truthy; require a real call)
|
// Handle tool_calls (empty array is truthy; require a real call)
|
||||||
if (delta.tool_calls && delta.tool_calls.length) {
|
if (delta.tool_calls && delta.tool_calls.length) {
|
||||||
|
closeReasoning(state, emit);
|
||||||
closeMessage(state, emit, idx);
|
closeMessage(state, emit, idx);
|
||||||
for (const tc of delta.tool_calls) {
|
for (const tc of delta.tool_calls) {
|
||||||
emitToolCall(state, emit, tc);
|
emitToolCall(state, emit, tc);
|
||||||
|
|||||||
@@ -17,21 +17,6 @@ exports[`GOLDEN response stream: Claude → OpenAI > text + thinking + tool_use
|
|||||||
"model": "claude-opus-4-6",
|
"model": "claude-opus-4-6",
|
||||||
"object": "chat.completion.chunk",
|
"object": "chat.completion.chunk",
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"choices": [
|
|
||||||
{
|
|
||||||
"delta": {
|
|
||||||
"content": "<think>",
|
|
||||||
},
|
|
||||||
"finish_reason": null,
|
|
||||||
"index": 0,
|
|
||||||
},
|
|
||||||
],
|
|
||||||
"created": 0,
|
|
||||||
"id": "chatcmpl-msg_1",
|
|
||||||
"model": "claude-opus-4-6",
|
|
||||||
"object": "chat.completion.chunk",
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"choices": [
|
"choices": [
|
||||||
{
|
{
|
||||||
@@ -47,21 +32,6 @@ exports[`GOLDEN response stream: Claude → OpenAI > text + thinking + tool_use
|
|||||||
"model": "claude-opus-4-6",
|
"model": "claude-opus-4-6",
|
||||||
"object": "chat.completion.chunk",
|
"object": "chat.completion.chunk",
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"choices": [
|
|
||||||
{
|
|
||||||
"delta": {
|
|
||||||
"content": "</think>",
|
|
||||||
},
|
|
||||||
"finish_reason": null,
|
|
||||||
"index": 0,
|
|
||||||
},
|
|
||||||
],
|
|
||||||
"created": 0,
|
|
||||||
"id": "chatcmpl-msg_1",
|
|
||||||
"model": "claude-opus-4-6",
|
|
||||||
"object": "chat.completion.chunk",
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"choices": [
|
"choices": [
|
||||||
{
|
{
|
||||||
|
|||||||
175
tests/unit/claude-thinking-stream-boundaries.test.js
Normal file
175
tests/unit/claude-thinking-stream-boundaries.test.js
Normal file
@@ -0,0 +1,175 @@
|
|||||||
|
// Thinking/answer boundaries across the OpenAI pivot.
|
||||||
|
//
|
||||||
|
// claude-to-openai used to mark a Claude thinking block with literal "<think>" /
|
||||||
|
// "</think>" chunks in delta.content while the thinking text itself went out in
|
||||||
|
// reasoning_content. The pair always arrived empty and adjacent, so OpenAI-format
|
||||||
|
// clients (opencode, DeepSeek Harness, ...) rendered a bare "<think></think>" above
|
||||||
|
// every answer (#3399, #4199).
|
||||||
|
//
|
||||||
|
// The Responses translators leaned on that "</think>" marker as their only signal
|
||||||
|
// to close the reasoning item before the answer. Dropping the marker therefore
|
||||||
|
// requires closing reasoning when the first message text or tool call arrives —
|
||||||
|
// which also fixes item ordering for every reasoning_content provider (DeepSeek,
|
||||||
|
// GLM, Qwen, Kimi), not just Claude.
|
||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { claudeToOpenAIResponse } from "../../open-sse/translator/response/claude-to-openai.js";
|
||||||
|
import { FORMATS } from "../../open-sse/translator/formats.js";
|
||||||
|
import { createSSETransformStreamWithLogger } from "../../open-sse/utils/stream.js";
|
||||||
|
import { createResponsesApiTransformStream } from "../../open-sse/transformer/responsesTransformer.js";
|
||||||
|
|
||||||
|
const THINKING = "391 factors as 17 times 23, so it's not prime.";
|
||||||
|
const ANSWER = "No — 391 = 17 × 23.";
|
||||||
|
|
||||||
|
function claudeThinkingStream({ thinkingText = THINKING, answer = ANSWER } = {}) {
|
||||||
|
const thinkingDeltas = thinkingText
|
||||||
|
? [{ type: "content_block_delta", index: 0, delta: { type: "thinking_delta", thinking: thinkingText } }]
|
||||||
|
: [];
|
||||||
|
return [
|
||||||
|
{ type: "message_start", message: { id: "msg_1", model: "claude-opus-5", role: "assistant", content: [], usage: { input_tokens: 10, output_tokens: 0 } } },
|
||||||
|
{ type: "content_block_start", index: 0, content_block: { type: "thinking", thinking: "", signature: "" } },
|
||||||
|
...thinkingDeltas,
|
||||||
|
{ type: "content_block_delta", index: 0, delta: { type: "signature_delta", signature: "sig_abc" } },
|
||||||
|
{ type: "content_block_stop", index: 0 },
|
||||||
|
{ type: "content_block_start", index: 1, content_block: { type: "text", text: "" } },
|
||||||
|
{ type: "content_block_delta", index: 1, delta: { type: "text_delta", text: answer } },
|
||||||
|
{ type: "content_block_stop", index: 1 },
|
||||||
|
{ type: "message_delta", delta: { stop_reason: "end_turn", stop_sequence: null }, usage: { output_tokens: 20 } },
|
||||||
|
{ type: "message_stop" },
|
||||||
|
];
|
||||||
|
}
|
||||||
|
|
||||||
|
function runClaudeToOpenAI(events) {
|
||||||
|
const state = {};
|
||||||
|
const out = [];
|
||||||
|
for (const ev of events) {
|
||||||
|
const r = claudeToOpenAIResponse(ev, state);
|
||||||
|
if (Array.isArray(r)) out.push(...r);
|
||||||
|
else if (r) out.push(r);
|
||||||
|
}
|
||||||
|
const deltas = out.map((c) => c.choices?.[0]?.delta || {});
|
||||||
|
return {
|
||||||
|
content: deltas.map((d) => d.content || "").join(""),
|
||||||
|
reasoning: deltas.map((d) => d.reasoning_content || "").join(""),
|
||||||
|
contentChunks: deltas.map((d) => d.content).filter((c) => c != null),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
async function drain(stream) {
|
||||||
|
const reader = stream.getReader();
|
||||||
|
const decoder = new TextDecoder();
|
||||||
|
let text = "";
|
||||||
|
for (;;) {
|
||||||
|
const { value, done } = await reader.read();
|
||||||
|
if (done) break;
|
||||||
|
text += typeof value === "string" ? value : decoder.decode(value, { stream: true });
|
||||||
|
}
|
||||||
|
return text + decoder.decode();
|
||||||
|
}
|
||||||
|
|
||||||
|
function sseStream(chunks) {
|
||||||
|
const encoder = new TextEncoder();
|
||||||
|
const body = chunks.map((c) => `data: ${JSON.stringify(c)}\n\n`).join("") + "data: [DONE]\n\n";
|
||||||
|
return new ReadableStream({
|
||||||
|
start(controller) {
|
||||||
|
controller.enqueue(encoder.encode(body));
|
||||||
|
controller.close();
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Upstream speaks `upstream`, client speaks the Responses API.
|
||||||
|
async function viaResponsesTranslator(chunks, upstream, provider, model) {
|
||||||
|
const out = sseStream(chunks).pipeThrough(
|
||||||
|
createSSETransformStreamWithLogger(upstream, FORMATS.OPENAI_RESPONSES, provider, null, null, model),
|
||||||
|
);
|
||||||
|
return parseEvents(await drain(out));
|
||||||
|
}
|
||||||
|
|
||||||
|
async function viaResponsesTransformer(chunks) {
|
||||||
|
return parseEvents(await drain(sseStream(chunks).pipeThrough(createResponsesApiTransformStream(null))));
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseEvents(text) {
|
||||||
|
return text
|
||||||
|
.split("\n")
|
||||||
|
.filter((l) => l.startsWith("data: ") && l.slice(6).trim() !== "[DONE]")
|
||||||
|
.map((l) => {
|
||||||
|
try { return JSON.parse(l.slice(6)); } catch { return null; }
|
||||||
|
})
|
||||||
|
.filter(Boolean);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Index of the event announcing/finishing an output item of the given type.
|
||||||
|
function itemEventIndex(events, eventType, itemType) {
|
||||||
|
return events.findIndex((e) => e.type === eventType && e.item?.type === itemType);
|
||||||
|
}
|
||||||
|
|
||||||
|
function expectReasoningClosedBefore(events, nextItemType) {
|
||||||
|
const reasoningDone = itemEventIndex(events, "response.output_item.done", "reasoning");
|
||||||
|
const nextAdded = itemEventIndex(events, "response.output_item.added", nextItemType);
|
||||||
|
expect(reasoningDone).toBeGreaterThanOrEqual(0);
|
||||||
|
expect(nextAdded).toBeGreaterThanOrEqual(0);
|
||||||
|
expect(reasoningDone).toBeLessThan(nextAdded);
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("claude-to-openai: thinking never leaks markers into content", () => {
|
||||||
|
it("summarized thinking goes to reasoning_content, answer to content, no <think> text", () => {
|
||||||
|
const { content, reasoning, contentChunks } = runClaudeToOpenAI(claudeThinkingStream());
|
||||||
|
expect(reasoning).toBe(THINKING);
|
||||||
|
expect(content).toBe(ANSWER);
|
||||||
|
expect(contentChunks.some((c) => c.includes("<think>") || c.includes("</think>"))).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("signature-only (redacted) thinking yields no stray markers", () => {
|
||||||
|
const { content, reasoning } = runClaudeToOpenAI(claudeThinkingStream({ thinkingText: "" }));
|
||||||
|
expect(reasoning).toBe("");
|
||||||
|
expect(content).toBe(ANSWER);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("Responses translator: reasoning closes before the answer", () => {
|
||||||
|
it("Claude upstream: reasoning item is done before the message item opens", async () => {
|
||||||
|
const events = await viaResponsesTranslator(claudeThinkingStream(), FORMATS.CLAUDE, "claude", "claude-opus-5");
|
||||||
|
expectReasoningClosedBefore(events, "message");
|
||||||
|
const summary = events.find((e) => e.type === "response.reasoning_summary_text.done");
|
||||||
|
expect(summary?.text).toBe(THINKING);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("reasoning_content upstream: reasoning item is done before the message item opens", async () => {
|
||||||
|
const events = await viaResponsesTranslator([
|
||||||
|
{ id: "c1", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] },
|
||||||
|
{ id: "c1", choices: [{ index: 0, delta: { content: ANSWER } }] },
|
||||||
|
{ id: "c1", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] },
|
||||||
|
], FORMATS.OPENAI, "deepseek", "deepseek-flash");
|
||||||
|
expectReasoningClosedBefore(events, "message");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("reasoning_content upstream: reasoning item is done before a tool call opens", async () => {
|
||||||
|
const events = await viaResponsesTranslator([
|
||||||
|
{ id: "c2", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] },
|
||||||
|
{ id: "c2", choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_1", type: "function", function: { name: "lookup", arguments: "{}" } }] } }] },
|
||||||
|
{ id: "c2", choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }] },
|
||||||
|
], FORMATS.OPENAI, "deepseek", "deepseek-flash");
|
||||||
|
expectReasoningClosedBefore(events, "function_call");
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("responsesTransformer (/v1/responses handler): reasoning closes before the answer", () => {
|
||||||
|
it("reasoning item is done before the message item opens", async () => {
|
||||||
|
const events = await viaResponsesTransformer([
|
||||||
|
{ id: "c3", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] },
|
||||||
|
{ id: "c3", choices: [{ index: 0, delta: { content: ANSWER } }] },
|
||||||
|
{ id: "c3", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] },
|
||||||
|
]);
|
||||||
|
expectReasoningClosedBefore(events, "message");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("reasoning item is done before a tool call opens", async () => {
|
||||||
|
const events = await viaResponsesTransformer([
|
||||||
|
{ id: "c4", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] },
|
||||||
|
{ id: "c4", choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_2", type: "function", function: { name: "lookup", arguments: "{}" } }] } }] },
|
||||||
|
{ id: "c4", choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }] },
|
||||||
|
]);
|
||||||
|
expectReasoningClosedBefore(events, "function_call");
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user