fix(responses): report usage on response.completed so clients can auto-compact
Map upstream Chat Completions usage to the Responses API shape and attach it to response.completed. Capture chunk.usage before the empty-choices guard so the usage-only trailer chunk survives, and defer completion to flushEvents() when usage is not yet known — only on the direct openai:openai-responses route, since a pivoted stream never reaches flushEvents. Fixes #3432.
This commit is contained in:
@@ -60,7 +60,13 @@ export function createSSEStream(options = {}) {
|
||||
const decoder = new TextDecoder("utf-8", { fatal: false });
|
||||
|
||||
const state = mode === STREAM_MODE.TRANSLATE
|
||||
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null }
|
||||
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null,
|
||||
// Which upstream format this stream came from. A response translator can be
|
||||
// reached either directly (target === its registered source) or as the second
|
||||
// hop of a pivot, and on the terminal null chunk the pivot drops it — so a
|
||||
// translator that defers closing events until flush needs to know which case
|
||||
// it is in. Absent/undefined means "unknown", i.e. do not defer.
|
||||
targetFormat }
|
||||
: null;
|
||||
|
||||
let totalContentLength = 0;
|
||||
|
||||
Reference in New Issue
Block a user