revert(qoder): drop the Responses usage plumbing from shared code

The merged Qoder work also rewrote shared translator/handler code so that
/v1/responses clients got token usage on response.completed. That changed
behaviour for every provider, not just Qoder: proxies saw input tokens
rise by the 2000-token context buffer, and the plain token mapping was
replaced by one that always adds input_tokens_details.

A probe confirms the Qoder benefit does not depend on those edits: the
executor's coalescer already emits one include_usage-style finish chunk, so
a Claude client receives input_tokens and cache_read_input_tokens with
every shared file at its original state. Only the Responses path relies on
the shared translator, and that path has no Qoder-owned seam to put it in.

Reverts the shared files to their pre-PR state and drops the Responses
usage test. The Cline envelope unwrap in nonStreamingHandler.js, which
landed after the PR in the same file, is kept.
This commit is contained in:
LLL
2026-09-10 22:52:29 +07:00
parent 998bb3d975
commit 248d7da01c
9 changed files with 74 additions and 341 deletions

View File

@@ -3,7 +3,6 @@ import { FORMATS } from "../translator/formats.js";
import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb.js";
import { extractUsage, mergeUsage, hasValidUsage, estimateUsage, logUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js";
import { parseSSELine, hasValuableContent, fixInvalidId, formatSSE } from "./streamHelpers.js";
import { toResponsesUsage } from "../translator/concerns/usage.js";
import { getOpenAIResponsesEventName, isOpenAIResponsesTerminalEvent, formatIncompleteOpenAIResponsesStreamFailure } from "./responsesStreamHelpers.js";
import { dbg, isDebugEnabled } from "./debugLog.js";
@@ -202,8 +201,7 @@ export function createSSEStream(options = {}) {
responsesTerminal = isOpenAIResponsesTerminalEvent(currentOpenAIResponsesEvent, parsed);
const isFinishChunk = parsed.choices?.[0]?.finish_reason
|| parsed.choices?.[0]?.delta?.finish_reason;
const isFinishChunk = parsed.choices?.[0]?.finish_reason;
if (isFinishChunk && !hasValidUsage(parsed.usage)) {
const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI);
parsed.usage = filterUsageForFormat(estimated, FORMATS.OPENAI);
@@ -367,19 +365,6 @@ export function createSSEStream(options = {}) {
item.usage = filterUsageForFormat(buffered, sourceFormat);
}
// Responses API clients (Codex, sub2api /v1/responses): usage lives on
// response.completed → response.usage. Same buffer/estimate policy as above.
const completedResponse = item.event === "response.completed" ? item.data?.response : null;
if (completedResponse && typeof completedResponse === "object") {
if (state.usage) {
completedResponse.usage = toResponsesUsage(addBufferToUsage(state.usage)) ?? completedResponse.usage;
} else if (!completedResponse.usage && totalContentLength > 0) {
const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI);
completedResponse.usage = toResponsesUsage(estimated);
state.usage = estimated;
}
}
const output = formatSSE(item, sourceFormat);
reqLogger?.appendConvertedChunk?.(output);
controller.enqueue(sharedEncoder.encode(output));