revert(qoder): drop the Responses usage plumbing from shared code

The merged Qoder work also rewrote shared translator/handler code so that
/v1/responses clients got token usage on response.completed. That changed
behaviour for every provider, not just Qoder: proxies saw input tokens
rise by the 2000-token context buffer, and the plain token mapping was
replaced by one that always adds input_tokens_details.

A probe confirms the Qoder benefit does not depend on those edits: the
executor's coalescer already emits one include_usage-style finish chunk, so
a Claude client receives input_tokens and cache_read_input_tokens with
every shared file at its original state. Only the Responses path relies on
the shared translator, and that path has no Qoder-owned seam to put it in.

Reverts the shared files to their pre-PR state and drops the Responses
usage test. The Cline envelope unwrap in nonStreamingHandler.js, which
landed after the PR in the same file, is kept.
This commit is contained in:
LLL
2026-09-10 22:52:29 +07:00
parent 998bb3d975
commit 248d7da01c
9 changed files with 74 additions and 341 deletions

View File

@@ -67,34 +67,3 @@ export function toOpenAIUsage(raw, kind) {
if (!extract || !raw || typeof raw !== "object") return null;
return buildUsage(extract(raw));
}
// Convert an OpenAI-shaped (or already-canonical / Claude-shaped) usage object into the
// Responses API shape emitted by `response.completed`. Details objects are always present
// (like the real API) so proxies that read `input_tokens_details.cached_tokens` never see undefined.
// Returns null when there is nothing countable.
export function toResponsesUsage(usage) {
if (!usage || typeof usage !== "object") return null;
const input = n(usage.prompt_tokens ?? usage.input_tokens);
const output = n(usage.completion_tokens ?? usage.output_tokens);
if (input === 0 && output === 0) return null;
const cached = n(
usage.input_tokens_details?.cached_tokens ??
usage.prompt_tokens_details?.cached_tokens ??
usage.cached_tokens ??
usage.cache_read_input_tokens
);
const reasoning = n(
usage.output_tokens_details?.reasoning_tokens ??
usage.completion_tokens_details?.reasoning_tokens ??
usage.reasoning_tokens
);
const out = {
input_tokens: input,
output_tokens: output,
total_tokens: typeof usage.total_tokens === "number" ? usage.total_tokens : input + output,
input_tokens_details: { cached_tokens: cached },
output_tokens_details: { reasoning_tokens: reasoning },
};
if (usage.estimated) out.estimated = true;
return out;
}