perf(usage): bound lastUsed overlay scan to 2-day window; reach max thinking tier
getUsageStats("all") shipped the entire usageHistory table to JS just to
refine lastUsed (~2s on 290K rows, on every statsEmitter update per SSE
listener). Bound the overlay to a 2-day indexed range scan; older entries
keep day-level lastUsed from usageDaily aggregates. Totals unaffected.
budgetToLevel now maps budgets > 80384 (midpoint of 32768/128000) to
"max" instead of clamping to "xhigh", so the top reasoning tier is
reachable from large budget_tokens requests.
This commit is contained in:
committed by
decolua
parent
ce9ac43da5
commit
d1de324586
@@ -34,6 +34,8 @@ export function effortToThinkingLevel(effort) {
|
||||
|
||||
// Numeric budget → nearest discrete level (reverse map via thresholds).
|
||||
// Returns null when budget <= 0 (no reasoning).
|
||||
// Thresholds are midpoints between LEVEL_TO_BUDGET values: max (128000) is
|
||||
// reachable, with the xhigh/max boundary at the 32768/128000 midpoint (80384).
|
||||
export function budgetToLevel(budget) {
|
||||
const b = Number(budget);
|
||||
if (!b || b <= 0) return null;
|
||||
@@ -41,7 +43,8 @@ export function budgetToLevel(budget) {
|
||||
if (b <= 4096) return "low";
|
||||
if (b <= 16384) return "medium";
|
||||
if (b <= 28672) return "high";
|
||||
return "xhigh";
|
||||
if (b <= 80384) return "xhigh";
|
||||
return "max";
|
||||
}
|
||||
|
||||
// Gemini thinkingBudget (numeric) → OpenAI reasoning_effort (antigravity reverse map).
|
||||
|
||||
@@ -537,8 +537,15 @@ export async function getUsageStats(period = "all") {
|
||||
}
|
||||
}
|
||||
|
||||
// Overlay precise lastUsed timestamps from history
|
||||
const overlayCutoff = maxDays ? Date.now() - maxDays * 86400000 : 0;
|
||||
// Overlay precise lastUsed timestamps from history.
|
||||
// ponytail: overlay scans only a recent window; entries older than that keep
|
||||
// day-level lastUsed from usageDaily. Upgrade to a materialized per-key
|
||||
// MAX(timestamp) table if exact old timestamps ever matter.
|
||||
const OVERLAY_WINDOW_MS = 2 * 86400000;
|
||||
const overlayCutoff = Math.max(
|
||||
maxDays ? Date.now() - maxDays * 86400000 : 0,
|
||||
Date.now() - OVERLAY_WINDOW_MS
|
||||
);
|
||||
const histRows = db.all(
|
||||
`SELECT timestamp, provider, model, connectionId, apiKey, endpoint FROM usageHistory WHERE timestamp >= ?`,
|
||||
[new Date(overlayCutoff).toISOString()]
|
||||
|
||||
39
tests/unit/thinking-budget-max-level.test.js
Normal file
39
tests/unit/thinking-budget-max-level.test.js
Normal file
@@ -0,0 +1,39 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { budgetToLevel } from "../../open-sse/translator/concerns/thinking.js";
|
||||
import { applyThinking } from "../../open-sse/translator/concerns/thinkingUnified.js";
|
||||
import { FORMATS } from "../../open-sse/translator/formats.js";
|
||||
|
||||
// Reverse map must be able to reach "max": LEVEL_TO_BUDGET.max = 128000 and
|
||||
// xhigh = 32768, so the xhigh/max threshold is their midpoint (80384).
|
||||
// Previously any budget > 28672 collapsed to "xhigh", making "max"
|
||||
// unreachable from Claude Code budget_tokens — its default thinking budget
|
||||
// (MAX_THINKING_TOKENS) could never produce effort "max".
|
||||
describe("budgetToLevel reaches max tier", () => {
|
||||
it("budget 98304 → \"max\"", () => {
|
||||
expect(budgetToLevel(98304)).toBe("max");
|
||||
});
|
||||
|
||||
it("budget 128000 → \"max\"", () => {
|
||||
expect(budgetToLevel(128000)).toBe("max");
|
||||
});
|
||||
|
||||
it("budget 80385 → \"max\" (just above midpoint)", () => {
|
||||
expect(budgetToLevel(80385)).toBe("max");
|
||||
});
|
||||
|
||||
it("budget 80384 → \"xhigh\" (midpoint still xhigh)", () => {
|
||||
expect(budgetToLevel(80384)).toBe("xhigh");
|
||||
});
|
||||
|
||||
it("budget 31999 stays \"xhigh\"", () => {
|
||||
expect(budgetToLevel(31999)).toBe("xhigh");
|
||||
});
|
||||
});
|
||||
|
||||
describe("applyThinking (openai-responses): large budgets map to max effort", () => {
|
||||
it("budget 98304 → reasoning_effort \"max\" for gpt-5.6-sol (openai wire)", () => {
|
||||
const body = { thinking: { type: "enabled", budget_tokens: 98304 } };
|
||||
const out = applyThinking(FORMATS.OPENAI_RESPONSES, "gpt-5.6-sol", body, "codex");
|
||||
expect(out?.reasoning_effort).toBe("max");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user