perf(usage): bound lastUsed overlay scan to 2-day window; reach max thinking tier
getUsageStats("all") shipped the entire usageHistory table to JS just to
refine lastUsed (~2s on 290K rows, on every statsEmitter update per SSE
listener). Bound the overlay to a 2-day indexed range scan; older entries
keep day-level lastUsed from usageDaily aggregates. Totals unaffected.
budgetToLevel now maps budgets > 80384 (midpoint of 32768/128000) to
"max" instead of clamping to "xhigh", so the top reasoning tier is
reachable from large budget_tokens requests.
This commit is contained in:
committed by
decolua
parent
ce9ac43da5
commit
d1de324586
@@ -34,6 +34,8 @@ export function effortToThinkingLevel(effort) {
|
|||||||
|
|
||||||
// Numeric budget → nearest discrete level (reverse map via thresholds).
|
// Numeric budget → nearest discrete level (reverse map via thresholds).
|
||||||
// Returns null when budget <= 0 (no reasoning).
|
// Returns null when budget <= 0 (no reasoning).
|
||||||
|
// Thresholds are midpoints between LEVEL_TO_BUDGET values: max (128000) is
|
||||||
|
// reachable, with the xhigh/max boundary at the 32768/128000 midpoint (80384).
|
||||||
export function budgetToLevel(budget) {
|
export function budgetToLevel(budget) {
|
||||||
const b = Number(budget);
|
const b = Number(budget);
|
||||||
if (!b || b <= 0) return null;
|
if (!b || b <= 0) return null;
|
||||||
@@ -41,7 +43,8 @@ export function budgetToLevel(budget) {
|
|||||||
if (b <= 4096) return "low";
|
if (b <= 4096) return "low";
|
||||||
if (b <= 16384) return "medium";
|
if (b <= 16384) return "medium";
|
||||||
if (b <= 28672) return "high";
|
if (b <= 28672) return "high";
|
||||||
return "xhigh";
|
if (b <= 80384) return "xhigh";
|
||||||
|
return "max";
|
||||||
}
|
}
|
||||||
|
|
||||||
// Gemini thinkingBudget (numeric) → OpenAI reasoning_effort (antigravity reverse map).
|
// Gemini thinkingBudget (numeric) → OpenAI reasoning_effort (antigravity reverse map).
|
||||||
|
|||||||
@@ -537,8 +537,15 @@ export async function getUsageStats(period = "all") {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Overlay precise lastUsed timestamps from history
|
// Overlay precise lastUsed timestamps from history.
|
||||||
const overlayCutoff = maxDays ? Date.now() - maxDays * 86400000 : 0;
|
// ponytail: overlay scans only a recent window; entries older than that keep
|
||||||
|
// day-level lastUsed from usageDaily. Upgrade to a materialized per-key
|
||||||
|
// MAX(timestamp) table if exact old timestamps ever matter.
|
||||||
|
const OVERLAY_WINDOW_MS = 2 * 86400000;
|
||||||
|
const overlayCutoff = Math.max(
|
||||||
|
maxDays ? Date.now() - maxDays * 86400000 : 0,
|
||||||
|
Date.now() - OVERLAY_WINDOW_MS
|
||||||
|
);
|
||||||
const histRows = db.all(
|
const histRows = db.all(
|
||||||
`SELECT timestamp, provider, model, connectionId, apiKey, endpoint FROM usageHistory WHERE timestamp >= ?`,
|
`SELECT timestamp, provider, model, connectionId, apiKey, endpoint FROM usageHistory WHERE timestamp >= ?`,
|
||||||
[new Date(overlayCutoff).toISOString()]
|
[new Date(overlayCutoff).toISOString()]
|
||||||
|
|||||||
39
tests/unit/thinking-budget-max-level.test.js
Normal file
39
tests/unit/thinking-budget-max-level.test.js
Normal file
@@ -0,0 +1,39 @@
|
|||||||
|
import { describe, expect, it } from "vitest";
|
||||||
|
import { budgetToLevel } from "../../open-sse/translator/concerns/thinking.js";
|
||||||
|
import { applyThinking } from "../../open-sse/translator/concerns/thinkingUnified.js";
|
||||||
|
import { FORMATS } from "../../open-sse/translator/formats.js";
|
||||||
|
|
||||||
|
// Reverse map must be able to reach "max": LEVEL_TO_BUDGET.max = 128000 and
|
||||||
|
// xhigh = 32768, so the xhigh/max threshold is their midpoint (80384).
|
||||||
|
// Previously any budget > 28672 collapsed to "xhigh", making "max"
|
||||||
|
// unreachable from Claude Code budget_tokens — its default thinking budget
|
||||||
|
// (MAX_THINKING_TOKENS) could never produce effort "max".
|
||||||
|
describe("budgetToLevel reaches max tier", () => {
|
||||||
|
it("budget 98304 → \"max\"", () => {
|
||||||
|
expect(budgetToLevel(98304)).toBe("max");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("budget 128000 → \"max\"", () => {
|
||||||
|
expect(budgetToLevel(128000)).toBe("max");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("budget 80385 → \"max\" (just above midpoint)", () => {
|
||||||
|
expect(budgetToLevel(80385)).toBe("max");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("budget 80384 → \"xhigh\" (midpoint still xhigh)", () => {
|
||||||
|
expect(budgetToLevel(80384)).toBe("xhigh");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("budget 31999 stays \"xhigh\"", () => {
|
||||||
|
expect(budgetToLevel(31999)).toBe("xhigh");
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("applyThinking (openai-responses): large budgets map to max effort", () => {
|
||||||
|
it("budget 98304 → reasoning_effort \"max\" for gpt-5.6-sol (openai wire)", () => {
|
||||||
|
const body = { thinking: { type: "enabled", budget_tokens: 98304 } };
|
||||||
|
const out = applyThinking(FORMATS.OPENAI_RESPONSES, "gpt-5.6-sol", body, "codex");
|
||||||
|
expect(out?.reasoning_effort).toBe("max");
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user