feat(headroom): make the compression request timeout configurable
The 3000 ms timeout on /v1/compress was fixed, so busy or slow machines timed out often and sent the LLM an inconsistently compressed body, hurting prompt caching. Add a headroomTimeoutMs setting, thread it from the chat handler down to compressWithHeadroom, expose it in the Token Saver dashboard, and normalize invalid values back to the 3000 ms default.
This commit is contained in:
@@ -7,6 +7,12 @@ import {
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 3000;
|
||||
|
||||
function normalizeTimeout(value) {
|
||||
return typeof value === "number" && Number.isFinite(value) && value > 0
|
||||
? value
|
||||
: DEFAULT_TIMEOUT_MS;
|
||||
}
|
||||
|
||||
function jsonBytes(value) {
|
||||
try {
|
||||
return new TextEncoder().encode(JSON.stringify(value) || "").length;
|
||||
@@ -240,6 +246,7 @@ async function callCompress(url, messages, model, timeoutMs, compressUserMessage
|
||||
// /v1/compress only understands OpenAI shape, so Claude bodies are translated
|
||||
// to OpenAI, compressed, then translated back using 9Router's own translators.
|
||||
export async function compressWithHeadroom(body, { enabled, url, model, format, compressUserMessages, timeoutMs = DEFAULT_TIMEOUT_MS, diagnostics = null } = {}) {
|
||||
timeoutMs = normalizeTimeout(timeoutMs);
|
||||
if (!enabled) {
|
||||
setDiagnostic(diagnostics, "disabled");
|
||||
return null;
|
||||
|
||||
Reference in New Issue
Block a user