diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 68e4fa0a..833635dc 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -6,15 +6,18 @@ // 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic // 4. DEFAULT_CAPABILITIES — safe floor (always returned) // -// Two extra layers then refine the result, and neither can override the hand -// written tables above (steps 1-2 short-circuit before they are consulted): +// Two extra layers then refine the result: // • the synced catalog — modalities keyed by model, limits keyed by provider // + model, refreshed from models.dev in the background. It reads a file, so // the server installs it via setCatalogSource(); this module stays free of // node:fs because the dashboard bundles it into the browser too. // • visionPatterns.js — name-based vision detection, last resort so a model // nobody has catalogued yet still accepts images. -// Both only ever turn a capability ON. +// Modalities only ever turn a capability ON. Limits from the catalog overlay +// the canonical exact entry (step 2) so a gateway-specific models.dev delta +// (Copilot's 32k Claude output, etc.) actually publishes. Step 1 still +// short-circuits: a hand-written PROVIDER_CAPABILITIES truncation is the +// gateway's own number and must not be overwritten. // // ── HOW TO ADD / UPDATE A MODEL ────────────────────────────────────── // Authoritative data source: https://models.dev/api.json (145 providers, 4000+ @@ -165,6 +168,11 @@ const CODEX_GPT_56_SOL_CAPS = { vision: true, reasoning: true, search: true, th const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 }; +// Devin CLI's registry declares a 200k context window for these GPT variants. +// Keep the GPT feature/output fields because provider overrides short-circuit +// the generic pattern rather than merging with it. +const DEVIN_CLI_GPT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 }; + /** * Provider-specific capability overrides. Keyed by provider alias/id. */ @@ -215,6 +223,15 @@ export const PROVIDER_CAPABILITIES = { "gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES, "gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES, }, + "devin-cli": { + "gpt-5.4-high": DEVIN_CLI_GPT_CAPS, + "gpt-5.4-medium": DEVIN_CLI_GPT_CAPS, + "gpt-5.4-low": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-xhigh": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-high": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-medium": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-low": DEVIN_CLI_GPT_CAPS, + }, // CodeBuddy.cn — authoritative per-model metadata from the gateway's model // config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision= // supportsImages). Every model reasons via OpenAI-style reasoning_effort @@ -278,6 +295,8 @@ export const PROVIDER_CAPABILITIES = { // the intl Qoder capability table verbatim (vision/reasoning/contextWindow). PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"]; PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex; +PROVIDER_CAPABILITIES.dv = PROVIDER_CAPABILITIES["devin-cli"]; +PROVIDER_CAPABILITIES.devin = PROVIDER_CAPABILITIES["devin-cli"]; /** * Pattern fallback — glob (* = wildcard), matched case-insensitively and @@ -315,11 +334,24 @@ export const PATTERN_CAPABILITIES = [ { pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } }, // ── OpenAI GPT-6.x (vision + thinking + web search) ────────────── - { pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } }, + // 1.05M is the API window for the whole gpt-6 family (astra, luna, sol alike). + // A gateway that truncates lower records its own number in + // PROVIDER_CAPABILITIES, which wins over this pattern — Kiro at 272k, Codex + // OAuth at 272k/372k (see CODEX_GPT_56_* above). This entry used to carry + // Kiro's 272k, so every other provider's gpt-6 models inherited one gateway's + // limit and were published at 3.9x under their real window. + { pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, // ── OpenAI GPT-5.x (vision + thinking + web search) ────────────── { pattern: "*gpt-5*image*", caps: { imageOutput: true } }, { pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, + // gpt-5.4 is where the 1.05M window starts, but the mini and nano tiers stayed + // at 400k — first match wins, so those two have to be listed ahead of it. + { pattern: "*gpt-5.4-mini*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, + { pattern: "*gpt-5.4-nano*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, + { pattern: "*gpt-5.4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, + { pattern: "*gpt-5.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, + { pattern: "*gpt-5.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, { pattern: "*gpt-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, { pattern: "*gpt-4o*", caps: { vision: true, search: true, contextWindow: 128000, maxOutput: 16384 } }, { pattern: "*gpt-4.1*", caps: { vision: true, contextWindow: 1000000, maxOutput: 32768 } }, @@ -618,9 +650,10 @@ export function getCapabilitiesForModel(provider, model) { if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] }; } - // 2. Canonical exact - if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] }; - if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] }; + // 2. Canonical exact, then catalog overlay so provider-scoped models.dev + // deltas still apply. Step 1 above still short-circuits. + if (MODEL_CAPABILITIES[baseModel]) return refine(MODEL_CAPABILITIES[baseModel], provider, model); + if (MODEL_CAPABILITIES[model]) return refine(MODEL_CAPABILITIES[model], provider, model); // 3. Pattern match (first match wins), refined by catalog + name heuristic for (const { pattern, caps } of PATTERN_CAPABILITIES) { diff --git a/src/app/api/v1/models/route.js b/src/app/api/v1/models/route.js index 389f141b..0b1fe342 100644 --- a/src/app/api/v1/models/route.js +++ b/src/app/api/v1/models/route.js @@ -1,5 +1,6 @@ import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS, getModelKind } from "@/shared/constants/models"; import { + ALIAS_TO_ID, AI_PROVIDERS, getProviderAlias, isAnthropicCompatibleProvider, @@ -40,6 +41,22 @@ async function resolveQoderLiveModels(conn, provider) { return { models: models.map((m) => ({ id: m.id, name: m.name })) }; } +// Combo seats use UI aliases; the model registry also has transport aliases. +// Capability overrides and catalog limits are keyed by provider id. +const ALIAS_TO_PROVIDER_ID = { + ...Object.fromEntries( + Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id]) + ), + ...ALIAS_TO_ID, +}; + +function comboSeatCapabilities(seat) { + const slash = seat.indexOf("/"); + if (slash <= 0) return null; + const alias = seat.slice(0, slash); + return getCapabilitiesForModel(ALIAS_TO_PROVIDER_ID[alias] || alias, seat.slice(slash + 1)); +} + // Per-provider live model resolvers. Each receives a connection record and // returns { models: [{ id, name? }, ...] } | null on failure. // Adding a provider here makes /v1/models prefer the live catalog for it. @@ -254,6 +271,47 @@ function comboMatchesKinds(combo, kindFilter) { return kindFilter.includes(kind); } +// Nested combo names are valid seats — the model selector exposes them and +// chat routing resolves them recursively — but a no-slash seat is otherwise +// treated as a literal model and publishes the 200k floor. Expand nested +// names (cycle-guarded) so the published window is the true min across the +// whole chain. +function comboSeatLimits(combo, combosByName, visiting = new Set()) { + const name = typeof combo?.name === "string" ? combo.name : null; + if (name) { + if (visiting.has(name)) return { contextWindow: undefined, maxOutput: undefined }; + visiting.add(name); + } + + let contextWindow = Infinity; + let maxOutput = Infinity; + try { + for (const seat of Array.isArray(combo?.models) ? combo.models : []) { + if (typeof seat !== "string") continue; + const slash = seat.indexOf("/"); + if (slash <= 0) { + const nested = combosByName.get(seat); + if (nested) { + const nestedLimits = comboSeatLimits(nested, combosByName, visiting); + if (Number.isFinite(nestedLimits.contextWindow)) contextWindow = Math.min(contextWindow, nestedLimits.contextWindow); + if (Number.isFinite(nestedLimits.maxOutput)) maxOutput = Math.min(maxOutput, nestedLimits.maxOutput); + continue; + } + } + const caps = comboSeatCapabilities(seat) || getCapabilitiesForModel(null, seat); + if (Number.isFinite(caps?.contextWindow)) contextWindow = Math.min(contextWindow, caps.contextWindow); + if (Number.isFinite(caps?.maxOutput)) maxOutput = Math.min(maxOutput, caps.maxOutput); + } + } finally { + if (name) visiting.delete(name); + } + + return { + contextWindow: Number.isFinite(contextWindow) ? contextWindow : undefined, + maxOutput: Number.isFinite(maxOutput) ? maxOutput : undefined, + }; +} + /** * Build OpenAI-format models list filtered by service kinds. * @param {string[]} kindFilter - List of service kinds to include (e.g. ["llm"], ["webSearch","webFetch"]). @@ -308,6 +366,9 @@ export async function buildModelsList(kindFilter, options = {}) { } const models = []; + const combosByName = new Map( + combos.filter((c) => typeof c?.name === "string").map((c) => [c.name, c]), + ); // Lookup map so aggregateComboCapabilities can recursively resolve nested combos const comboByName = Object.fromEntries(combos.map((c) => [c.name, c.models])); @@ -323,19 +384,23 @@ export async function buildModelsList(kindFilter, options = {}) { if (combo.kind === "webSearch" || combo.kind === "webFetch") { entry.kind = combo.kind; } else { - const comboCaps = aggregateComboCapabilities(combo.models, comboByName); + const comboCaps = aggregateComboCapabilities(combo.models, comboByName, comboSeatCapabilities); if (comboCaps) entry.capabilities = comboCaps; + // Any seat can serve the request, so the only window a combo can promise is + // its smallest. Combo entries were the only models on this endpoint that + // published no limits at all, which leaves a client to guess from the name — + // and it guesses high (see the snake_case note on the per-provider path). + const { contextWindow, maxOutput } = comboSeatLimits(combo, combosByName); + if (Number.isFinite(contextWindow)) entry.context_length = contextWindow; + if (Number.isFinite(maxOutput)) entry.max_completion_tokens = maxOutput; } models.push(entry); } if (connections.length === 0) { // DB unavailable -> return static models, filtered by per-model kind - const aliasToProviderId = Object.fromEntries( - Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id]) - ); for (const [alias, providerModels] of Object.entries(PROVIDER_MODELS)) { - const providerId = aliasToProviderId[alias] || alias; + const providerId = ALIAS_TO_PROVIDER_ID[alias] || alias; if (!providerMatchesKinds(providerId, kindFilter)) continue; for (const model of providerModels) { if (!kindFilter.includes(modelKind(model))) continue; diff --git a/src/lib/modelCatalog/sync.js b/src/lib/modelCatalog/sync.js index 973c18b3..54f72d5a 100644 --- a/src/lib/modelCatalog/sync.js +++ b/src/lib/modelCatalog/sync.js @@ -24,6 +24,7 @@ const LIMIT_TOLERANCE = 0.1; // while building rather than on every lookup. Providers absent here keep whatever // the local pattern table resolves; names that already match need no entry. export const PROVIDER_ALIASES = { + "github": "github-copilot", "glm": "zai", "glm-cn": "zhipuai", "claude": "anthropic", diff --git a/tests/unit/gpt-6-context-window.test.js b/tests/unit/gpt-6-context-window.test.js new file mode 100644 index 00000000..e3f394cd --- /dev/null +++ b/tests/unit/gpt-6-context-window.test.js @@ -0,0 +1,112 @@ +import { describe, expect, it } from "vitest"; + +import { getCapabilitiesForModel, setCatalogSource } from "../../open-sse/providers/capabilities.js"; +import { PROVIDER_ALIASES, build } from "../../src/lib/modelCatalog/sync.js"; + +// The gpt-6 family's API window is 1.05M. The pattern table published 272,000 for +// it — Kiro's own truncation, copied into the global glob — so every other +// provider's gpt-6 models were advertised at 3.9x under their real window, and a +// client reading context_length compacted (or refused) far too early. +const API_WINDOW = 1050000; +const LEGACY_GPT5_WINDOW = 400000; + +describe("gpt-6 / gpt-5.4+ context windows", () => { + it("reports the 1.05M API window for gpt-6 models on ordinary providers", () => { + for (const [provider, model] of [ + ["github", "gpt-6-luna"], + ["azure", "gpt-6-luna"], + ["openai", "gpt-6-luna"], + ["github", "gpt-6-sol"], + ["openai", "gpt-6-astra"], + ]) { + expect(getCapabilitiesForModel(provider, model).contextWindow, `${provider}/${model}`).toBe(API_WINDOW); + } + }); + + // These two gateways really do truncate below the API, and their numbers live in + // PROVIDER_CAPABILITIES, which outranks the pattern. Correcting the pattern must + // leave them alone — that split is the whole point of the layering. + it("leaves the gateways that truncate lower on their own numbers", () => { + expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna").contextWindow).toBe(272000); + expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna-thinking-agentic").contextWindow).toBe(272000); + expect(getCapabilitiesForModel("codex", "gpt-5.6-luna").contextWindow).toBe(272000); + expect(getCapabilitiesForModel("codex", "gpt-5.6-sol").contextWindow).toBe(372000); + expect(getCapabilitiesForModel("codex", "gpt-6-astra").contextWindow).toBe(272000); + }); + + it("keeps Devin CLI's seven GPT-5.4/5.5 variants at the gateway's 200k limit", () => { + for (const model of [ + "gpt-5.4-high", "gpt-5.4-medium", "gpt-5.4-low", + "gpt-5.5-xhigh", "gpt-5.5-high", "gpt-5.5-medium", "gpt-5.5-low", + ]) { + expect(getCapabilitiesForModel("devin-cli", model), model).toMatchObject({ + contextWindow: 200000, + maxOutput: 128000, + vision: true, + reasoning: true, + thinkingFormat: "openai", + }); + } + expect(getCapabilitiesForModel("dv", "gpt-5.5-high").contextWindow).toBe(200000); + expect(getCapabilitiesForModel("devin", "gpt-5.5-high").contextWindow).toBe(200000); + }); + + // gpt-5.4 is where the 1.05M window starts and the mini/nano tiers are the + // exception that stayed at 400k. Pattern resolution is first-match-wins, so this + // is really a guard on the ORDER of the entries: move the tier patterns above + // the mini/nano ones and both tiers silently report 1.05M. + it("splits the 1.05M tiers from the 400k ones", () => { + for (const model of [ + "gpt-5.4", "gpt-5.4-pro", "gpt-5.5", "gpt-5.5-pro", "gpt-5.6", "gpt-5.6-luna", "gpt-5.6-terra", + ]) { + expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(API_WINDOW); + } + for (const model of ["gpt-5.4-mini", "gpt-5.4-nano", "gpt-5", "gpt-5.1", "gpt-5.2", "gpt-5.3-codex"]) { + expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(LEGACY_GPT5_WINDOW); + } + }); +}); + +// Copilot is "github" locally and "github-copilot" upstream. Without that mapping +// build() resolved no upstream provider for it and skipped every Copilot model, so +// the daily models.dev sync could never correct a stale hand-written number — which +// is how the gpt-6 window stayed 3.9x wrong without anything noticing. +describe("models.dev sync reaches GitHub Copilot", () => { + it("maps the local github id onto the upstream github-copilot id", () => { + expect(PROVIDER_ALIASES.github).toBe("github-copilot"); + }); + + it("records a Copilot limit that disagrees with the local tables", () => { + // Copilot caps Claude output at 32k where the local floor assumes 64k. + const upstream = { + "github-copilot": { + models: { "claude-sonnet-4.6": { limit: { context: 200000, output: 32000 } } }, + }, + }; + const entries = [ + { provider: "github", model: "claude-sonnet-4.6", current: { contextWindow: 200000, maxOutput: 64000 } }, + ]; + + // Context agrees, so only the output delta is recorded — and it is filed under + // the local id, which is what the reader looks up. + expect(build(upstream, entries).providers.github).toEqual({ + "claude-sonnet-4.6": { maxOutput: 32000 }, + }); + }); + + it("applies those Copilot deltas to an exact MODEL_CAPABILITIES id", () => { + // claude-sonnet-4.6 is canonical-exact (128k). Without refine() on that + // path the 32k Copilot delta from build() would never be read. + setCatalogSource({ + getModalities: () => null, + getLimits: (provider, model) => + provider === "github" && model === "claude-sonnet-4.6" ? { maxOutput: 32000 } : null, + }); + try { + expect(getCapabilitiesForModel("github", "claude-sonnet-4.6").maxOutput).toBe(32000); + expect(getCapabilitiesForModel("claude", "claude-sonnet-4.6").maxOutput).toBe(128000); + } finally { + setCatalogSource(null); + } + }); +}); diff --git a/tests/unit/v1-models-combo-context.test.js b/tests/unit/v1-models-combo-context.test.js new file mode 100644 index 00000000..9f52bed2 --- /dev/null +++ b/tests/unit/v1-models-combo-context.test.js @@ -0,0 +1,83 @@ +import { describe, expect, it, vi } from "vitest"; +import { setCatalogSource } from "../../open-sse/providers/capabilities.js"; + +const db = vi.hoisted(() => ({ + getProviderConnections: vi.fn(), + getCombos: vi.fn(), + getCustomModels: vi.fn(async () => []), + getModelAliases: vi.fn(async () => ({})), +})); + +vi.mock("@/lib/localDb", () => db); +vi.mock("@/lib/disabledModelsDb", () => ({ + getDisabledModels: vi.fn(async () => ({})), +})); + +const { buildModelsList } = await import("../../src/app/api/v1/models/route.js"); + +const syncedLimits = { contextWindow: 180000, maxOutput: 16000 }; + +async function modelsWithCombo(providerId, modelId, combos) { + db.getProviderConnections.mockResolvedValue([{ + id: 1, + provider: providerId, + isActive: true, + providerSpecificData: { enabledModels: [modelId] }, + }]); + db.getCombos.mockResolvedValue(combos); + setCatalogSource({ + getModalities: () => null, + getLimits: (provider, model) => + provider === providerId && model === modelId ? syncedLimits : null, + }); + try { + return await buildModelsList(["llm"]); + } finally { + setCatalogSource(null); + } +} + +describe("/v1/models combo limits", () => { + it.each([ + ["ocg", "opencode-go", "mimo-v2.5"], + ["xmtp", "xiaomi-tokenplan", "mimo-v2.5"], + ["ps", "poolside", "custom-model"], + ["ds", "deepseek", "deepseek-chat"], + ])("uses the real provider for a %s UI-alias seat", async (uiAlias, providerId, modelId) => { + const combo = { name: "ui-alias-combo", models: [`${uiAlias}/${modelId}`] }; + const models = await modelsWithCombo(providerId, modelId, [combo]); + const published = models.find((model) => model.id === combo.name); + + expect(published).toMatchObject({ + context_length: 180000, + max_completion_tokens: 16000, + capabilities: { contextWindow: 180000, maxOutput: 16000 }, + }); + }); + + it("carries provider-scoped limits through a nested combo", async () => { + const models = await modelsWithCombo("opencode-go", "mimo-v2.5", [ + { name: "inner-combo", models: ["ocg/mimo-v2.5"] }, + { name: "outer-combo", models: ["inner-combo"] }, + ]); + const outer = models.find((model) => model.id === "outer-combo"); + + expect(outer).toMatchObject({ + context_length: 180000, + max_completion_tokens: 16000, + capabilities: { contextWindow: 180000, maxOutput: 16000 }, + }); + }); + + it("publishes Devin CLI's 200k limit for a dv combo seat", async () => { + const models = await modelsWithCombo("devin-cli", "gpt-5.5-high", [ + { name: "devin-combo", models: ["dv/gpt-5.5-high"] }, + ]); + const combo = models.find((model) => model.id === "devin-combo"); + + expect(combo).toMatchObject({ + context_length: 200000, + capabilities: { contextWindow: 200000 }, + }); + }); +});