fix(capabilities): publish real GPT-6/GPT-5.4+ context windows and combo token limits

This commit is contained in:
Amirsalar Sojoudi committed 2026-10-01 10:25:30 +07:00
1 parent 49c761cd67
commit 89ffac5a2c
5 files changed
+306 -12

No files matched your search

+40 -7
View File
@@ -6,15 +6,18 @@
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic // 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
// 4. DEFAULT_CAPABILITIES — safe floor (always returned) // 4. DEFAULT_CAPABILITIES — safe floor (always returned)
// //
// Two extra layers then refine the result, and neither can override the hand // Two extra layers then refine the result:
// written tables above (steps 1-2 short-circuit before they are consulted):
// • the synced catalog — modalities keyed by model, limits keyed by provider // • the synced catalog — modalities keyed by model, limits keyed by provider
// + model, refreshed from models.dev in the background. It reads a file, so // + model, refreshed from models.dev in the background. It reads a file, so
// the server installs it via setCatalogSource(); this module stays free of // the server installs it via setCatalogSource(); this module stays free of
// node:fs because the dashboard bundles it into the browser too. // node:fs because the dashboard bundles it into the browser too.
// • visionPatterns.js — name-based vision detection, last resort so a model // • visionPatterns.js — name-based vision detection, last resort so a model
// nobody has catalogued yet still accepts images. // nobody has catalogued yet still accepts images.
// Both only ever turn a capability ON. // Modalities only ever turn a capability ON. Limits from the catalog overlay
// the canonical exact entry (step 2) so a gateway-specific models.dev delta
// (Copilot's 32k Claude output, etc.) actually publishes. Step 1 still
// short-circuits: a hand-written PROVIDER_CAPABILITIES truncation is the
// gateway's own number and must not be overwritten.
// //
// ── HOW TO ADD / UPDATE A MODEL ────────────────────────────────────── // ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+ // Authoritative data source: https://models.dev/api.json (145 providers, 4000+
@@ -165,6 +168,11 @@ const CODEX_GPT_56_SOL_CAPS = { vision: true, reasoning: true, search: true, th
const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 }; const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 };
// Devin CLI's registry declares a 200k context window for these GPT variants.
// Keep the GPT feature/output fields because provider overrides short-circuit
// the generic pattern rather than merging with it.
const DEVIN_CLI_GPT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 };
/** /**
* Provider-specific capability overrides. Keyed by provider alias/id. * Provider-specific capability overrides. Keyed by provider alias/id.
*/ */
@@ -215,6 +223,15 @@ export const PROVIDER_CAPABILITIES = {
"gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES, "gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
"gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES, "gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
}, },
"devin-cli": {
"gpt-5.4-high": DEVIN_CLI_GPT_CAPS,
"gpt-5.4-medium": DEVIN_CLI_GPT_CAPS,
"gpt-5.4-low": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-xhigh": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-high": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-medium": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-low": DEVIN_CLI_GPT_CAPS,
},
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model // CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision= // config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort // supportsImages). Every model reasons via OpenAI-style reasoning_effort
@@ -278,6 +295,8 @@ export const PROVIDER_CAPABILITIES = {
// the intl Qoder capability table verbatim (vision/reasoning/contextWindow). // the intl Qoder capability table verbatim (vision/reasoning/contextWindow).
PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"]; PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"];
PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex; PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex;
PROVIDER_CAPABILITIES.dv = PROVIDER_CAPABILITIES["devin-cli"];
PROVIDER_CAPABILITIES.devin = PROVIDER_CAPABILITIES["devin-cli"];
/** /**
* Pattern fallback — glob (* = wildcard), matched case-insensitively and * Pattern fallback — glob (* = wildcard), matched case-insensitively and
@@ -315,11 +334,24 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } }, { pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
// ── OpenAI GPT-6.x (vision + thinking + web search) ────────────── // ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } }, // 1.05M is the API window for the whole gpt-6 family (astra, luna, sol alike).
// A gateway that truncates lower records its own number in
// PROVIDER_CAPABILITIES, which wins over this pattern — Kiro at 272k, Codex
// OAuth at 272k/372k (see CODEX_GPT_56_* above). This entry used to carry
// Kiro's 272k, so every other provider's gpt-6 models inherited one gateway's
// limit and were published at 3.9x under their real window.
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
// ── OpenAI GPT-5.x (vision + thinking + web search) ────────────── // ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } }, { pattern: "*gpt-5*image*", caps: { imageOutput: true } },
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, { pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
// gpt-5.4 is where the 1.05M window starts, but the mini and nano tiers stayed
// at 400k — first match wins, so those two have to be listed ahead of it.
{ pattern: "*gpt-5.4-mini*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
{ pattern: "*gpt-5.4-nano*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
{ pattern: "*gpt-5.4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
{ pattern: "*gpt-5.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
{ pattern: "*gpt-5.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
{ pattern: "*gpt-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, { pattern: "*gpt-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
{ pattern: "*gpt-4o*", caps: { vision: true, search: true, contextWindow: 128000, maxOutput: 16384 } }, { pattern: "*gpt-4o*", caps: { vision: true, search: true, contextWindow: 128000, maxOutput: 16384 } },
{ pattern: "*gpt-4.1*", caps: { vision: true, contextWindow: 1000000, maxOutput: 32768 } }, { pattern: "*gpt-4.1*", caps: { vision: true, contextWindow: 1000000, maxOutput: 32768 } },
@@ -618,9 +650,10 @@ export function getCapabilitiesForModel(provider, model) {
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] }; if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
} }
// 2. Canonical exact // 2. Canonical exact, then catalog overlay so provider-scoped models.dev
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] }; // deltas still apply. Step 1 above still short-circuits.
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] }; if (MODEL_CAPABILITIES[baseModel]) return refine(MODEL_CAPABILITIES[baseModel], provider, model);
if (MODEL_CAPABILITIES[model]) return refine(MODEL_CAPABILITIES[model], provider, model);
// 3. Pattern match (first match wins), refined by catalog + name heuristic // 3. Pattern match (first match wins), refined by catalog + name heuristic
for (const { pattern, caps } of PATTERN_CAPABILITIES) { for (const { pattern, caps } of PATTERN_CAPABILITIES) {
+70 -5
View File
@@ -1,5 +1,6 @@
import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS, getModelKind } from "@/shared/constants/models"; import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS, getModelKind } from "@/shared/constants/models";
import { import {
ALIAS_TO_ID,
AI_PROVIDERS, AI_PROVIDERS,
getProviderAlias, getProviderAlias,
isAnthropicCompatibleProvider, isAnthropicCompatibleProvider,
@@ -40,6 +41,22 @@ async function resolveQoderLiveModels(conn, provider) {
return { models: models.map((m) => ({ id: m.id, name: m.name })) }; return { models: models.map((m) => ({ id: m.id, name: m.name })) };
} }
// Combo seats use UI aliases; the model registry also has transport aliases.
// Capability overrides and catalog limits are keyed by provider id.
const ALIAS_TO_PROVIDER_ID = {
...Object.fromEntries(
Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id])
),
...ALIAS_TO_ID,
};
function comboSeatCapabilities(seat) {
const slash = seat.indexOf("/");
if (slash <= 0) return null;
const alias = seat.slice(0, slash);
return getCapabilitiesForModel(ALIAS_TO_PROVIDER_ID[alias] || alias, seat.slice(slash + 1));
}
// Per-provider live model resolvers. Each receives a connection record and // Per-provider live model resolvers. Each receives a connection record and
// returns { models: [{ id, name? }, ...] } | null on failure. // returns { models: [{ id, name? }, ...] } | null on failure.
// Adding a provider here makes /v1/models prefer the live catalog for it. // Adding a provider here makes /v1/models prefer the live catalog for it.
@@ -254,6 +271,47 @@ function comboMatchesKinds(combo, kindFilter) {
return kindFilter.includes(kind); return kindFilter.includes(kind);
} }
// Nested combo names are valid seats — the model selector exposes them and
// chat routing resolves them recursively — but a no-slash seat is otherwise
// treated as a literal model and publishes the 200k floor. Expand nested
// names (cycle-guarded) so the published window is the true min across the
// whole chain.
function comboSeatLimits(combo, combosByName, visiting = new Set()) {
const name = typeof combo?.name === "string" ? combo.name : null;
if (name) {
if (visiting.has(name)) return { contextWindow: undefined, maxOutput: undefined };
visiting.add(name);
}
let contextWindow = Infinity;
let maxOutput = Infinity;
try {
for (const seat of Array.isArray(combo?.models) ? combo.models : []) {
if (typeof seat !== "string") continue;
const slash = seat.indexOf("/");
if (slash <= 0) {
const nested = combosByName.get(seat);
if (nested) {
const nestedLimits = comboSeatLimits(nested, combosByName, visiting);
if (Number.isFinite(nestedLimits.contextWindow)) contextWindow = Math.min(contextWindow, nestedLimits.contextWindow);
if (Number.isFinite(nestedLimits.maxOutput)) maxOutput = Math.min(maxOutput, nestedLimits.maxOutput);
continue;
}
}
const caps = comboSeatCapabilities(seat) || getCapabilitiesForModel(null, seat);
if (Number.isFinite(caps?.contextWindow)) contextWindow = Math.min(contextWindow, caps.contextWindow);
if (Number.isFinite(caps?.maxOutput)) maxOutput = Math.min(maxOutput, caps.maxOutput);
}
} finally {
if (name) visiting.delete(name);
}
return {
contextWindow: Number.isFinite(contextWindow) ? contextWindow : undefined,
maxOutput: Number.isFinite(maxOutput) ? maxOutput : undefined,
};
}
/** /**
* Build OpenAI-format models list filtered by service kinds. * Build OpenAI-format models list filtered by service kinds.
* @param {string[]} kindFilter - List of service kinds to include (e.g. ["llm"], ["webSearch","webFetch"]). * @param {string[]} kindFilter - List of service kinds to include (e.g. ["llm"], ["webSearch","webFetch"]).
@@ -308,6 +366,9 @@ export async function buildModelsList(kindFilter, options = {}) {
} }
const models = []; const models = [];
const combosByName = new Map(
combos.filter((c) => typeof c?.name === "string").map((c) => [c.name, c]),
);
// Lookup map so aggregateComboCapabilities can recursively resolve nested combos // Lookup map so aggregateComboCapabilities can recursively resolve nested combos
const comboByName = Object.fromEntries(combos.map((c) => [c.name, c.models])); const comboByName = Object.fromEntries(combos.map((c) => [c.name, c.models]));
@@ -323,19 +384,23 @@ export async function buildModelsList(kindFilter, options = {}) {
if (combo.kind === "webSearch" || combo.kind === "webFetch") { if (combo.kind === "webSearch" || combo.kind === "webFetch") {
entry.kind = combo.kind; entry.kind = combo.kind;
} else { } else {
const comboCaps = aggregateComboCapabilities(combo.models, comboByName); const comboCaps = aggregateComboCapabilities(combo.models, comboByName, comboSeatCapabilities);
if (comboCaps) entry.capabilities = comboCaps; if (comboCaps) entry.capabilities = comboCaps;
// Any seat can serve the request, so the only window a combo can promise is
// its smallest. Combo entries were the only models on this endpoint that
// published no limits at all, which leaves a client to guess from the name —
// and it guesses high (see the snake_case note on the per-provider path).
const { contextWindow, maxOutput } = comboSeatLimits(combo, combosByName);
if (Number.isFinite(contextWindow)) entry.context_length = contextWindow;
if (Number.isFinite(maxOutput)) entry.max_completion_tokens = maxOutput;
} }
models.push(entry); models.push(entry);
} }
if (connections.length === 0) { if (connections.length === 0) {
// DB unavailable -> return static models, filtered by per-model kind // DB unavailable -> return static models, filtered by per-model kind
const aliasToProviderId = Object.fromEntries(
Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id])
);
for (const [alias, providerModels] of Object.entries(PROVIDER_MODELS)) { for (const [alias, providerModels] of Object.entries(PROVIDER_MODELS)) {
const providerId = aliasToProviderId[alias] || alias; const providerId = ALIAS_TO_PROVIDER_ID[alias] || alias;
if (!providerMatchesKinds(providerId, kindFilter)) continue; if (!providerMatchesKinds(providerId, kindFilter)) continue;
for (const model of providerModels) { for (const model of providerModels) {
if (!kindFilter.includes(modelKind(model))) continue; if (!kindFilter.includes(modelKind(model))) continue;
+1
View File
@@ -24,6 +24,7 @@ const LIMIT_TOLERANCE = 0.1;
// while building rather than on every lookup. Providers absent here keep whatever // while building rather than on every lookup. Providers absent here keep whatever
// the local pattern table resolves; names that already match need no entry. // the local pattern table resolves; names that already match need no entry.
export const PROVIDER_ALIASES = { export const PROVIDER_ALIASES = {
"github": "github-copilot",
"glm": "zai", "glm": "zai",
"glm-cn": "zhipuai", "glm-cn": "zhipuai",
"claude": "anthropic", "claude": "anthropic",
+112
View File
@@ -0,0 +1,112 @@
import { describe, expect, it } from "vitest";
import { getCapabilitiesForModel, setCatalogSource } from "../../open-sse/providers/capabilities.js";
import { PROVIDER_ALIASES, build } from "../../src/lib/modelCatalog/sync.js";
// The gpt-6 family's API window is 1.05M. The pattern table published 272,000 for
// it — Kiro's own truncation, copied into the global glob — so every other
// provider's gpt-6 models were advertised at 3.9x under their real window, and a
// client reading context_length compacted (or refused) far too early.
const API_WINDOW = 1050000;
const LEGACY_GPT5_WINDOW = 400000;
describe("gpt-6 / gpt-5.4+ context windows", () => {
it("reports the 1.05M API window for gpt-6 models on ordinary providers", () => {
for (const [provider, model] of [
["github", "gpt-6-luna"],
["azure", "gpt-6-luna"],
["openai", "gpt-6-luna"],
["github", "gpt-6-sol"],
["openai", "gpt-6-astra"],
]) {
expect(getCapabilitiesForModel(provider, model).contextWindow, `${provider}/${model}`).toBe(API_WINDOW);
}
});
// These two gateways really do truncate below the API, and their numbers live in
// PROVIDER_CAPABILITIES, which outranks the pattern. Correcting the pattern must
// leave them alone — that split is the whole point of the layering.
it("leaves the gateways that truncate lower on their own numbers", () => {
expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna").contextWindow).toBe(272000);
expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna-thinking-agentic").contextWindow).toBe(272000);
expect(getCapabilitiesForModel("codex", "gpt-5.6-luna").contextWindow).toBe(272000);
expect(getCapabilitiesForModel("codex", "gpt-5.6-sol").contextWindow).toBe(372000);
expect(getCapabilitiesForModel("codex", "gpt-6-astra").contextWindow).toBe(272000);
});
it("keeps Devin CLI's seven GPT-5.4/5.5 variants at the gateway's 200k limit", () => {
for (const model of [
"gpt-5.4-high", "gpt-5.4-medium", "gpt-5.4-low",
"gpt-5.5-xhigh", "gpt-5.5-high", "gpt-5.5-medium", "gpt-5.5-low",
]) {
expect(getCapabilitiesForModel("devin-cli", model), model).toMatchObject({
contextWindow: 200000,
maxOutput: 128000,
vision: true,
reasoning: true,
thinkingFormat: "openai",
});
}
expect(getCapabilitiesForModel("dv", "gpt-5.5-high").contextWindow).toBe(200000);
expect(getCapabilitiesForModel("devin", "gpt-5.5-high").contextWindow).toBe(200000);
});
// gpt-5.4 is where the 1.05M window starts and the mini/nano tiers are the
// exception that stayed at 400k. Pattern resolution is first-match-wins, so this
// is really a guard on the ORDER of the entries: move the tier patterns above
// the mini/nano ones and both tiers silently report 1.05M.
it("splits the 1.05M tiers from the 400k ones", () => {
for (const model of [
"gpt-5.4", "gpt-5.4-pro", "gpt-5.5", "gpt-5.5-pro", "gpt-5.6", "gpt-5.6-luna", "gpt-5.6-terra",
]) {
expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(API_WINDOW);
}
for (const model of ["gpt-5.4-mini", "gpt-5.4-nano", "gpt-5", "gpt-5.1", "gpt-5.2", "gpt-5.3-codex"]) {
expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(LEGACY_GPT5_WINDOW);
}
});
});
// Copilot is "github" locally and "github-copilot" upstream. Without that mapping
// build() resolved no upstream provider for it and skipped every Copilot model, so
// the daily models.dev sync could never correct a stale hand-written number — which
// is how the gpt-6 window stayed 3.9x wrong without anything noticing.
describe("models.dev sync reaches GitHub Copilot", () => {
it("maps the local github id onto the upstream github-copilot id", () => {
expect(PROVIDER_ALIASES.github).toBe("github-copilot");
});
it("records a Copilot limit that disagrees with the local tables", () => {
// Copilot caps Claude output at 32k where the local floor assumes 64k.
const upstream = {
"github-copilot": {
models: { "claude-sonnet-4.6": { limit: { context: 200000, output: 32000 } } },
},
};
const entries = [
{ provider: "github", model: "claude-sonnet-4.6", current: { contextWindow: 200000, maxOutput: 64000 } },
];
// Context agrees, so only the output delta is recorded — and it is filed under
// the local id, which is what the reader looks up.
expect(build(upstream, entries).providers.github).toEqual({
"claude-sonnet-4.6": { maxOutput: 32000 },
});
});
it("applies those Copilot deltas to an exact MODEL_CAPABILITIES id", () => {
// claude-sonnet-4.6 is canonical-exact (128k). Without refine() on that
// path the 32k Copilot delta from build() would never be read.
setCatalogSource({
getModalities: () => null,
getLimits: (provider, model) =>
provider === "github" && model === "claude-sonnet-4.6" ? { maxOutput: 32000 } : null,
});
try {
expect(getCapabilitiesForModel("github", "claude-sonnet-4.6").maxOutput).toBe(32000);
expect(getCapabilitiesForModel("claude", "claude-sonnet-4.6").maxOutput).toBe(128000);
} finally {
setCatalogSource(null);
}
});
});
@@ -0,0 +1,83 @@
import { describe, expect, it, vi } from "vitest";
import { setCatalogSource } from "../../open-sse/providers/capabilities.js";
const db = vi.hoisted(() => ({
getProviderConnections: vi.fn(),
getCombos: vi.fn(),
getCustomModels: vi.fn(async () => []),
getModelAliases: vi.fn(async () => ({})),
}));
vi.mock("@/lib/localDb", () => db);
vi.mock("@/lib/disabledModelsDb", () => ({
getDisabledModels: vi.fn(async () => ({})),
}));
const { buildModelsList } = await import("../../src/app/api/v1/models/route.js");
const syncedLimits = { contextWindow: 180000, maxOutput: 16000 };
async function modelsWithCombo(providerId, modelId, combos) {
db.getProviderConnections.mockResolvedValue([{
id: 1,
provider: providerId,
isActive: true,
providerSpecificData: { enabledModels: [modelId] },
}]);
db.getCombos.mockResolvedValue(combos);
setCatalogSource({
getModalities: () => null,
getLimits: (provider, model) =>
provider === providerId && model === modelId ? syncedLimits : null,
});
try {
return await buildModelsList(["llm"]);
} finally {
setCatalogSource(null);
}
}
describe("/v1/models combo limits", () => {
it.each([
["ocg", "opencode-go", "mimo-v2.5"],
["xmtp", "xiaomi-tokenplan", "mimo-v2.5"],
["ps", "poolside", "custom-model"],
["ds", "deepseek", "deepseek-chat"],
])("uses the real provider for a %s UI-alias seat", async (uiAlias, providerId, modelId) => {
const combo = { name: "ui-alias-combo", models: [`${uiAlias}/${modelId}`] };
const models = await modelsWithCombo(providerId, modelId, [combo]);
const published = models.find((model) => model.id === combo.name);
expect(published).toMatchObject({
context_length: 180000,
max_completion_tokens: 16000,
capabilities: { contextWindow: 180000, maxOutput: 16000 },
});
});
it("carries provider-scoped limits through a nested combo", async () => {
const models = await modelsWithCombo("opencode-go", "mimo-v2.5", [
{ name: "inner-combo", models: ["ocg/mimo-v2.5"] },
{ name: "outer-combo", models: ["inner-combo"] },
]);
const outer = models.find((model) => model.id === "outer-combo");
expect(outer).toMatchObject({
context_length: 180000,
max_completion_tokens: 16000,
capabilities: { contextWindow: 180000, maxOutput: 16000 },
});
});
it("publishes Devin CLI's 200k limit for a dv combo seat", async () => {
const models = await modelsWithCombo("devin-cli", "gpt-5.5-high", [
{ name: "devin-combo", models: ["dv/gpt-5.5-high"] },
]);
const combo = models.find((model) => model.id === "devin-combo");
expect(combo).toMatchObject({
context_length: 200000,
capabilities: { contextWindow: 200000 },
});
});
});