fix(capabilities): publish real GPT-6/GPT-5.4+ context windows and combo token limits
This commit is contained in:
1 parent
49c761cd67
commit
89ffac5a2c
5 files changed
+306
-12
No files matched your search
@@ -6,15 +6,18 @@
|
||||
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
|
||||
// 4. DEFAULT_CAPABILITIES — safe floor (always returned)
|
||||
//
|
||||
// Two extra layers then refine the result, and neither can override the hand
|
||||
// written tables above (steps 1-2 short-circuit before they are consulted):
|
||||
// Two extra layers then refine the result:
|
||||
// • the synced catalog — modalities keyed by model, limits keyed by provider
|
||||
// + model, refreshed from models.dev in the background. It reads a file, so
|
||||
// the server installs it via setCatalogSource(); this module stays free of
|
||||
// node:fs because the dashboard bundles it into the browser too.
|
||||
// • visionPatterns.js — name-based vision detection, last resort so a model
|
||||
// nobody has catalogued yet still accepts images.
|
||||
// Both only ever turn a capability ON.
|
||||
// Modalities only ever turn a capability ON. Limits from the catalog overlay
|
||||
// the canonical exact entry (step 2) so a gateway-specific models.dev delta
|
||||
// (Copilot's 32k Claude output, etc.) actually publishes. Step 1 still
|
||||
// short-circuits: a hand-written PROVIDER_CAPABILITIES truncation is the
|
||||
// gateway's own number and must not be overwritten.
|
||||
//
|
||||
// ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
|
||||
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+
|
||||
@@ -165,6 +168,11 @@ const CODEX_GPT_56_SOL_CAPS = { vision: true, reasoning: true, search: true, th
|
||||
const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 };
|
||||
|
||||
// Devin CLI's registry declares a 200k context window for these GPT variants.
|
||||
// Keep the GPT feature/output fields because provider overrides short-circuit
|
||||
// the generic pattern rather than merging with it.
|
||||
const DEVIN_CLI_GPT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 };
|
||||
|
||||
/**
|
||||
* Provider-specific capability overrides. Keyed by provider alias/id.
|
||||
*/
|
||||
@@ -215,6 +223,15 @@ export const PROVIDER_CAPABILITIES = {
|
||||
"gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
},
|
||||
"devin-cli": {
|
||||
"gpt-5.4-high": DEVIN_CLI_GPT_CAPS,
|
||||
"gpt-5.4-medium": DEVIN_CLI_GPT_CAPS,
|
||||
"gpt-5.4-low": DEVIN_CLI_GPT_CAPS,
|
||||
"gpt-5.5-xhigh": DEVIN_CLI_GPT_CAPS,
|
||||
"gpt-5.5-high": DEVIN_CLI_GPT_CAPS,
|
||||
"gpt-5.5-medium": DEVIN_CLI_GPT_CAPS,
|
||||
"gpt-5.5-low": DEVIN_CLI_GPT_CAPS,
|
||||
},
|
||||
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
|
||||
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
|
||||
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
|
||||
@@ -278,6 +295,8 @@ export const PROVIDER_CAPABILITIES = {
|
||||
// the intl Qoder capability table verbatim (vision/reasoning/contextWindow).
|
||||
PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"];
|
||||
PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex;
|
||||
PROVIDER_CAPABILITIES.dv = PROVIDER_CAPABILITIES["devin-cli"];
|
||||
PROVIDER_CAPABILITIES.devin = PROVIDER_CAPABILITIES["devin-cli"];
|
||||
|
||||
/**
|
||||
* Pattern fallback — glob (* = wildcard), matched case-insensitively and
|
||||
@@ -315,11 +334,24 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
|
||||
|
||||
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
|
||||
// 1.05M is the API window for the whole gpt-6 family (astra, luna, sol alike).
|
||||
// A gateway that truncates lower records its own number in
|
||||
// PROVIDER_CAPABILITIES, which wins over this pattern — Kiro at 272k, Codex
|
||||
// OAuth at 272k/372k (see CODEX_GPT_56_* above). This entry used to carry
|
||||
// Kiro's 272k, so every other provider's gpt-6 models inherited one gateway's
|
||||
// limit and were published at 3.9x under their real window.
|
||||
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
|
||||
|
||||
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
// gpt-5.4 is where the 1.05M window starts, but the mini and nano tiers stayed
|
||||
// at 400k — first match wins, so those two have to be listed ahead of it.
|
||||
{ pattern: "*gpt-5.4-mini*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
{ pattern: "*gpt-5.4-nano*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
{ pattern: "*gpt-5.4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
|
||||
{ pattern: "*gpt-5.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
|
||||
{ pattern: "*gpt-5.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
|
||||
{ pattern: "*gpt-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
{ pattern: "*gpt-4o*", caps: { vision: true, search: true, contextWindow: 128000, maxOutput: 16384 } },
|
||||
{ pattern: "*gpt-4.1*", caps: { vision: true, contextWindow: 1000000, maxOutput: 32768 } },
|
||||
@@ -618,9 +650,10 @@ export function getCapabilitiesForModel(provider, model) {
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
}
|
||||
|
||||
// 2. Canonical exact
|
||||
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
|
||||
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
|
||||
// 2. Canonical exact, then catalog overlay so provider-scoped models.dev
|
||||
// deltas still apply. Step 1 above still short-circuits.
|
||||
if (MODEL_CAPABILITIES[baseModel]) return refine(MODEL_CAPABILITIES[baseModel], provider, model);
|
||||
if (MODEL_CAPABILITIES[model]) return refine(MODEL_CAPABILITIES[model], provider, model);
|
||||
|
||||
// 3. Pattern match (first match wins), refined by catalog + name heuristic
|
||||
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS, getModelKind } from "@/shared/constants/models";
|
||||
import {
|
||||
ALIAS_TO_ID,
|
||||
AI_PROVIDERS,
|
||||
getProviderAlias,
|
||||
isAnthropicCompatibleProvider,
|
||||
@@ -40,6 +41,22 @@ async function resolveQoderLiveModels(conn, provider) {
|
||||
return { models: models.map((m) => ({ id: m.id, name: m.name })) };
|
||||
}
|
||||
|
||||
// Combo seats use UI aliases; the model registry also has transport aliases.
|
||||
// Capability overrides and catalog limits are keyed by provider id.
|
||||
const ALIAS_TO_PROVIDER_ID = {
|
||||
...Object.fromEntries(
|
||||
Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id])
|
||||
),
|
||||
...ALIAS_TO_ID,
|
||||
};
|
||||
|
||||
function comboSeatCapabilities(seat) {
|
||||
const slash = seat.indexOf("/");
|
||||
if (slash <= 0) return null;
|
||||
const alias = seat.slice(0, slash);
|
||||
return getCapabilitiesForModel(ALIAS_TO_PROVIDER_ID[alias] || alias, seat.slice(slash + 1));
|
||||
}
|
||||
|
||||
// Per-provider live model resolvers. Each receives a connection record and
|
||||
// returns { models: [{ id, name? }, ...] } | null on failure.
|
||||
// Adding a provider here makes /v1/models prefer the live catalog for it.
|
||||
@@ -254,6 +271,47 @@ function comboMatchesKinds(combo, kindFilter) {
|
||||
return kindFilter.includes(kind);
|
||||
}
|
||||
|
||||
// Nested combo names are valid seats — the model selector exposes them and
|
||||
// chat routing resolves them recursively — but a no-slash seat is otherwise
|
||||
// treated as a literal model and publishes the 200k floor. Expand nested
|
||||
// names (cycle-guarded) so the published window is the true min across the
|
||||
// whole chain.
|
||||
function comboSeatLimits(combo, combosByName, visiting = new Set()) {
|
||||
const name = typeof combo?.name === "string" ? combo.name : null;
|
||||
if (name) {
|
||||
if (visiting.has(name)) return { contextWindow: undefined, maxOutput: undefined };
|
||||
visiting.add(name);
|
||||
}
|
||||
|
||||
let contextWindow = Infinity;
|
||||
let maxOutput = Infinity;
|
||||
try {
|
||||
for (const seat of Array.isArray(combo?.models) ? combo.models : []) {
|
||||
if (typeof seat !== "string") continue;
|
||||
const slash = seat.indexOf("/");
|
||||
if (slash <= 0) {
|
||||
const nested = combosByName.get(seat);
|
||||
if (nested) {
|
||||
const nestedLimits = comboSeatLimits(nested, combosByName, visiting);
|
||||
if (Number.isFinite(nestedLimits.contextWindow)) contextWindow = Math.min(contextWindow, nestedLimits.contextWindow);
|
||||
if (Number.isFinite(nestedLimits.maxOutput)) maxOutput = Math.min(maxOutput, nestedLimits.maxOutput);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
const caps = comboSeatCapabilities(seat) || getCapabilitiesForModel(null, seat);
|
||||
if (Number.isFinite(caps?.contextWindow)) contextWindow = Math.min(contextWindow, caps.contextWindow);
|
||||
if (Number.isFinite(caps?.maxOutput)) maxOutput = Math.min(maxOutput, caps.maxOutput);
|
||||
}
|
||||
} finally {
|
||||
if (name) visiting.delete(name);
|
||||
}
|
||||
|
||||
return {
|
||||
contextWindow: Number.isFinite(contextWindow) ? contextWindow : undefined,
|
||||
maxOutput: Number.isFinite(maxOutput) ? maxOutput : undefined,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Build OpenAI-format models list filtered by service kinds.
|
||||
* @param {string[]} kindFilter - List of service kinds to include (e.g. ["llm"], ["webSearch","webFetch"]).
|
||||
@@ -308,6 +366,9 @@ export async function buildModelsList(kindFilter, options = {}) {
|
||||
}
|
||||
|
||||
const models = [];
|
||||
const combosByName = new Map(
|
||||
combos.filter((c) => typeof c?.name === "string").map((c) => [c.name, c]),
|
||||
);
|
||||
|
||||
// Lookup map so aggregateComboCapabilities can recursively resolve nested combos
|
||||
const comboByName = Object.fromEntries(combos.map((c) => [c.name, c.models]));
|
||||
@@ -323,19 +384,23 @@ export async function buildModelsList(kindFilter, options = {}) {
|
||||
if (combo.kind === "webSearch" || combo.kind === "webFetch") {
|
||||
entry.kind = combo.kind;
|
||||
} else {
|
||||
const comboCaps = aggregateComboCapabilities(combo.models, comboByName);
|
||||
const comboCaps = aggregateComboCapabilities(combo.models, comboByName, comboSeatCapabilities);
|
||||
if (comboCaps) entry.capabilities = comboCaps;
|
||||
// Any seat can serve the request, so the only window a combo can promise is
|
||||
// its smallest. Combo entries were the only models on this endpoint that
|
||||
// published no limits at all, which leaves a client to guess from the name —
|
||||
// and it guesses high (see the snake_case note on the per-provider path).
|
||||
const { contextWindow, maxOutput } = comboSeatLimits(combo, combosByName);
|
||||
if (Number.isFinite(contextWindow)) entry.context_length = contextWindow;
|
||||
if (Number.isFinite(maxOutput)) entry.max_completion_tokens = maxOutput;
|
||||
}
|
||||
models.push(entry);
|
||||
}
|
||||
|
||||
if (connections.length === 0) {
|
||||
// DB unavailable -> return static models, filtered by per-model kind
|
||||
const aliasToProviderId = Object.fromEntries(
|
||||
Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id])
|
||||
);
|
||||
for (const [alias, providerModels] of Object.entries(PROVIDER_MODELS)) {
|
||||
const providerId = aliasToProviderId[alias] || alias;
|
||||
const providerId = ALIAS_TO_PROVIDER_ID[alias] || alias;
|
||||
if (!providerMatchesKinds(providerId, kindFilter)) continue;
|
||||
for (const model of providerModels) {
|
||||
if (!kindFilter.includes(modelKind(model))) continue;
|
||||
|
||||
@@ -24,6 +24,7 @@ const LIMIT_TOLERANCE = 0.1;
|
||||
// while building rather than on every lookup. Providers absent here keep whatever
|
||||
// the local pattern table resolves; names that already match need no entry.
|
||||
export const PROVIDER_ALIASES = {
|
||||
"github": "github-copilot",
|
||||
"glm": "zai",
|
||||
"glm-cn": "zhipuai",
|
||||
"claude": "anthropic",
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
|
||||
import { getCapabilitiesForModel, setCatalogSource } from "../../open-sse/providers/capabilities.js";
|
||||
import { PROVIDER_ALIASES, build } from "../../src/lib/modelCatalog/sync.js";
|
||||
|
||||
// The gpt-6 family's API window is 1.05M. The pattern table published 272,000 for
|
||||
// it — Kiro's own truncation, copied into the global glob — so every other
|
||||
// provider's gpt-6 models were advertised at 3.9x under their real window, and a
|
||||
// client reading context_length compacted (or refused) far too early.
|
||||
const API_WINDOW = 1050000;
|
||||
const LEGACY_GPT5_WINDOW = 400000;
|
||||
|
||||
describe("gpt-6 / gpt-5.4+ context windows", () => {
|
||||
it("reports the 1.05M API window for gpt-6 models on ordinary providers", () => {
|
||||
for (const [provider, model] of [
|
||||
["github", "gpt-6-luna"],
|
||||
["azure", "gpt-6-luna"],
|
||||
["openai", "gpt-6-luna"],
|
||||
["github", "gpt-6-sol"],
|
||||
["openai", "gpt-6-astra"],
|
||||
]) {
|
||||
expect(getCapabilitiesForModel(provider, model).contextWindow, `${provider}/${model}`).toBe(API_WINDOW);
|
||||
}
|
||||
});
|
||||
|
||||
// These two gateways really do truncate below the API, and their numbers live in
|
||||
// PROVIDER_CAPABILITIES, which outranks the pattern. Correcting the pattern must
|
||||
// leave them alone — that split is the whole point of the layering.
|
||||
it("leaves the gateways that truncate lower on their own numbers", () => {
|
||||
expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna").contextWindow).toBe(272000);
|
||||
expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna-thinking-agentic").contextWindow).toBe(272000);
|
||||
expect(getCapabilitiesForModel("codex", "gpt-5.6-luna").contextWindow).toBe(272000);
|
||||
expect(getCapabilitiesForModel("codex", "gpt-5.6-sol").contextWindow).toBe(372000);
|
||||
expect(getCapabilitiesForModel("codex", "gpt-6-astra").contextWindow).toBe(272000);
|
||||
});
|
||||
|
||||
it("keeps Devin CLI's seven GPT-5.4/5.5 variants at the gateway's 200k limit", () => {
|
||||
for (const model of [
|
||||
"gpt-5.4-high", "gpt-5.4-medium", "gpt-5.4-low",
|
||||
"gpt-5.5-xhigh", "gpt-5.5-high", "gpt-5.5-medium", "gpt-5.5-low",
|
||||
]) {
|
||||
expect(getCapabilitiesForModel("devin-cli", model), model).toMatchObject({
|
||||
contextWindow: 200000,
|
||||
maxOutput: 128000,
|
||||
vision: true,
|
||||
reasoning: true,
|
||||
thinkingFormat: "openai",
|
||||
});
|
||||
}
|
||||
expect(getCapabilitiesForModel("dv", "gpt-5.5-high").contextWindow).toBe(200000);
|
||||
expect(getCapabilitiesForModel("devin", "gpt-5.5-high").contextWindow).toBe(200000);
|
||||
});
|
||||
|
||||
// gpt-5.4 is where the 1.05M window starts and the mini/nano tiers are the
|
||||
// exception that stayed at 400k. Pattern resolution is first-match-wins, so this
|
||||
// is really a guard on the ORDER of the entries: move the tier patterns above
|
||||
// the mini/nano ones and both tiers silently report 1.05M.
|
||||
it("splits the 1.05M tiers from the 400k ones", () => {
|
||||
for (const model of [
|
||||
"gpt-5.4", "gpt-5.4-pro", "gpt-5.5", "gpt-5.5-pro", "gpt-5.6", "gpt-5.6-luna", "gpt-5.6-terra",
|
||||
]) {
|
||||
expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(API_WINDOW);
|
||||
}
|
||||
for (const model of ["gpt-5.4-mini", "gpt-5.4-nano", "gpt-5", "gpt-5.1", "gpt-5.2", "gpt-5.3-codex"]) {
|
||||
expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(LEGACY_GPT5_WINDOW);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// Copilot is "github" locally and "github-copilot" upstream. Without that mapping
|
||||
// build() resolved no upstream provider for it and skipped every Copilot model, so
|
||||
// the daily models.dev sync could never correct a stale hand-written number — which
|
||||
// is how the gpt-6 window stayed 3.9x wrong without anything noticing.
|
||||
describe("models.dev sync reaches GitHub Copilot", () => {
|
||||
it("maps the local github id onto the upstream github-copilot id", () => {
|
||||
expect(PROVIDER_ALIASES.github).toBe("github-copilot");
|
||||
});
|
||||
|
||||
it("records a Copilot limit that disagrees with the local tables", () => {
|
||||
// Copilot caps Claude output at 32k where the local floor assumes 64k.
|
||||
const upstream = {
|
||||
"github-copilot": {
|
||||
models: { "claude-sonnet-4.6": { limit: { context: 200000, output: 32000 } } },
|
||||
},
|
||||
};
|
||||
const entries = [
|
||||
{ provider: "github", model: "claude-sonnet-4.6", current: { contextWindow: 200000, maxOutput: 64000 } },
|
||||
];
|
||||
|
||||
// Context agrees, so only the output delta is recorded — and it is filed under
|
||||
// the local id, which is what the reader looks up.
|
||||
expect(build(upstream, entries).providers.github).toEqual({
|
||||
"claude-sonnet-4.6": { maxOutput: 32000 },
|
||||
});
|
||||
});
|
||||
|
||||
it("applies those Copilot deltas to an exact MODEL_CAPABILITIES id", () => {
|
||||
// claude-sonnet-4.6 is canonical-exact (128k). Without refine() on that
|
||||
// path the 32k Copilot delta from build() would never be read.
|
||||
setCatalogSource({
|
||||
getModalities: () => null,
|
||||
getLimits: (provider, model) =>
|
||||
provider === "github" && model === "claude-sonnet-4.6" ? { maxOutput: 32000 } : null,
|
||||
});
|
||||
try {
|
||||
expect(getCapabilitiesForModel("github", "claude-sonnet-4.6").maxOutput).toBe(32000);
|
||||
expect(getCapabilitiesForModel("claude", "claude-sonnet-4.6").maxOutput).toBe(128000);
|
||||
} finally {
|
||||
setCatalogSource(null);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,83 @@
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
import { setCatalogSource } from "../../open-sse/providers/capabilities.js";
|
||||
|
||||
const db = vi.hoisted(() => ({
|
||||
getProviderConnections: vi.fn(),
|
||||
getCombos: vi.fn(),
|
||||
getCustomModels: vi.fn(async () => []),
|
||||
getModelAliases: vi.fn(async () => ({})),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/localDb", () => db);
|
||||
vi.mock("@/lib/disabledModelsDb", () => ({
|
||||
getDisabledModels: vi.fn(async () => ({})),
|
||||
}));
|
||||
|
||||
const { buildModelsList } = await import("../../src/app/api/v1/models/route.js");
|
||||
|
||||
const syncedLimits = { contextWindow: 180000, maxOutput: 16000 };
|
||||
|
||||
async function modelsWithCombo(providerId, modelId, combos) {
|
||||
db.getProviderConnections.mockResolvedValue([{
|
||||
id: 1,
|
||||
provider: providerId,
|
||||
isActive: true,
|
||||
providerSpecificData: { enabledModels: [modelId] },
|
||||
}]);
|
||||
db.getCombos.mockResolvedValue(combos);
|
||||
setCatalogSource({
|
||||
getModalities: () => null,
|
||||
getLimits: (provider, model) =>
|
||||
provider === providerId && model === modelId ? syncedLimits : null,
|
||||
});
|
||||
try {
|
||||
return await buildModelsList(["llm"]);
|
||||
} finally {
|
||||
setCatalogSource(null);
|
||||
}
|
||||
}
|
||||
|
||||
describe("/v1/models combo limits", () => {
|
||||
it.each([
|
||||
["ocg", "opencode-go", "mimo-v2.5"],
|
||||
["xmtp", "xiaomi-tokenplan", "mimo-v2.5"],
|
||||
["ps", "poolside", "custom-model"],
|
||||
["ds", "deepseek", "deepseek-chat"],
|
||||
])("uses the real provider for a %s UI-alias seat", async (uiAlias, providerId, modelId) => {
|
||||
const combo = { name: "ui-alias-combo", models: [`${uiAlias}/${modelId}`] };
|
||||
const models = await modelsWithCombo(providerId, modelId, [combo]);
|
||||
const published = models.find((model) => model.id === combo.name);
|
||||
|
||||
expect(published).toMatchObject({
|
||||
context_length: 180000,
|
||||
max_completion_tokens: 16000,
|
||||
capabilities: { contextWindow: 180000, maxOutput: 16000 },
|
||||
});
|
||||
});
|
||||
|
||||
it("carries provider-scoped limits through a nested combo", async () => {
|
||||
const models = await modelsWithCombo("opencode-go", "mimo-v2.5", [
|
||||
{ name: "inner-combo", models: ["ocg/mimo-v2.5"] },
|
||||
{ name: "outer-combo", models: ["inner-combo"] },
|
||||
]);
|
||||
const outer = models.find((model) => model.id === "outer-combo");
|
||||
|
||||
expect(outer).toMatchObject({
|
||||
context_length: 180000,
|
||||
max_completion_tokens: 16000,
|
||||
capabilities: { contextWindow: 180000, maxOutput: 16000 },
|
||||
});
|
||||
});
|
||||
|
||||
it("publishes Devin CLI's 200k limit for a dv combo seat", async () => {
|
||||
const models = await modelsWithCombo("devin-cli", "gpt-5.5-high", [
|
||||
{ name: "devin-combo", models: ["dv/gpt-5.5-high"] },
|
||||
]);
|
||||
const combo = models.find((model) => model.id === "devin-combo");
|
||||
|
||||
expect(combo).toMatchObject({
|
||||
context_length: 200000,
|
||||
capabilities: { contextWindow: 200000 },
|
||||
});
|
||||
});
|
||||
});
|
||||
Reference in new issue
Block a user