feat(models): refresh model capabilities from models.dev in the background
Capability tables are hand-maintained, so a model gains vision or a wider context only when someone notices and edits the file. This adds a daily sync that fills the gap for models already in the registry. How it decides: - Modalities (vision/pdf/audio/video) belong to the MODEL — every gateway serving glm-5.3-flash serves the same weights — so they are keyed by model id and shared. A majority of sources must declare one, which keeps out lone mis-declarations: minimax-m2.5 (1 of 45), glm-4.7 (1 of 44) and gpt-oss-120b (2 of 76) are text-only despite a reseller claiming vision. - Context/output limits belong to the GATEWAY — each truncates differently (glm-5 ships as 202752/16384 on one host and 204800/131072 on another) — so they are keyed by provider + model and only the matching provider's own numbers are trusted. Both layers are strictly additive and sit BELOW the hand-written tables, which short-circuit first. A capability already true stays true. Mechanics: worker thread (the 4MB parse would block the loop ~20ms), ETag so an unchanged catalog costs one empty request, 60s startup delay, 30min backoff on failure, MODEL_CATALOG_SYNC=off to disable. Only the ~57KB delta is kept; lookups cost ~0.1us via an mtime-guarded cache. capabilities.js is bundled into the browser through useModelCaps, so it cannot import node:fs — the server injects the reader via setCatalogSource() from instrumentation. visionPatterns.js is the last resort: a model nobody has catalogued yet still accepts images when its id says so (qwen3-vl-plus, glm-4.6v, llava), with image-generation and embedding ids excluded. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -6,6 +6,16 @@
|
||||
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
|
||||
// 4. DEFAULT_CAPABILITIES — safe floor (always returned)
|
||||
//
|
||||
// Two extra layers then refine the result, and neither can override the hand
|
||||
// written tables above (steps 1-2 short-circuit before they are consulted):
|
||||
// • the synced catalog — modalities keyed by model, limits keyed by provider
|
||||
// + model, refreshed from models.dev in the background. It reads a file, so
|
||||
// the server installs it via setCatalogSource(); this module stays free of
|
||||
// node:fs because the dashboard bundles it into the browser too.
|
||||
// • visionPatterns.js — name-based vision detection, last resort so a model
|
||||
// nobody has catalogued yet still accepts images.
|
||||
// Both only ever turn a capability ON.
|
||||
//
|
||||
// ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
|
||||
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+
|
||||
// models, MIT). Each model exposes the exact fields we map below:
|
||||
@@ -23,6 +33,7 @@
|
||||
// 2.0+, Grok, Perplexity). Verify with: curl -s https://models.dev/api.json
|
||||
|
||||
import { matchPattern } from "./pricing.js";
|
||||
import { looksLikeVisionModel } from "./visionPatterns.js";
|
||||
|
||||
/**
|
||||
* Safe floor — every resolved result is merged over this so consumers
|
||||
@@ -94,8 +105,8 @@ export const MODEL_CAPABILITIES = {
|
||||
// Gemini image-gen / OpenAI image / xai image variants
|
||||
"gpt-image-1": { imageOutput: true, tools: false },
|
||||
|
||||
// GLM vision variants (text GLM has no vision) — 5.3-Flash is natively
|
||||
// multimodal per z.ai and carries the full 1M window.
|
||||
// GLM vision variants (text GLM has no vision) — 5.3-Flash and 5V-Turbo are
|
||||
// natively multimodal per z.ai, and 5.3-Flash carries the full 1M window.
|
||||
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "zai", contextWindow: 1000000, maxOutput: 131072 },
|
||||
"glm-4.6v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000, maxOutput: 32768 },
|
||||
"glm-4.5v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 64000, maxOutput: 16384 },
|
||||
@@ -333,6 +344,46 @@ export const PATTERN_CAPABILITIES = [
|
||||
* @param {string} model
|
||||
* @returns {object} full capabilities object
|
||||
*/
|
||||
const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
|
||||
// Catalog lookups, installed by the server at startup. Left as no-ops in the
|
||||
// browser bundle, where there is no file to read.
|
||||
let catalogSource = null;
|
||||
|
||||
/**
|
||||
* Install the synced catalog reader (server only).
|
||||
* @param {{ getModalities: Function, getLimits: Function } | null} source
|
||||
*/
|
||||
export function setCatalogSource(source) {
|
||||
catalogSource = source;
|
||||
}
|
||||
|
||||
// Apply the synced catalog + name heuristic on top of a table-resolved result.
|
||||
// Strictly additive: a capability already true stays true, and a false one only
|
||||
// flips when an outside source positively declares support.
|
||||
function refine(base, provider, model) {
|
||||
const result = { ...DEFAULT_CAPABILITIES, ...base };
|
||||
|
||||
if (catalogSource) {
|
||||
const modalities = catalogSource.getModalities(model);
|
||||
if (modalities) {
|
||||
for (const key of MODALITY_KEYS) {
|
||||
if (modalities[key] === true) result[key] = true;
|
||||
}
|
||||
}
|
||||
|
||||
const limits = catalogSource.getLimits(provider, model);
|
||||
if (limits) {
|
||||
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
|
||||
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
|
||||
}
|
||||
}
|
||||
|
||||
if (!result.vision && looksLikeVisionModel(model)) result.vision = true;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
export function getCapabilitiesForModel(provider, model) {
|
||||
if (!model) return { ...DEFAULT_CAPABILITIES };
|
||||
|
||||
@@ -350,13 +401,13 @@ export function getCapabilitiesForModel(provider, model) {
|
||||
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
|
||||
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
|
||||
|
||||
// 3. Pattern match (first match wins)
|
||||
// 3. Pattern match (first match wins), refined by catalog + name heuristic
|
||||
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
|
||||
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
|
||||
return { ...DEFAULT_CAPABILITIES, ...caps };
|
||||
return refine(caps, provider, model);
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Floor
|
||||
return { ...DEFAULT_CAPABILITIES };
|
||||
return refine(null, provider, model);
|
||||
}
|
||||
|
||||
72
open-sse/providers/catalogOverride.js
Normal file
72
open-sse/providers/catalogOverride.js
Normal file
@@ -0,0 +1,72 @@
|
||||
// Read side of the model catalog synced from models.dev.
|
||||
//
|
||||
// The file is the source of truth; the only thing held in memory is a parsed
|
||||
// copy dropped as soon as the file's mtime changes. getCapabilitiesForModel is
|
||||
// synchronous and runs per request, so the hot path is one stat (~1us) and the
|
||||
// parse (~0.1ms on a ~18KB file) only reruns after a sync.
|
||||
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import { DATA_DIR } from "@/lib/dataDir.js";
|
||||
|
||||
export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
|
||||
// Trimmed upstream catalog, read by the add-models skill (not by the router).
|
||||
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
|
||||
|
||||
const EMPTY = { models: {}, providers: {} };
|
||||
let cache = EMPTY;
|
||||
let cachedMtime = -1;
|
||||
|
||||
// "zai-org/GLM-4.6V:free" -> "glm-4.6v"
|
||||
function baseId(model) {
|
||||
if (!model) return "";
|
||||
const withoutVendor = model.includes("/") ? model.split("/").pop() : model;
|
||||
return withoutVendor.toLowerCase().split(":")[0];
|
||||
}
|
||||
|
||||
function load() {
|
||||
let mtime;
|
||||
try {
|
||||
mtime = fs.statSync(CATALOG_FILE).mtimeMs;
|
||||
} catch {
|
||||
cache = EMPTY;
|
||||
cachedMtime = -1;
|
||||
return cache;
|
||||
}
|
||||
if (mtime === cachedMtime) return cache;
|
||||
|
||||
cachedMtime = mtime;
|
||||
try {
|
||||
const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8"));
|
||||
cache = { models: parsed?.models || {}, providers: parsed?.providers || {} };
|
||||
} catch {
|
||||
cache = EMPTY;
|
||||
}
|
||||
return cache;
|
||||
}
|
||||
|
||||
// Modality is a property of the model itself — any gateway serving it inherits
|
||||
// the same image/video/pdf support, so this is keyed by model id alone.
|
||||
export function getCatalogModalities(model) {
|
||||
return load().models[baseId(model)] || null;
|
||||
}
|
||||
|
||||
// Context and output limits are a property of the gateway, not the model: each
|
||||
// one truncates differently, so these stay keyed by provider + model.
|
||||
export function getCatalogLimits(provider, model) {
|
||||
const byProvider = provider && load().providers[provider];
|
||||
if (!byProvider) return null;
|
||||
return byProvider[model] || byProvider[baseId(model)] || null;
|
||||
}
|
||||
|
||||
// Force a re-read on the next lookup (called right after a sync writes the file).
|
||||
export function invalidateCatalog() {
|
||||
cachedMtime = -1;
|
||||
}
|
||||
|
||||
// Hand the reader to capabilities.js. That module is bundled into the browser
|
||||
// too, so it cannot import this file directly — the server pushes it in.
|
||||
export async function installCatalogSource() {
|
||||
const { setCatalogSource } = await import("./capabilities.js");
|
||||
setCatalogSource({ getModalities: getCatalogModalities, getLimits: getCatalogLimits });
|
||||
}
|
||||
42
open-sse/providers/visionPatterns.js
Normal file
42
open-sse/providers/visionPatterns.js
Normal file
@@ -0,0 +1,42 @@
|
||||
// Name-based vision detection — last resort when neither the catalog file nor
|
||||
// the capability tables know a model. Vendors put the modality in the id
|
||||
// ("qwen3-vl-plus", "glm-4.6v", "deepseek-v4-flash-vision-exp"), so a custom or
|
||||
// freshly released model still gets image input instead of silently dropping it.
|
||||
//
|
||||
// Only ever turns vision ON. Never used to turn a declared capability off.
|
||||
|
||||
const SEP = "[-_/:.]";
|
||||
|
||||
// Image GENERATION, video generation, and non-chat models also carry these
|
||||
// words but take no image input — checked first so they can never match.
|
||||
const NOT_VISION = new RegExp(
|
||||
[
|
||||
`(^|${SEP})(image|img)(${SEP}|$)`,
|
||||
"stable-image", "gen[0-9]_image", "nanobanana", "imagine",
|
||||
"t2v", "i2v", "flux", "dall", "sdxl", "diffusion",
|
||||
"embed", "rerank", "guard", "moderation",
|
||||
"tts", "stt", "whisper", "voice", "speech", "audio",
|
||||
].join("|"),
|
||||
"i"
|
||||
);
|
||||
|
||||
// Explicit modality words, plus the "<digit>v" suffix vendors use for vision
|
||||
// variants (glm-4.6v, glm-5v-turbo). The digit-v branch requires a dotted
|
||||
// version so the never-shipped `gpt-4v` cannot match.
|
||||
const VISION_NAME = new RegExp(
|
||||
[
|
||||
`(^|${SEP})(vision|vl|vlm|multimodal|omni|visual)(${SEP}|$)`,
|
||||
`[0-9]\\.[0-9]+v(${SEP}|$)`,
|
||||
`(^|${SEP})glm-[0-9]+v(${SEP}|$)`,
|
||||
"(^|[-_/:.])(llava|pixtral|internvl|cogvlm|minicpm-v|moondream|idefics|fuyu)",
|
||||
].join("|"),
|
||||
"i"
|
||||
);
|
||||
|
||||
// Does this model id look like a vision model? Name signal only.
|
||||
export function looksLikeVisionModel(modelId) {
|
||||
if (!modelId) return false;
|
||||
const id = String(modelId).toLowerCase();
|
||||
if (NOT_VISION.test(id)) return false;
|
||||
return VISION_NAME.test(id);
|
||||
}
|
||||
Reference in New Issue
Block a user