363 Commits

Author SHA1 Message Date
ddfa789a31 fix(providers): add missing providerStrategies state + restore notify store
* page.js referenced setProviderStrategies (line 166) and
  providerStrategies (line 597) but never declared the
  useState — providers page threw ReferenceError on mount.
* The destructure of useNotificationStore() was also dropped
  by the merge of origin/master; 5 sites in the file called
  notify.error/.success/.warning.
* Both were present before the merge (commits de9e00c6 +
  upstream master versions). The statusFilter commit
  (d1d4e0f0) was the last state-block edit and survived,
  but adjacent state lines were lost during the conflict
  resolution.
2026-09-07 15:45:14 +07:00
6f52d7020c fix(chatCore): add missing capsOverride + streamErrorPatterns to destructure
* handleChatCore() referenced capsOverride (line 162) and
  streamErrorPatterns (line 475) but the destructured param list
  did not include them. Callers that did not pass these (e.g.
  open-sse/handlers/responsesHandler.js, unit callers, older
  client builds) would trigger 'capsOverride is not defined' /
  'streamErrorPatterns is not defined' ReferenceError mid-request.
* Both defaulted to null. capsOverride is read by capability merge
  (line 162, already guarded by '|| {}'). streamErrorPatterns is
  read in the early-peek hook (line 484) which was null-safe
  only because the variable happened to be in scope when chat.js
  spread it in; responsesHandler never passed it and would crash.
* Unblocks all callers regardless of which fields they pass.
2026-09-07 15:23:48 +07:00
59f17b3725 feat(usage): raw request detail modal + raw stream capture
* Add /api/usage/request-details/raw endpoint serving a single
  stored request detail verbatim (raw payloads), with /raw doc
  clarifying it stays gated by the dashboard auth layer.
* Add RawDetailModal opened from a new 'Raw' button in
  RequestDetailsTab. Modal loads /raw, exposes per-section copy
  buttons and a 'Copy all (JSON)' that bundles every section.
* Capture the raw provider SSE text inside the streaming
  transform (cap 64KB) and forward it through
  onStreamComplete.rawProviderText so handler stores it as the
  providerResponse. response.content stays the extracted user
  text. Tool-call-only turns remain so the marker.
* Accumulate from translated client-facing chunks instead of
  raw provider shapes so Responses, Claude delta types, and
  Gemini/Antigravity parts all contribute.
* Drop redaction from the list endpoint; raw access is now via
  the dedicated /raw endpoint. Tests cover the new behavior.
2026-09-07 14:23:48 +07:00
a835771c97 Merge origin/master (v0.5.69) into gitea/new_feature 2026-09-07 14:10:11 +07:00
decolua
eb712ca821 # v0.5.69 (2026-09-05)
## Features
- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities
- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`)
- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys
- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819)
- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through
- **CLI tools**: replace Copilot MITM with VS Code extension setup guide
- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace

## Fixes
- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792)
- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813)
- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797)
- **Dashboard**: dynamic mode label for local/remote detection (#3801)
- **Codex**: format reset credit API errors cleanly (#3778)
- **Security**: guard cowork MCP tools probe against SSRF (#3783)
- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800)
- **Logger**: suppress noisy background token refresh logs
- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory
2026-09-05 22:57:00 +07:00
Sina Sadeghi
11222eff0f feat(opencode-go): muse-spark-1.2 and Responses tool fixes (#3820)
- Add muse-spark-1.2-contributor as responses-only model on OpenCode Go
- Normalize object tool schemas without properties in OpenCode Go executor
- Make fallback Responses call_ids unique across same-millisecond calls
- Make Responses output coercion fail-soft for circular and non-stringifiable values
2026-09-05 22:39:22 +07:00
decolua
e214fb1c30 feat(usage): add Claude Fable quota tracker support
- Recognize Fable weekly windows and normalize to weekly fable (7d)
- Fall back to 100% available weekly Fable window when Anthropic payload omits it
- Forward remaining percentages and enforce canonical Claude quota order in Quota Tracker

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-05 22:05:18 +07:00
decolua
f615a83cb2 feat(dashboard): group Antigravity model quotas and trim hidden keys
- Group Antigravity Gemini text models into single 'Gemini (Flash / Pro)' quota
- Group Claude models into single 'Claude (Sonnet / Opus)' quota
- Prune stale or legacy model keys from hidden quota visibility list

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-05 21:56:07 +07:00
Sina Sadeghi
e74db4d0a6 feat(opencode-go): add muse-spark-1.3-contributor and fix parallel tool calls on Responses paths (#3819)
- Add muse-spark-1.3-contributor as responses-only model on OpenCode Go with dedicated executor
- Key Responses→chat streaming tool calls by item_id to prevent parallel tool calls merging into index 0
- Standardize tool coercions and call_id clamping in Responses API translation
2026-09-05 21:49:53 +07:00
Sutarto Jordan Chrisfivo
77e6a227fe fix(claude): normalize adaptive auto effort (#3792)
Claude adaptive requests without an explicit effort are normalized to
output_config.effort: "high" instead of forwarding the unsupported
literal value "auto" which Anthropic rejects with HTTP 400.
2026-09-05 21:41:25 +07:00
Hifzi
1442cc73ce fix(antigravity): prevent Google anti-abuse rate limits on multi-account refresh (#3813) 2026-09-05 21:28:30 +07:00
Federico Liva
fb9fab0206 fix(anthropic-compatible): send Claude beta flags to nodes fronting Anthropic (#3797) 2026-09-05 21:26:16 +07:00
vianhanif
28cfd9facf fix(dashboard): dynamic mode label for local/remote detection (#3801) 2026-09-05 21:20:53 +07:00
Raisal P Wardana
1a3d446831 fix(codex): format reset credit API errors (#3778) 2026-09-05 21:18:07 +07:00
soroush5
97f3ab97b1 fix(security): guard cowork-mcp-tools probe against SSRF (#3783) 2026-09-05 21:12:16 +07:00
JOJO
0da803eef4 fix(usage): track OpenCode Go quota (#3791)
OpenCode Go API-key connections now appear in the Quota Tracker and report rolling, weekly, and monthly subscription usage.
2026-09-05 21:10:10 +07:00
turingcat
81f4f93082 fix(opencode-go): send stable session header (#3800)
- add a dedicated OpenCode Go executor that always sends x-opencode-session
- preserve a valid caller-provided native OpenCode session header
- translate downstream Agent session IDs into opaque, stable, Agent-scoped IDs
- forward the original provider session seed and client tool on both initial and credential-refresh requests
2026-09-05 21:09:49 +07:00
zmf
cec672d9d9 feat(providers): align codebuddy-cn catalog/capabilities with server config
- Sync codebuddy-cn catalog and capabilities with copilot.tencent.com server payload
- Fix thinkingCanDisable semantics for glm-5.3 and deepseek-v4 models
- Add missing glm-5.2 thinking levels to thinkingLevels.js
- Add glm-5-turbo model to glm and glm-cn registries
2026-09-05 21:03:02 +07:00
An Nguyen
ed963931b4 feat(codex): add GPT-5.6 Sol, Terra, and Luna image aliases (#3806) 2026-09-05 21:01:52 +07:00
decolua
f388b5e56b chore(logger): remove noisy background token refresh logs
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-04 10:17:26 +07:00
decolua
b84681d5a4 feat(cli-tools): replace copilot mitm with vscode extension setup guide
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 23:27:04 +07:00
hangyu
2ab6a4c949 feat(qoder): refresh model catalog, add capability mapping and image pass-through
- Registry/constants: drop qmodel_preview/gm51model, add lite,
  qmodel_38max (Qwen3.8-Max), qfmodel (Qwen3.8-Flash), gmodel (GLM-5.3),
  gfmodel (GLM-5.3-Flash)
- capabilities: add PROVIDER_CAPABILITIES['qoder'] so opaque internal
  ids resolve to their real models' context windows and limits
- executor: preserve image blocks instead of flattening away, convert
  Claude-style image blocks, and hash images into chat_record_id
- tests: cover image preservation, data-URI and Claude-block conversion
- build(docker): use CN mirrors for apk and npm
2026-09-03 23:02:52 +07:00
decolua
c08efdbe2b feat(gemini): persist and replay thoughtSignature with session namespace
- Add open-sse/services/thoughtSignatureStore.js managing LRU Map (2k) + SQLite kv table
- Store thoughtSignature with sessionId namespace and toolCallId fallback
- Replay cached signature by sessionId:tool_call_id to prevent multi-process collisions
- Normalize Antigravity sessionId to numeric int64 format

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 18:20:04 +07:00
decolua
4eda76e2ab # v0.5.65 (2026-09-03)
## Features
- **Fetch**: add Ollama Cloud web fetch provider
- **Gemini / Antigravity**: add Gemini 3.8 Flash support and bump IDE fingerprint to 2.11.0
- **Claude**: add Claude Fable 5.1 support (adaptive thinking with `output_config.effort`), bump Claude Code fingerprint to 2.1.258 for new-model access
- **Providers**: add client-side status filter (All / Active / Inactive / No connection) on the Providers dashboard; add max height and scroll for connection list
- **Providers & Models**: streamline tokenrouter model catalog down to 22 flagship/newest models and add missing provider icons; refresh Codebuddy-CN catalog (add hy4-preview/hy3/glm-5.3/kimi-k3-1, drop EOL glm-5.0/glm-4.7)
- **Models**: capability toggles (vision, reasoning) when adding custom models with upsert and live caps refresh
- **CLI tools**: support saving and managing custom API key presets
- **Quota**: add usage and rate-limit tracking for Groq via `x-ratelimit-*` headers
- **i18n**: complete Indonesian translation (1391 keys)

## Fixes
- **Security**: close SSRF guard bypasses in `ssrfGuard.js` (alternate IPv6 encodings, hostname trailing dots, wildcard DNS resolution check, safe redirect handling) (#3714)
- **Model markers**: strip the `[1m]` context marker Claude Code appends to model names (`claude-opus-5[1m]`) preventing model resolution failures (#3690)
- **Claude**: drop `server_tool_use` blocks carrying foreign IDs to avoid Anthropic 400 rejections; never anchor cache breakpoints on `defer_loading` tools (#3567)
- **Antigravity**: strike-break optimistic quota readings that keep 429ing by blocking the connection+model pair for 15m after 3 strikes (#3681); preserve client identity on model catalog requests (#3414)
- **Auth**: protect root `/responses` rewrite requiring API key validation in dashboardGuard
- **Chat & Docker**: return 503 Service Unavailable when all credentials are rate-limited; explicitly bundle `node-machine-id` into standalone Docker runtime image
- **OpenCode**: route Muse Spark models to `/zen/v1/responses` and declare vision support; filter inactive free model
- **Kiro**: preserve inline images as OpenAI-compatible `image_url` parts in OpenAI MITM; remove redundant top-level `systemPrompt` from payload
- **Usage**: read Responses-shape `cached_tokens` in `extractUsageFromResponse` for non-streaming traffic
- **Models**: support single model lookup with provider-prefixed IDs (e.g. `cc/claude-sonnet-5`)
- **Translator**: route Gemini thinking through `reasoning_effort` on OpenAI-compatible wire; convert `prefixItems` and ensure array items in Gemini schema sanitizer
- **UI**: apply persisted theme before first paint to prevent flash on reload; translate combo vision adapter label
2026-09-03 10:36:34 +07:00
Sami Basra
e0ffc7e2a1 feat(fetch): add Ollama Cloud web fetch provider 2026-09-03 10:13:15 +07:00
Federico Liva
6ab9ca9eb1 fix(claude): never anchor cache breakpoint on defer_loading tools (#3567) 2026-09-03 10:05:35 +07:00
decolua
6efb97904b feat(providers): streamline tokenrouter models and add missing provider icons
- Prune tokenrouter seed models from 121 to 22 flagship/newest models
- Add z-ai/glm-5.3-free with 0 pricing
- Add missing 128x128 icons for alims-intl, alitp-intl, fish-audio, and selfhosted-* providers

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 09:59:30 +07:00
decolua
831001c322 feat(providers): add max height and scroll for connection list
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 09:54:43 +07:00
IEatCodeDaily
e7dd72a8d7 fix(usage): read Responses-shape cached_tokens in extractUsageFromResponse
Non-streaming codex traffic recorded cached_tokens: 0 even when upstream
prompt caching worked. The Claude-format branch (which OpenAI Responses
usage also matches) never read input_tokens_details, and the OpenAI
branch ignored a top-level flat cached_tokens. Read both in both
branches; Responses prompts are cache-inclusive so canonicalizeUsage
passes the value through without folding. 5 new regression tests.
2026-09-03 09:48:19 +07:00
Sutarto Jordan Chrisfivo
5caa72f5fb fix(models): support single model lookup
Support single model lookup by replacing the one-segment models route
with a catch-all route that preserves capability kind paths while
allowing provider-prefixed IDs like cc/claude-sonnet-5.
2026-09-03 09:43:01 +07:00
zmf
e014cb537f feat(codebuddy-cn): refresh model catalog — add hy4-preview/hy3/glm-5.3/kimi-k3, drop EOL glm-5.0/glm-4.7
- Add hy3, hy3-x, hy4-preview, hy4-preview-x, glm-5.3, glm-5.3-flash, kimi-k3-1
- Remove dead models glm-5.0, glm-4.7 (API 11102)
- Register capabilities and context windows in PROVIDER_CAPABILITIES
- Configure supported effort sets in PATTERN_THINKING
2026-09-03 09:41:02 +07:00
openhands
d1d4e0f02b feat(providers): add status filter to providers dashboard
Adds a client-side status filter (All / Active / Inactive / No
connection) to the Providers page, applied over the already-fetched
provider + connection list. Status derives from getProviderStats
(total, allDisabled); noAuth providers count as Active. Filter composes
with the existing search across all provider sections. Part of #3699.
2026-09-03 09:39:01 +07:00
louis-cai
ac98dd9d32 fix(antigravity): strike-break optimistic quota readings that keep 429ing
Google's quota API can report remaining quota while generation endpoints
keep returning 429 (sprint/weekly dual-pool mismatch). handleAntigravityQuotaError
trusted remainingPercentage > 0 as healthy and returned null, causing 429 retry
loops across multi-account pools.

Add a strike-based circuit breaker to the optimistic and unavailable quota paths:
- After 3 strikes (429/409) within 60s for the same connection+model, cache-block
  that pair for 15 minutes by synthesizing an entry in the shared RAM quota cache.
- Re-assert active strike blocks across refreshes so optimistic readings cannot
  resurrect a broken pair prematurely.
- Reset strikes and clear synthesized cache entry upon successful request.
- Keep exact-resetAt handling for genuine 0% exhausted readings.

Closes #3681
2026-09-03 09:34:24 +07:00
Teguh Rijanandi
a58902e4a7 feat(i18n): complete Indonesian translation (1391 keys) 2026-09-03 09:33:01 +07:00
Sutarto Jordan Chrisfivo
98579f98c1 fix(auth): protect root /responses rewrite
Add /responses to PUBLIC_PREFIXES in dashboardGuard so pre-rewrite remote
requests require API key validation as intended.
2026-09-03 09:29:12 +07:00
vianhanif
15687d1913 fix(chat,docker): return 503 for rate-limited providers and bundle node-machine-id
- chat: always return 503 Service Unavailable when all credentials are rate-limited
- Dockerfile: explicitly copy node-machine-id into standalone runtime image
2026-09-03 09:25:17 +07:00
anojndr
acb5c34cdc fix(opencode): route Muse Spark models to Responses API and declare vision
Route all Muse Spark models (not just 1.2) on OpenCode Free to
/zen/v1/responses via isMuseSparkModel(), fixing HTTP 500 on
muse-spark-1.3-contributor-free. Declare vision:true on Muse Spark
models so image input is no longer stripped; register 1.3 in the
registry and capabilities. Scoped to opencode only — other providers
keep Chat Completions routing.
2026-09-03 09:24:18 +07:00
openhands
b870b5d41b fix(security): close SSRF guard bypasses in ssrfGuard.js (#3714)
Closes four SSRF guard bypasses reported in #3714:
- Block alternate IPv6 encodings (hex format, NAT64, IPv4-compatible, IPv4-mapped) by parsing to 16-bit groups
- Normalize trailing dots on hostnames to prevent FQDN bypasses
- Add assertPublicUrlResolved() with DNS resolution to block wildcard DNS domains resolving to private/metadata IPs
- Add fetchPublic() to safely handle and validate HTTP redirects
2026-09-03 09:22:22 +07:00
Outis
1f190bd00b fix(kiro): preserve inline images in OpenAI MITM
Forward Kiro userInputMessage.images as OpenAI-compatible image_url content parts.
2026-09-03 09:21:37 +07:00
Zafar
70f15aa50b feat(antigravity,gemini): add Gemini 3.8 Flash support and bump IDE fingerprint to 2.11.0
Co-authored-by: Schnee111 <daffamaarif.dev@gmail.com>
Co-authored-by: AhooraZen <ahoora935137@gmail.com>
Co-authored-by: anojndr <anojndr@gmail.com>
Co-authored-by: Emirhan <emirhan551952@gmail.com>
2026-09-03 09:13:45 +07:00
Lek Huda
1fe996db6a fix(translator): route Gemini thinking through reasoning_effort on OpenAI-compatible wire 2026-09-03 09:12:34 +07:00
decolua
c24a854278 feat(cli-tools): support saving and managing custom API key presets
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 09:06:08 +07:00
decolua
1fc2a81d65 fix(kiro): remove redundant top-level systemPrompt field from payload
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 09:06:02 +07:00
decolua
f6c59d30b0 fix(gemini): convert prefixItems and ensure array items in schema sanitizer
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-03 09:05:57 +07:00
LucasOl1337
ac9120fde3 fix(claude): support Fable 5.1
- add claude-fable-5-1 to the Claude Code model catalog (1M context,
  permanent adaptive thinking)
- centralize the spoofed Claude Code version and update both request
  and billing identities to 2.1.257 (Fable 5.1 rejects < 2.1.251)
- send output_config.effort without the redundant thinking switch for
  permanently adaptive models
- add regression coverage for capabilities, headers, billing identity
  and adaptive-effort payload

# Conflicts:
#	open-sse/providers/registry/claude.js
#	open-sse/providers/shared.js
#	open-sse/utils/claudeCloaking.js
#	tests/__baseline__/providers-baseline.json
2026-09-02 20:42:53 +07:00
Federico Liva
ee7a961633 fix: strip the [1m] context marker Claude Code appends to the model name
With the 1M-context beta enabled, Claude Code sends model: "claude-opus-5[1m]".
The marker is a client-side annotation — it matches no combo name, no alias and
no provider/model pair — so the request dies at model resolution with
"Invalid model format" and the client reports "There's an issue with the
selected model". Every request from that session fails until the beta is
switched off.

New open-sse/utils/modelMarkers.js exporting stripModelContextMarker(modelStr)
-> { model, contextMarker }. handleChat strips the marker before resolution and
normalizes body.model so downstream logging and translation see the real name.
Only a trailing marker is stripped, so a model whose name genuinely contains
brackets is left alone.

The capability itself travels in anthropic-beta: context-1m-2025-08-07, which
the default executor already forwards untouched — only the routing key needed
cleaning.

Fixes #3690.

Tests: tests/unit/model-context-marker.test.js (6 cases).
2026-09-02 20:20:32 +07:00
Matt Van Horn
f68d2f5ee5 fix(antigravity): preserve client identity on model catalog requests
Restrict the legacy IDE-version override to generation endpoints so
catalog and other passthrough requests keep their original User-Agent
and metadata.ideVersion, letting newer Antigravity releases see current
models like Gemini 3.7 Flash in the MITM selector.

Fixes #3414
2026-09-02 20:14:59 +07:00
Federico Liva
ed1bd0c528 fix(claude): drop server_tool_use blocks carrying a foreign id
Anthropic validates server_tool_use.id against ^srvtoolu_[a-zA-Z0-9_]+$
and 400s the whole request when one does not match. A combo that falls
back to a provider with its own built-in tools (z.ai/glm emits
OpenAI-style call_ ids for analyze_image) leaves such blocks in the
history, so every later Claude turn fails.

Extend normalizeClaudePassthrough to drop those blocks (reusing the
existing loop), drop the paired tool_result / web_search_tool_result
referencing a dropped id, and drop empty text blocks plus messages left
with no content. Well-formed srvtoolu_ blocks and regular tool_use ids
are untouched.
2026-09-02 20:07:16 +07:00
docaohieu2808
925cb4aade fix(ui): apply persisted theme before first paint to avoid flash on reload
Theme was applied from the client store in useEffect (after hydration),
so a reload painted the default light theme for a frame before the
stored dark theme was reapplied. Add a blocking head script that reads
the persisted zustand theme key and sets the dark class on
documentElement before first paint, mirroring applyTheme() including
system -> prefers-color-scheme resolution.
2026-09-02 20:05:52 +07:00
openhands
b9c92cb83c feat(quota): add usage tracking for Groq
First slice of #3701: quota tracking for Groq via x-ratelimit-* response
headers on the models endpoint (no dedicated quota endpoint exists, and
reading usage costs zero tokens).

- usage/groq.js: parse request+token limit/remaining headers; Go-style
  duration reset headers ("2m59.56s") resolve to future timestamps;
  missing key/401/403 -> message, 2xx without headers -> soft
  "not tracked yet" with quotas:{}
- registry/groq.js: transport.usage.url (reuses validateUrl) +
  features {usage, usageApikey}
- services/usage.js: groq entry in USAGE_HANDLERS
- ProviderLimits/utils.js: parseQuotaData case (absolute used/total,
  codex/kiro style)
- tests: groq-usage.test.js (registry flags, header parsing, soft
  not-tracked path, missing key/401, parseQuotaData)
2026-09-02 20:04:35 +07:00
decolua
44e4b80bbe fix(models): filter dead opencode free model
Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-02 20:01:15 +07:00
dajinglingpake
9d3f7646d1 fix(i18n): translate combo vision adapter label 2026-09-02 19:45:34 +07:00
decolua
009cac6326 fix(claude): bump CC fingerprint to 2.1.258 for new-model access
Anthropic gates newly released models (e.g. claude-fable-5-1) to Claude
Code >= 2.1.251; the spoofed 2.1.92 client got HTTP 400 on every request.
Bump User-Agent + billing-header version to 2.1.258 and refresh the
providers baseline snapshot.

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-02 11:27:21 +07:00
decolua
38f031f4c9 feat(models): capability toggles for custom models with upsert and live caps refresh
- AddCustomModelModal lets users pick vision/reasoning caps when adding a model
- POST /api/models/custom whitelists caps to booleans
- aliasRepo.addCustomModel upserts — re-adding updates caps/name in place
- /api/models includes custom llm models with stored caps overriding the heuristic
- useModelCaps refetches on customModelChanged instead of trusting a stale cache

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-01 10:45:51 +07:00
decolua
90b52e06ff # v0.5.59 (2026-08-29)
## Features
- **Search**: new web search providers — Antigravity (Google Search grounding
  on the existing OAuth account pool, citations keyed and merged by URL) and
  Xquik (X search with `x-api-key` auth, cursor pagination, credit-based
  usage), both on `POST /v1/search`. Based on #3437 by @Nautilaceae
- **Search**: ollama-search and zai-search borrow a chat provider's API key
  instead of requiring their own connection, driven by a new
  `credentialFallback` registry field. zai-search later folded into the `glm`
  provider itself so the web search page shows the shared connection
- **Models**: daily background sync of model capabilities from models.dev —
  modalities keyed by model id (majority of sources must declare one),
  context/output limits keyed by provider + model, strictly additive and
  sitting below the hand-written tables. ETag + mtime cache, 60s startup
  delay, `MODEL_CATALOG_SYNC=off` to disable
- **Models**: add GLM-5.3-Flash (1M context, natively multimodal), DeepSeek
  V4 Vision, Grok 4.5/4.6 (500k context); correct glm-4.6v/4.5v video input
  and output limits, backfill glm-4.6v on glm-cn
- **Usage**: show the Zed plan quota on the dashboard — plan, edit
  predictions, hosted model requests and billing-cycle reset; unlimited rows
  render as "N used · Unlimited"
- **Usage**: track GPT-5.3-Codex-Spark quota windows (spark_session /
  spark_weekly) from the Codex usage response (#3431)
- **Antigravity**: quota-aware routing — on 409/429 fetch live quota for the
  exact per-model resetAt and skip only the exhausted account/model pair;
  report the earliest reset when every account is blocked (#3561)
- **Antigravity**: map image `size` to the aspect-ratio model suffix (-WxH);
  add the Gemini 3.7 Flash tiers to MITM defaultModels so they show up in
  the dashboard model-mapping table
- **Dashboard**: bulk import Grok CLI accounts from JSON — paste an array or
  drag-drop multiple .json files, all OAuth connections created in a single
  call, mirroring the codex flow
- **CLI tools**: endpoint presets shared across every tool card through one
  live-resyncing store, instead of per-card localStorage copies that never
  saw each other's saved endpoints
- **Token Saver**: configurable compression timeout (`headroomTimeoutMs`) —
  the fixed 3000 ms made busy machines time out and send inconsistently
  compressed bodies, hurting prompt caching
- **i18n**: pt-BR expanded to 1132 terms

## Fixes
- **Stream**: record usage when a client closes on the terminal event — the
  Responses API has no [DONE] sentinel, so codex closed the socket on
  `response.completed` and cancelled the reader before flush() ran its usage
  side effects; the tail now lives in a once-guarded finalizeStream(). Also
  stop logging a disconnect for every completed Responses call
- **Stream**: parse the trailing NDJSON line an Ollama stream leaves behind
  without a closing newline — the final chunk carrying `done_reason` and the
  token counts was dropped
- **Session**: read the Claude Code session id from the
  `x-claude-code-session-id` header — `metadata.user_id` is dropped by
  Responses translation, splitting one conversation across several
  `prompt_cache_key` values and missing the upstream prefix cache
- **Usage**: preserve nested `cached_tokens` — the top-level-only read
  persisted `cached_tokens: 0` for every Responses-format provider (codex,
  grok-cli, …), billing cache hits at the full input rate
- **Usage**: GLM quotas accept CREDIT_LIMIT plans and multi-interval windows
  (5h session / 7d weekly) instead of overwriting a single "session" key
- **Models**: the catalog sync no longer erases its own output — deltas were
  measured against the previous run's writes (the second run cut `providers`
  from 20 entries to 5); one vote per provider in the modality tally, ETag
  restored from file on startup, and the worker thread dropped after the
  bundler rewrote its path into a module-not-found error
- **Executor**: CommandCode returns errors as a `type:"error"` event inside
  an HTTP 200 NDJSON stream — peek the first events before committing, abort
  and return a real 4xx/5xx so combo/account fallback triggers instead of
  streaming the error text as content
- **Search**: scope failure locks on the credential-fallback path — a failing
  search locked `modelLock___all` and took the shared glm key offline for
  chat as well; locks are now attributed to the connection's owner and
  scoped to `websearch:<provider>`
- **Providers**: connection tests get a 15s AbortSignal timeout instead of
  hanging and exhausting the browser socket pool; guard undefined provider
  names on the providers page
- **Antigravity**: sanitize competing-client branding via a config-driven
  rule table (Zed's Claude-agent prompt, opencode → antigravity) — upstream
  answers 429 Quota Exhausted. Applied in the executor so the shared
  openai-to-gemini translator leaves gemini/vertex/zed untouched
- **MiniMax**: preserve images on the sourceFormat-matched OpenAI transport
  — MiniMax-M3 resolved a Claude-shaped body posted to the OpenAI endpoint,
  silently dropping `image_url` blocks (#3418)
- **Claude**: decloak tool names in same-format streaming passthrough —
  OAuth-cloaked names (CLAUDE_TOOL_SUFFIX) leaked to the client and every
  tool call was rejected as unknown
- **Tools**: default a missing `tools[].type` to "custom" on Claude-format
  requests — strict Anthropic-compatible gateways (MiniMax) reject the
  request with 400 otherwise
- **Translator**: zai thinkingFormat sends the top-level `reasoning_effort`
  object GLM-5.2+ requires — every GLM-5.x request ran at the model default
  (max); gated on GLM-5.2+ since older GLM does not read it (#2721)
- **RTK**: system prompt injection matches each target wire format
  (Chat/Responses/Claude/Gemini/Kiro) and is exact-idempotent across retries,
  so distinct prompts sharing a long prefix are no longer collapsed (#3202).
  Also set the diagnostic before the silent null return on Responses
  translation failure so the panel is no longer blank
- **OpenCode**: route muse-spark through /zen/v1/responses (it 500s on
  chat/completions), normalizing the Chat fields the Responses API rejects
  and clamping max/ultra effort to xhigh
- **CLI**: install better-sqlite3 without build tools on Node 22+ (N-API
  13.0.3 ships per-platform prebuilds, `--ignore-scripts` skips the implicit
  node-gyp build); Node < 22 stays on 12.6.2, working installs untouched
- **CLI tools**: send the API key Codex actually reads —
  `[model_providers.9router.http_headers]` instead of auth.json (which left
  every request 401 and clobbered an existing ChatGPT login); subagent model
  moved to `agents.default_subagent_model`
- **OAuth**: refresh Cline tokens with the extension JSON contract
- **Dashboard**: clamp the API key mask length — keys shorter than 8 chars
  threw RangeError and crashed the media-provider detail page
- **UI**: wait for the Material Symbols font itself before revealing icons —
  `document.fonts.ready` resolved before the 4MB woff2 even started loading,
  leaving icons blank until a second load
2026-08-29 17:59:36 +07:00
decolua
2203cd8f2b test(translator): drop the golden url/header snapshot
The committed snapshot had drifted from the registry: five providers
mismatched on a plain checkout and alitp-intl was missing entirely.
Remove it so the suite regenerates from the current registry.
2026-08-28 18:19:53 +07:00
decolua
2fd99eae5d fix(session): read Claude Code session id from its request header
Claude Code carries the session in metadata.user_id, which the Responses
API translation drops before the executor resolves a cache session. The
request then fell through to the assistant-text hash and the per-connection
fallback, so one conversation was split across several prompt_cache_key
values and the upstream prefix cache kept missing.

Fall back to the x-claude-code-session-id header, which survives every
translation. The body stays authoritative when both are present.
2026-08-28 18:17:55 +07:00
Agung Gunawnan
df85e16d7a fix(providers): time out connection tests and guard undefined names
Apply a 15s AbortSignal timeout in fetchWithConnectionProxy when the
caller supplies none, so provider connection tests stop hanging and
exhausting the browser socket pool. Also make matchSearch return false
for falsy provider names instead of crashing the providers page.
2026-08-28 17:06:36 +07:00
Ahoora5678
dff648496c fix(antigravity): sanitize competing-client branding in system prompts
Antigravity flags requests whose system prompt identifies another vendor's
client and answers 429 Quota Exhausted. Move the existing Zed/Claude prompt
rewrite into a config-driven rule table and add case-preserving opencode ->
antigravity mapping.

Applied in the executor so only Antigravity requests are rewritten - the
shared openai-to-gemini translator also serves gemini, gemini-cli, vertex
and zed, which must not be touched.
2026-08-28 17:01:18 +07:00
Paulo Schuller
88676b3037 fix(oauth): refresh Cline tokens with extension JSON contract 2026-08-28 16:58:27 +07:00
fasilu
bb3cb43e09 fix(dashboard): clamp API key mask length for short keys
"•".repeat(apiKey.length - 8) threw RangeError when the key was
shorter than 8 chars, crashing the media-provider detail page.
2026-08-28 16:53:14 +07:00
Óscar Fonseca
4a371d1d9f fix(usage): preserve nested cached_tokens in canonicalizeUsage
buildUsage() only emits cache reads under prompt_tokens_details, so the
top-level-only read dropped the count for every Responses-format provider
(codex, grok-cli, ...), persisting cached_tokens: 0 and billing cache hits
at the full input rate. Mirror the cache_creation fallback already used
just above.
2026-08-28 16:46:05 +07:00
alfep
d91e8b85e0 feat(antigravity): add Gemini 3.7 Flash tiers to MITM defaultModels
Registry/pricing/CLI catalog already had gemini-3.7-flash-{high,medium,low}
but MITM_TOOLS.antigravity.defaultModels was missing them, so the tiers
never showed up in the dashboard model-mapping table.
2026-08-28 16:44:07 +07:00
snower
993c6eb469 feat(headroom): make the compression request timeout configurable
The 3000 ms timeout on /v1/compress was fixed, so busy or slow machines
timed out often and sent the LLM an inconsistently compressed body,
hurting prompt caching. Add a headroomTimeoutMs setting, thread it from
the chat handler down to compressWithHeadroom, expose it in the Token
Saver dashboard, and normalize invalid values back to the 3000 ms default.
2026-08-28 16:34:33 +07:00
turingcat
28d005772a fix(minimax): preserve images on matched OpenAI transport
Prefer the sourceFormat-matched runtime transport over a model's
declared targetFormat when both apply. MiniMax-M3 previously resolved
to a Claude-shaped body while being posted to the already-selected
OpenAI endpoint, silently dropping image_url blocks from OpenAI
clients. Fixes #3418.
2026-08-28 16:34:05 +07:00
fasilu
2a9213c5bd feat(antigravity): map image size to aspect-ratio model suffix
Resolve body.size through sizeToAspectRatio and append the ratio as a
-WxH suffix so the executor's parseImageConfig picks it up. Also fall
back to gemini-3.1-flash-image when a non-image model reaches the
image handler.
2026-08-28 16:33:23 +07:00
decolua
ec6692808b fix(search): scope failure locks so search cannot take chat offline
Two problems on the credentialFallback path, where a search provider
borrows a chat provider's connection:

- the lock was attributed to the search provider id, but the connection
  belongs to the chat provider, so markAccountUnavailable looked it up
  under the wrong provider and read a stale backoffLevel
- with no model argument the lock key is `modelLock___all`, which
  isModelLockActive treats as blocking every model — one failing search
  would have taken the shared glm key offline for chat as well

Attribute the lock to the provider that owns the connection, and scope
it to `websearch:<provider>`, passed to getProviderCredentials too so
the lock is read back under the same key.
2026-08-28 16:18:46 +07:00
Amir Seify
e5a13c3ab7 feat(usage): show Zed plan quota on the dashboard
Add a Zed usage handler so connected Zed accounts appear on
/dashboard/quota. Reads GET /client/users/me for plan, edit
predictions, optional hosted model requests and billing-cycle reset.

Render unlimited rows as "N used · Unlimited" instead of 0 / ∞, and
surface overdue-invoice / token-billing messages.
2026-08-28 16:16:01 +07:00
Bertho Joris
67d9182e1a fix(executor): handle CommandCode in-stream errors for combo and account fallback
CommandCode returns errors as a type:"error" event inside an HTTP 200
NDJSON stream instead of a non-200 status, so the existing combo/account
fallback logic (keyed off response.status) never triggered and the error
text was streamed to the client as if it were content.

Peek the first NDJSON events before committing to a stream; on a
type:"error" event, abort and return a proper 4xx/5xx Response instead.
Normal streams are replayed losslessly (buffered prefix + rest of the
stream) through the existing translator, so the happy path is unchanged.
Add CommandCodeExecutor.parseError() so parseUpstreamError() can extract
a clean message/status from the synthesized error body.
2026-08-28 16:15:19 +07:00
decolua
9dbdca0e5e refactor(search): fold zai-search into the glm provider
The separate zai-search entry showed "No connections" on the web search
page because credentials live on the `glm` connection, not on it. Every
other provider that does both chat and search (antigravity, kimi, xai,
gemini) declares webSearch on the provider itself, so do the same here.

- glm gains serviceKinds ["llm", "webSearch"] and the MCP searchConfig
- the request builder / normalizer move from "zai-search" to "glm"
- drop the zai-search registry entry and its svg logo, which also
  removes the only need for svg logo support in getProviderIconSrc

ollama-search keeps its own entry and credentialFallback: its search
endpoint is unrelated to the ollama chat transport.
2026-08-28 16:12:04 +07:00
decolua
5a86f6a8d2 feat(search): add ollama-search and zai-search with credential fallback
Register two web search providers that reuse an existing chat provider's
API key instead of requiring their own connection:

- ollama-search (POST ollama.com/api/web_search) reuses the `ollama` key
- zai-search (POST api.z.ai MCP web_search_prime) reuses the `glm` key

A new `credentialFallback` registry field drives this: when a search
provider has no connection of its own, the search handler falls back to
the linked chat provider's credentials.

Also teach getProviderIconSrc to serve .svg logos for providers that
ship vector art.
2026-08-28 16:04:45 +07:00
huohua-dev
eb312bd470 fix(claude): decloak tool names in same-format streaming passthrough
translateResponse() short-circuited untouched on claude->claude streaming,
so OAuth-cloaked tool names (CLAUDE_TOOL_SUFFIX) leaked to the client and
every tool call was rejected as unknown. Add decloakStreamChunk(), the
streaming counterpart of decloakToolNames(), and call it on the same-format
path using the already-plumbed state.toolNameMap.
2026-08-28 15:40:40 +07:00
Daniel Gonçalves Araujo
fcfcced4ab fix(usage): support CREDIT_LIMIT and multi-interval GLM quotas
GLM quota parsing only accepted TOKENS_LIMIT and wrote every limit to a
single "session" key, so credit-based plans showed nothing and later
intervals overwrote earlier ones. Accept CREDIT_LIMIT too and derive the
quota key from the limit unit (5h session, 7d weekly, tokens, custom).
Moves the parser into its own usage/glm.js, re-exported from misc.js.
2026-08-28 15:35:23 +07:00
qingyong
56a40765e9 fix(translator): zai thinkingFormat sends reasoning.effort object
Z.ai / GLM-5.2+ require a top-level reasoning_effort (low/high/max)
alongside thinking:{type:"enabled"} to control reasoning depth; the zai
branch previously only set thinking and dropped reasoning_effort, so every
GLM-5.x request ran at the model default (max). Gate the field behind
GLM-5.2+ (thinkingEffortSupported in capabilities.js) since older GLM
(4.x, 5.0, 5.1, 5-turbo, 5v-turbo) do not read it, and map client levels
to the exact low/high/max values z.ai accepts.

extractThinking now checks reasoning_effort/reasoning.effort before the
thinking object so a client-supplied effort is not overwritten by
thinking:{type:"enabled"} mapping to mode:auto.

Fixes #2721
2026-08-28 12:32:41 +07:00
KunN-21
cadef6c4ff fix(rtk): make system prompt injection format-safe and idempotent
Caveman/Ponytail injection now matches each target wire format instead of
assuming an OpenAI-shaped body:

- Chat arrays append a text block; Responses arrays append input_text and
  create typed message items
- Claude inserts before the final cache-control block; Gemini preserves the
  snake/camel systemInstruction wrapper
- Kiro updates systemPrompt and its mirrored first-user prefix atomically,
  rolling back if the pair fails to converge
- Format label decides Claude/Gemini before the wire-shape sniff, since their
  bodies also carry messages[]/contents[] and Anthropic rejects a "system"
  role inside messages[]
- Delimiter-aware dedup makes injection exact-idempotent across retries, so
  distinct prompts sharing a long prefix are no longer collapsed
- Every write is fail-open on frozen or proxied bodies

Saver order and X-9Router-Token-Saver: off behavior are unchanged.

Fixes #3202.
2026-08-28 11:47:56 +07:00
anojndr
ab044e6d6d fix(opencode): route Muse Spark through the Responses API
muse-spark-1.2-contributor-free returned HTTP 500 on /zen/v1/chat/completions.
The model is only served by /zen/v1/responses, so route it there via a per-model
targetFormat and normalize the Chat fields the Responses API rejects
(max_tokens -> max_output_tokens, reasoning_effort -> reasoning{effort,summary}),
clamping max/ultra down to the highest effort the model accepts (xhigh).

Routing stays per-model: the other free models (big-pickle, hy3-free, mimo,
nemotron, laguna) are not served by /responses and keep /chat/completions.
2026-08-28 11:33:14 +07:00
decolua
14401c433c fix(ui): wait for the icon font itself before revealing Material Symbols
`document.fonts.ready` resolved before the 4MB Material Symbols woff2 even
started loading — it runs in <head>, ahead of any element that would trigger
the lazy fetch. The `fonts-loaded` class landed early, so icons rendered
blank until a second load served the font from disk cache.

Load the face explicitly and swap visibility for opacity, with a 3s fallback
so icons never stay hidden if the font fails.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-28 11:32:28 +07:00
warelik
e08ac6dada fix(tools): default Claude tool type when missing
Strict Anthropic-compatible gateways (e.g. MiniMax) reject Claude-format
requests with HTTP 400 when tools[].type is missing. Normalize each
missing/falsy tools[].type to "custom" before dispatch when the final
request format is Claude. Built-in tool types (computer_use, bash,
web_search_*) are passed through untouched.
2026-08-28 11:16:44 +07:00
Nguyen Thanh Dat
f9d82c6575 fix(stream): parse the trailing NDJSON line an Ollama stream leaves behind
createSSEStream splits on "\n" and keeps the remainder, which only flush()
parses. That call omitted targetFormat, so parseSSELine required a "data: "
prefix and dropped whatever an NDJSON provider left without a closing
newline. The !parsed.done guard compounded it: the SSE sentinel and an
Ollama final chunk both carry done:true, but the latter is the real last
chunk holding done_reason and the token counts.

Pass targetFormat and scope the sentinel check to formats that emit one, so
the tail reaches the translator. Accumulate its usage into state the same way
the transform loop does, so finalizeStream logs those tokens instead of null.
2026-08-27 20:53:02 +07:00
decolua
2f17352cc2 feat(search): add Antigravity as a web search provider
Route POST /v1/search with provider "antigravity" through Google Search
grounding on v1internal:generateContent, using the existing Antigravity
OAuth account pool. Grounding chunks become citations with the grounded
sentence as snippet and its surrounding answer text as content.

Upstream repeats a source across chunks, so citations are keyed by URL
and their snippets merged. A missing projectId is reported up front —
upstream answers a fabricated or absent project with a misleading
"no valid license" 403.

Based on the approach in #3437 by @Nautilaceae.
2026-08-27 20:48:43 +07:00
vianhanif
90a0005845 fix(cli): install better-sqlite3 without build tools on Node 22+
The runtime hook pinned better-sqlite3 12.6.2, whose prebuilds stop at
Node ABI 141 — on Node 26 the install fell back to a node-gyp source
build and failed on machines without build tools, silently degrading to
the sql.js fallback.

Node >= 22 now installs 13.0.3, which is N-API and ships per-platform
prebuilds inside the package. Two things were needed to make that
actually work:

- npm injects an implicit `node-gyp rebuild` for any package shipping a
  binding.gyp, so the install still demanded build tools; `--ignore-scripts`
  skips it and uses the bundled prebuild as-is.
- the binary check only looked at build/Release, which 13.x no longer
  creates, so every start re-ran npm install; it now also accepts
  prebuilds/<platform>-<arch>.node.

Node < 22 stays on 12.6.2 (13.x requires Node >= 22), and an existing
working install is left untouched either way.
2026-08-27 20:28:26 +07:00
Fábio A.
e79ae6e7c5 i18n(pt-BR): expand translation to 1132 terms
Add 144 missing pt-BR strings covering Usage, Endpoint & Key security
notices, 9Remote, Media Providers, Proxy Pools, Combo & Vision Adapter,
Token Saver, Agent Skills and Quota Tracker.
2026-08-27 20:06:10 +07:00
decolua
a68ada1c83 feat(cli-tools): share endpoint presets across every tool card
Each card kept its own copy of the localStorage preset logic inside
BaseUrlSelect, so an endpoint saved on one card was invisible to the
others until a reload, and a URL typed into the custom field was
forgotten the moment the card collapsed.

Move the store into cliEndpointPresets.js and publish a change event
so open cards resync live. Applying settings now remembers the
endpoint unless it matches a built-in option, and each card passes
its configured URL as currentUrl so BaseUrlSelect can preselect the
matching preset instead of always falling back to 127.0.0.1.
Deleting a preset falls back to the first real option rather than
clearing the field.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 18:54:44 +07:00
decolua
c4af43faa3 fix(stream): stop logging a disconnect for every completed Responses call
Responses-API clients (codex, droid) close the socket on
response.completed because the protocol has no [DONE] sentinel, so
every successful request printed "⚡ DISCONNECT: ResponseAborted"
after its own "📊 done" line. Keep the dbg("CTRL", …) trace and drop
the console line; ABORTED and ERROR still print.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 18:52:50 +07:00
decolua
d7f7d70dd5 fix(stream): record usage when a client closes on the terminal event
The Responses API has no [DONE] sentinel, so codex closes the socket
as soon as response.completed arrives. That cancels the reader before
flush() runs — and flush() held every usage side effect, so a fully
successful request logged nothing: no 📊 done line, no token stats,
no request detail.

Extract that tail into a once-guarded finalizeStream() and also call
it right after the terminal event is forwarded, in both passthrough
and translate mode. flush() still calls it; the guard makes the
second call a no-op. Streams that end normally are unaffected, and a
terminal event carrying no usage falls through to the existing
estimate/null path rather than blocking.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 18:52:42 +07:00
decolua
9c45b27cd7 fix(cli-tools): send the API key Codex actually reads
Codex only authenticates a custom model provider from env_key,
http_headers, env_http_headers or a token command — auth.json is
read solely by the built-in openai provider. Writing OPENAI_API_KEY
there left every request unauthenticated (401 Missing API key) while
clobbering an existing ChatGPT login.

Put the key in [model_providers.9router.http_headers] instead, and
drop the auth.json write. Also move the subagent model to the
agents.default_subagent_model scalar: agents.<role> now declares a
custom role and requires a description, so the old [agents.subagent]
table was discarded with a startup warning. DELETE still clears
auth.json to repair machines configured by the previous version.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 18:52:32 +07:00
decolua
e6f5724b4b fix(models): stop the catalog sync from erasing its own output
collectEntries() computed each model's "current" capabilities with the
previous catalog still installed, so every delta was measured against the
last one. An upstream value that still agreed with what we had written
looked like no change and was dropped: the second run cut `providers`
from 20 entries to 5, taking glm-5.3's 1M context correction with it.

The baseline has to be the hand-written tables alone, so the reader is
detached for the snapshot and restored in a finally — a mid-sync failure
must not leave capabilities.js without it.

Two smaller corrections:

- One vote per provider in the modality tally. Ids that normalize to the
  same model (claude-opus-4-thinking:1024, :8192, :32768 …) were each
  counted, giving nano-gpt five votes where other gateways had one. No
  model's result actually flipped — the variants agree with each other —
  but the majority rule only means something if the denominator does.
- Restore the etag from the file on startup. It lived only in module
  state, so every restart re-downloaded 4.3MB to be told nothing changed.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 18:44:30 +07:00
decolua
d01724556a fix(models): drop the worker thread from the catalog sync
The worker resolved its own path through import.meta.url, which the
bundler rewrites — so the running server looked for the file at a path
that does not exist there:

  [modelCatalog] sync failed: Cannot find module
  '/Users/Working/router4/9router/src/lib/modelCatalog/worker.js'

It was guarding against a 23ms JSON.parse that runs once a day, 60s after
boot. Inlining it into sync.js costs that 23ms on an otherwise idle tick
and removes both the failure mode and a whole file.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 18:39:44 +07:00
DarRahman
40eed18688 feat(usage): track GPT-5.3-Codex-Spark quota windows
Extract Spark rate limit windows from the Codex usage response and expose
them as spark_session/spark_weekly quotas, reusing the existing prefix
mechanism. Map codex quota types to readable dashboard labels.

Fixes #3431
2026-08-27 17:56:26 +07:00
decolua
0532f00d84 feat(models): refresh model capabilities from models.dev in the background
Capability tables are hand-maintained, so a model gains vision or a wider
context only when someone notices and edits the file. This adds a daily
sync that fills the gap for models already in the registry.

How it decides:

- Modalities (vision/pdf/audio/video) belong to the MODEL — every gateway
  serving glm-5.3-flash serves the same weights — so they are keyed by
  model id and shared. A majority of sources must declare one, which keeps
  out lone mis-declarations: minimax-m2.5 (1 of 45), glm-4.7 (1 of 44) and
  gpt-oss-120b (2 of 76) are text-only despite a reseller claiming vision.
- Context/output limits belong to the GATEWAY — each truncates differently
  (glm-5 ships as 202752/16384 on one host and 204800/131072 on another) —
  so they are keyed by provider + model and only the matching provider's
  own numbers are trusted.

Both layers are strictly additive and sit BELOW the hand-written tables,
which short-circuit first. A capability already true stays true.

Mechanics: worker thread (the 4MB parse would block the loop ~20ms),
ETag so an unchanged catalog costs one empty request, 60s startup delay,
30min backoff on failure, MODEL_CATALOG_SYNC=off to disable. Only the
~57KB delta is kept; lookups cost ~0.1us via an mtime-guarded cache.

capabilities.js is bundled into the browser through useModelCaps, so it
cannot import node:fs — the server injects the reader via
setCatalogSource() from instrumentation.

visionPatterns.js is the last resort: a model nobody has catalogued yet
still accepts images when its id says so (qwen3-vl-plus, glm-4.6v, llava),
with image-generation and embedding ids excluded.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 17:53:53 +07:00
decolua
9c650e1d54 feat(models): add GLM-5.3-Flash, DeepSeek V4 Vision, Grok 4.5/4.6
Vendors shipped four multimodal models the registry did not carry:

- glm-5.3-flash — z.ai's first natively multimodal GLM-5, 1M context,
  image + video + pdf input (glm, glm-cn, opencode-go)
- deepseek-v4-flash-vision-exp — image input at V4-Flash text parity,
  1M context / 384k output (deepseek, opencode-go)
- grok-4.6, grok-4.5 — 500k context; 4.6 has no text output limit (xai)

Capabilities needed hand entries because the existing globs mis-matched:
*glm-5* and *deepseek-v4* carry no vision, and *grok-4* would have capped
grok-4.6 at 256k instead of 500k. The grok-4.6 pattern sits above the
generic *grok-4* so it wins the first-match lookup.

Also corrects glm-4.6v / glm-4.5v, which were missing video input and
declared no maxOutput, and backfills glm-4.6v on glm-cn — zhipuai serves
it and the sibling provider already listed it.

tests/unit/opencode-go-models.test.js pins the opencode-go model list, so
its expected array moves with the registry.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 17:52:01 +07:00
Nim
1a3db1efae feat(antigravity): quota-aware routing with reset-aware fallback
On a 409/429 from Antigravity, fetch live quota to learn the exact
per-model resetAt instead of guessing a backoff, then skip only the
exhausted account/model pair until that time.

- antigravityQuota.js: in-memory quota cache, coalesced concurrent
  refreshes, 30s throttle per connection (applied to failures too),
  keeps known cache when upstream returns 401/403 error payloads
- auth.js: pre-filter exhausted account/model pairs; report the
  earliest quota reset when every account is blocked; skip the
  30-minute cooldown cap so the upstream resetAt is not truncated
- chat.js: antigravity 409/429 falls back on the RAM cache only, no
  persistent modelLock_* for this path
- Logs identify accounts by id prefix, never email or name

Closes #3561
2026-08-27 16:18:08 +07:00
kriptoburak
f0a6d35818 feat(search): add Xquik as an X search provider
Xquik needs a GET request with x-api-key auth and a tweets envelope
normalizer, neither of which the generic search fallback provides. Adds a
dedicated request builder and normalizer, cursor pagination passthrough,
result-based credit usage reporting, and a validateUrl probe so key
validation hits the no-charge credits endpoint.
2026-08-27 16:10:00 +07:00
vianhanif
548e32aacf fix(rtk): set diagnostic before silent null return on Responses translation failure
The openai-responses branch of compressWithHeadroom returned null without
recording a reason, leaving the diagnostics panel blank and making Codex
translation failures indistinguishable from a successful compression.
2026-08-27 15:52:46 +07:00
ariesho2903
abb20d9f39 feat(dashboard): bulk import Grok CLI accounts from JSON
Add a "Bulk Add" flow for the grok-cli provider, mirroring the existing
codex one: paste a JSON array/object or drag-drop multiple .json files,
then create all OAuth connections in a single call.

- BulkImportGrokCliModal: flexible JSON parsing (array, single object,
  {accounts:[...]}, concatenated objects) + multi-file upload
- POST /api/oauth/grok-cli/bulk-import: serial createProviderConnection,
  snake_case/camelCase token fields, email backfilled from id_token or
  access_token, authMethod "device_code" to match the login flow
2026-08-27 15:49:35 +07:00
86112cee6d feat(combos): drag-and-drop reorder with manual sortOrder
- schema.js: add sortOrder REAL column to combos, bump SCHEMA_VERSION 4->5
- combosRepo: persist sortOrder, append new combos to end, add reorderCombos() atomic reorder
- export/import DB: include sortOrder for round-trip
- API: PUT /api/combos/order accepts { ids: string[] } and persists new order
- UI: dnd-kit DndContext + SortableContext wrap flat list; drag handle (grip icon) on each ComboCard; optimistic local reorder with revert on failure

Grouped-by-tag view keeps server-side order; reorder is only available in the unfiltered flat list.
2026-08-27 14:21:25 +07:00
5c6048759c fix(db): add missing apiKeys columns + restore ApiExplorerModal export
- schema.js: add createdAt/allowedModels to apiKeys, bump SCHEMA_VERSION 3->4
- shared/components/index.js: restore ApiExplorerModal barrel export (was replaced by TagInput)
- combos: tag support, repo + api + page updates
2026-08-27 13:58:23 +07:00
d0f202a75d feat(commandcode): bump version header and expand model catalog 2026-08-27 10:34:09 +07:00
f0adfb205a feat(dashboard): per-key model restrictions, pin header routing, combo side-panel picker
- Endpoint: per-API-key model allowlist (schema v3) enforced on chat (403)
  and /v1/models; Full-access toggle + multi-select picker in Keys UI.
- Providers: honor x-connection-id in /v1/chat/completions — pinned requests
  no longer rotate to another account on failure.
- Providers: strategy saves merge into stored enabled:false override;
  Test All groups match grid sections; 1-by-1 skips disabled connections.
- Dashboard: provider-card toggle syncs from server on failure; grid toggles
  always visible; connection rows get clear-✕ for stale error banners.
- Combo editor: on desktop (xl+) the Add-Model picker opens as a floating
  side panel beside the untouched combo popup instead of stacking on top;
  mobile keeps the full-screen overlay.
- Long API-key overflow fixed in key rows + provider model sections.
2026-08-27 09:26:17 +07:00
1d56e2dbc5 chore(deps): bump dependencies 2026-08-22 14:34:02 +07:00
55f10c11e5 fix(dashboard): label usage-by-provider column as Provider / API Key 2026-08-22 14:34:02 +07:00
eedad6c5ea fix(combos): show effective strategy vs global default; keep explicit fallback override 2026-08-22 14:34:02 +07:00
99752a397c fix(usage): commandcode monthly total = consumed + remaining credits 2026-08-22 14:33:53 +07:00
bb8d67ba9c feat(caps): user-registered models trust upstream vision instead of stripping media 2026-08-22 14:33:53 +07:00
144dda2ac2 fix(routing): fall through to compatible node when built-in alias has no credentials 2026-08-22 14:33:53 +07:00
bc9719fac7 feat(dashboard): bulk enable/disable selected API keys on provider page 2026-08-22 14:33:20 +07:00
6770f6ba0b fix(dashboard): ReferenceError getProviderLabel in RecentRequests
getProviderLabel was a useCallback inside the UsageStats component, but
RecentRequests (a sibling module-level component) called it too, causing
"getProviderLabel is not defined" at runtime.

Extract a module-level resolveProviderLabel() (registry lookup by
id/alias/display prefix) used by RecentRequests; keep the component-scoped
getProviderLabel (which adds connected-provider nodeName/name lookup) for
the main table render paths.
2026-08-17 16:30:26 +07:00
c61dc6de46 fix(dashboard): show provider names instead of node ids in Usage by Provider
The Usage by Provider table (and related provider cells) rendered raw
provider keys, so requests routed through a custom OpenAI-compatible node
appeared as "openai-compatible-chat-096baf9a-6433-4f22-b079-20371092555a"
instead of the node's display name.

- UsageStats: add getProviderLabel() resolving a provider key (built-in id,
  alias, or custom node id) to a friendly name via the providers state
  (nodeName > connection name) and AI_PROVIDERS/getProviderByAlias fallback
- Apply the label in group headers, detail rows, badges, and recent requests
  (raw key kept as hover title)
- UsageTable: accept providerLabel prop, use it for provider group headers
2026-08-17 14:24:13 +07:00
b5c0f10610 feat(settings): restore per-provider connect timeout overrides
Re-apply the settings/UI layer of the per-provider timeout feature that
was dropped during the origin/master merge (core providerTimeout.js +
executor wiring survived; the settings keys and dashboard UI did not):

- settingsRepo: providerTimeouts:{} + defaultTimeoutMs:null defaults
- Profile page: "Default Connect Timeout" card (global fallback, ms)
- Provider detail page: per-provider "Connect Timeout" input, saved to
  providerTimeouts[providerId].timeoutMs, applied via
  resolveProviderTimeoutMs() priority: per-provider > global > registry > env
2026-08-17 09:49:44 +07:00
1256f29d92 Merge remote-tracking branch 'origin/master' into gitea/new_feature
# Conflicts:
#	open-sse/handlers/chatCore.js
#	open-sse/services/combo.js
#	src/app/(dashboard)/dashboard/profile/page.js
#	src/app/api/v1/models/route.js
#	src/lib/db/repos/settingsRepo.js
2026-08-17 00:21:51 +07:00
de9e00c66d feat(settings): runtime log level + free provider enable/disable
- Add LOG_LEVEL env + runtime setLogLevel (dashboard Settings → Logging),
  applied immediately, persisted across restarts; WARN/ERROR quiet production
  INFO lines (▶ POST / 📊 DONE / [COMBO] / [CHAT])
- Allow toggling free/noAuth providers (gemini-cli, kilo, etc.) off via
  providerStrategies.enabled from Providers page and provider detail page
- auth.js: honor disabled override before noAuth/connection branches
- CompatibleModelsSection: parallel model testing
2026-08-17 00:17:55 +07:00
decolua
699edac327 # v0.5.55 (2026-08-14)
## Features
- **Auth**: native SAML 2.0 SSO alongside OIDC — AuthnRequest generation, ACS
  assertion handling, SP metadata export, admin config test, replay-protected
  via a `saml_state` cookie matched against `InResponseTo`
- **Providers**: add Alibaba Token Plan (`token-plan.ap-southeast-1`) — the
  fourth Alibaba key type, Singapore-only and OpenAI-compatible transport only
- **Providers**: add `glm-5.3` to GLM Coding and GLM (China)
- **Providers**: Kimchi accepts API keys as well as OAuth (dual auth), with a
  working Test Connection for both modes
- **Antigravity**: add Gemini 3.7 Flash and its tiered high/medium/low variants
  (also in the Gemini registry) with pricing and quota tracking
- **TTS**: add Fish Audio — model id travels in an HTTP `model` header, voice
  is a `reference_id` (preset or cloned voice model)
- **OpenCode-Go**: route by request format via declared transports instead of
  forcing every client into `/messages` — Codex/OpenAI clients no longer pay a
  lossy Responses→OpenAI→Claude double translation. Per-model `supportedFormats`
  guard; the bespoke executor is gone (its shared `_lastModel` cache could cross
  auth headers between concurrent requests)
- **Usage**: dedup + cache Claude quota calls (120s TTL keyed by access token,
  in-flight promise dedup, last-good read on soft failure) to stop multiple
  tabs tripping 429; manual refresh (↻) sends `force=1` to bypass the cache

## Fixes
- **Docker**: ship `sql.js` in the image so the pure-JS DB fallback can start —
  file tracing carried the package's JS without `dist/sql-wasm.wasm`, so a
  container with no native driver aborted with ENOENT and never got a database
  (#3248)
- **Usage**: read Gemini `usageMetadata` out of the antigravity `{ response }`
  envelope — every non-streaming antigravity request logged `IN 0 | OUT 0`
  (#3260)
- **Claude**: re-anchor passthrough cache breakpoints — the client's own
  `cache_control` markers point at pre-normalization offsets, so the tail was
  re-cached every request. Last system block and last tool pinned at 1h TTL,
  last assistant turn at 5m, mid-conversation system messages folded into the
  neighbouring user turn instead of hoisted into `body.system`
- **Combos**: detect images from Hermes and attachment payloads (`images[]`,
  `experimental_attachments`, message-level `image_url`/`audio_url`, inline
  `data:` URIs) so the Vision Adapter auto-switch fires for Hermes/Ollama/
  Vercel AI SDK shapes
- **Kiro**: intercept chat via `x-amz-target` — Kiro IDE 1.0.228+ moved
  `GenerateAssistantResponse` to `POST /` + header, bypassing MITM. Also emit
  the now-mandatory initial-response frame and map the `auto` model slot
- **Kiro**: report real output tokens and stop discarding usable turns
- **Qoder**: detect billing blocks at stream start and return a synthetic 403
  so combo/account fallback triggers instead of leaking the error into chat
- **Antigravity**: strip competitive system prompts (Zed IDE's Claude-agent
  prompt) that Antigravity flags with a 429 Quota Exhausted
- **OpenCode**: send the official client fingerprint on free-tier requests so
  the Console stops classifying traffic as unidentified and rate-limiting it;
  session id resolves conversation-stable to preserve prompt caching
- **Responses**: don't close the message on an empty `tool_calls` array — some
  providers attach one to every chunk, and the truthy check ended the message
  on the first content token (#3234)
- **Translator**: preserve `prompt_cache_key` when converting chat to responses
- **Models**: expose snake_case token limits on `/v1/models`
- **Combos**: strip `stream_options` from the Fusion panel fan-out to avoid a
  DeepSeek 400 (#3024); raise the dashboard model-test probe budget to 1024 and
  soft-pass reasoning-only responses (#3010)
- **Headroom**: the toggle reflects the `headroomEnabled` setting even when the
  proxy is down — it previously showed OFF while the engine kept calling
  `/v1/compress`; proxy status stays visible via the status chip
- **Hermes**: add the `api_key` parameter to the model block in YAML config
- **Providers**: add llm7 to provider test support

## Docs
- **i18n**: add Spanish, French, and Brazilian Portuguese README translations

## Security
- **Real IP**: `x-9r-real-ip` and the Host fallback were trusted from
  client-controlled headers whenever `custom-server.js` was not in the request
  path (`npm run start`, `start:bun`), letting a remote caller pose as local to
  skip API key auth and reach `LOCAL_ONLY_PATHS` (`/api/mcp/*`,
  `/api/tunnel/enable`, `/api/auth/reset-password`). The server now stamps a
  per-process `x-9r-peer-token` on every request it sanitizes and only trusts
  `x-9r-real-ip` behind it — falling back to Host in development and failing
  closed in production (GHSA-pjm4-8fpg-f9p6). Also fixes IPv6 loopback
  detection (`::1`, `::ffff:127.0.0.1`) and routes `npm run start` /
  `start:bun` through `custom-server.js`
- **Search**: `resolveBaseUrl()` rejects client-supplied non-public baseUrls
  (SSRF guard on `/v1/search`)
- **Login**: fresh-install remote login with the default password returns 403
  without issuing a JWT
- **Usage**: `/api/usage/request-details` redacts request/response payloads
2026-08-14 17:08:02 +07:00
decolua
540ebbe682 test(baseline): regenerate provider snapshot for opencode-go transports 2026-08-14 16:53:04 +07:00
KiMelody
e1115e2839 feat(opencode-go): route by request format via transports + per-model guard
opencode-go hard-coded targetFormat: claude per model, so every client
format was force-routed to /messages (Codex/OpenAI clients paid a lossy
Responses->OpenAI->Claude double translation). Declare the existing
upstream multi-endpoint transports [openai, claude, openai-responses]
and guard per model via registry supportedFormats: kimi/glm/mimo only
support /chat/completions, minimax/qwen add /messages, deepseek adds
/responses. Undeclared models keep the upstream default.

Drop the bespoke OpenCodeGoExecutor (its shared _lastModel cache could
cross auth headers between concurrent requests); DefaultExecutor already
consumes runtimeTransport and injects reasoning content.
2026-08-14 16:52:37 +07:00
Nguyen Thanh Dat
27f3710c8b fix(docker): ship sql.js so the pure-JS DB fallback can start
Next file tracing follows JS imports, and sql.js loads dist/sql-wasm.wasm by
path at runtime, so the standalone output carries the package's JS without its
wasm binary. When both native drivers fail the last-resort adapter then aborts
with ENOENT on the missing binary and the container never gets a database.

The CLI bundle already guards this explicitly (build-cli.js step 3b,
ensureModuleInBundle("sql.js")); the image just never got the same treatment.
Copy the package the same way node-forge and next already are.

Fixes #3248
2026-08-14 16:40:53 +07:00
Nguyen Thanh Dat
59d858b639 fix(usage): read Gemini usageMetadata out of the antigravity response envelope
Antigravity and gemini-cli wrap their payload in { response: {...} }.
extractUsageFromResponse only tested top-level usageMetadata, so every
non-streaming antigravity request logged zero usage (IN 0 | OUT 0) and
zeroed rows in the usage dashboard. Read the envelope the same way
usageTracking.js and nonStreamingHandler.js already do; top-level
metadata keeps priority and the OpenAI/Claude branches are untouched.

Fixes #3260
2026-08-14 16:34:54 +07:00
Nguyen Thanh Dat
92259214db fix(security): require proof that x-9r-real-ip came from the socket (GHSA-pjm4-8fpg-f9p6)
x-9r-real-ip and the Host fallback were trusted from client-controlled
headers whenever custom-server.js was not in the request path (npm run
start, start:bun), letting a remote caller pose as local to skip API key
auth and reach LOCAL_ONLY_PATHS (/api/mcp/*, /api/tunnel/enable,
/api/auth/reset-password).

custom-server.js now generates a per-process secret at boot and stamps it
as x-9r-peer-token on every request it sanitizes. hasTrustedPeerHeaders()
(src/lib/auth/trustedPeer.js) gates trust in x-9r-real-ip on that secret;
otherwise the guard falls back to Host only in development, and fails
closed in production. Same gate on loginLimiter.getClientIp() so a spoofed
header cannot rotate the login lockout bucket.

Also: fix isLoopbackHostname for IPv6 (::1, ::ffff:127.0.0.1) which the
old split(":")[0] reduced to empty string; route npm run start /
start:bun through custom-server.js (postbuild copies it into
.next/standalone, build-cli.js fails without it) so documented deployments
keep passwordless local access.
2026-08-14 16:33:58 +07:00
Nguyen Thanh Dat
b04c03c6b5 feat(providers): add Alibaba Token Plan (token-plan.ap-southeast-1)
Fourth Alibaba key type — Coding Plan (alicode/alicode-intl) and Model Studio
(alims-intl) both reject Token Plan keys. Registry entry only; PROVIDER_MODELS
builds from providers/registry so no executor or translator work is needed.

Singapore-only (eu-central-1 answers IllegalEndpoint) and OpenAI-compatible
transport only (the Anthropic surface is not authorized for this plan).

Closes #2754
Closes #2806
2026-08-14 16:32:54 +07:00
AlexNoVibe
8b2b2fefb5 docs(i18n): add Spanish and French README translations
Add i18n/README.es.md and i18n/README.fr.md mirroring the English
README structure, and link both from the language switcher.
2026-08-14 16:29:43 +07:00
Azriel Akbar Ferry Ardiansyah Kusumawardhana
86694ed8d0 feat(antigravity): add Gemini 3.7 Flash models (#3286, #3281)
Add gemini-3.7-flash and its tiered high/medium/low variants to the
Antigravity and Gemini registries, with matching capabilities, pricing
and Antigravity quota tracking.

extractModel now recognises gemini-3.7-flash-tiered alongside 3.6 and
derives the version from the request, so thinkingLevel still maps to the
right tiered alias.

Closes #3286
Closes #3281
2026-08-14 16:27:10 +07:00
Nguyen Thanh Dat
8af5e752da feat(tts): add Fish Audio as a text-to-speech provider
Registry entry plus one config-driven FORMAT_HANDLERS handler. The model id
travels in an HTTP `model` header rather than the JSON body, and the voice is
a reference_id (preset or cloned voice model).

Closes #2411
2026-08-14 16:21:11 +07:00
zmf
8ed9da7165 feat(providers): add glm-5.3 to GLM Coding and GLM (China) registries
Zhipu released GLM-5.3 on both api.z.ai and open.bigmodel.cn coding
endpoints. Verified live against both, returning model:"glm-5.3" with
native reasoning_content.

No other changes needed: the '*glm-5*' family pattern in capabilities.js
and 'glm-5*' in pricing.js already cover it.
2026-08-14 16:16:46 +07:00
decolua
7e5f5a8813 fix(claude): re-anchor passthrough cache breakpoints with 1h TTL
Passthrough kept the client's own cache_control markers, which point at
pre-normalization offsets. Once normalize/dedupe reshaped system and tools,
the breakpoints landed mid-array and the tail was re-cached every request.

- Pin the last system block and last tool at ttl 1h (was the client's 5m)
- Anchor the last assistant turn at 5m, falling back to the final message
  so a first turn still gets a breakpoint
- Fold mid-conversation system messages into the neighbouring user turn
  instead of hoisting them into body.system, where the volatile token
  counters invalidated the prefix on every request
- Run the anchoring after every token saver, at the final body

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-14 16:08:30 +07:00
Bertho Joris
345cdcf6a5 fix(combo): detect images from Hermes and attachment payloads for Vision Adapter
Inspect images[], experimental_attachments/attachments, message-level
image/image_url/audio_url, and inline data:image|audio|pdf URIs on trailing
user turns so Vision Adapter auto-switch fires for Hermes/Ollama/Vercel AI
SDK shapes. stripOpenAI now also drops msg.images and image attachments when
the active model lacks vision support.
2026-08-13 18:30:44 +07:00
Duc Nguyen
65197ad11c feat(auth): add native SAML 2.0 SSO integration
Add SAML 2.0 as a second SSO protocol alongside OIDC under a unified
authMode/ssoType model. SP flows via @node-saml/node-saml: AuthnRequest
generation, ACS POST assertion handling, SP metadata export, and admin
config test endpoint. Replay-protected via saml_state cookie (httpOnly,
SameSite=Lax) matched against InResponseTo; wantAssertionsSigned enforced.

- src/lib/auth/saml.js: SAML instance builder, X.509 cert formatter, claim pickers
- 4 routes under src/app/api/auth/saml/: start, acs, metadata, test
- settingsRepo: ssoType + saml* defaults; login/status routes dispatch by type
- profile page: SSO protocol switcher, IdP metadata XML + cert uploaders
- login page: dynamic SAML sign-in button; Header: SAML user badge
2026-08-13 17:56:34 +07:00
Fadjrir Herlambang
e02bde4a70 feat(providers): add Kimchi API key support (dual OAuth + API key)
Kimchi's transport is OpenAI-compatible (Authorization: Bearer) but the
registry declared it OAuth-only, so the dashboard, /api/providers, and
the connection test all rejected API keys. Enable dual auth
(authModes: ["oauth", "apikey"]) and add a kimchi case to
testApiKeyConnection so the Test Connection button works for both modes.
Regenerate the golden snapshot with the Kimchi entries (+ other
previously-missing providers).
2026-08-13 17:53:44 +07:00
haumanto
30fec4318e fix(models): expose snake_case token limits on /v1/models 2026-08-13 12:19:06 +07:00
CyrixJD115
67271d859e fix(opencode): send official client headers on free-tier requests
Mirror the official opencode CLI fingerprint (User-Agent, x-opencode-session, x-opencode-request, x-opencode-project) on free-tier requests so the Console no longer classifies traffic as an unidentified client and rate-limits it with FreeUsageLimitError / HTTP 429.

Session id resolves conversation-stable via resolveSessionId (client session to assistant-text hash to connection) to preserve prompt caching, normalized into opencode ses_ format with a generated fallback. When the downstream client is already opencode, its headers are forwarded as-is.
2026-08-13 12:13:51 +07:00
stoXmod
b566b20ade fix(antigravity): strip competitive system prompts to prevent 429 quota errors
Zed IDE injects a Claude-agent system prompt that Antigravity flags as
competitive, blocking the request with a 429 Quota Exhausted response.
Scan systemInstruction.parts and remove the prompt before dispatch.
2026-08-13 12:11:30 +07:00
Clayton Tavares
6d30ce6de5 fix: Fusion strip stream_options + reasoning model test probe
- combos: strip stream_options from Fusion panel fan-out to avoid DeepSeek 400 (#3024)
- dashboard: raise model-test probe budget to 1024 + soft-pass reasoning-only responses (#3010)
2026-08-13 11:56:43 +07:00
rm1dev
5b417f9bf2 fix(kiro): intercept chat via x-amz-target and prepend initial-response frame
Kiro IDE 1.0.228+ moved GenerateAssistantResponse from path
/generateAssistantResponse to POST / + x-amz-target header, so chat turns
bypassed MITM. The SmithyMessageDecoderStream also now requires an
initial-response frame at stream start, and agent/vibe mode sends
modelId "auto" which had no mappable slot.

- Add isChatRequest() header-based match for kiro in mitm/config.js
- Add buildInitialResponseFrame/withInitialFrame to emit the mandatory
  initial-response once per stream (kiro.js)
- Add "auto" model slot and update mitmDomain to runtime.us-east-1.kiro.dev
2026-08-13 11:56:31 +07:00
Cokky Turnip
b57c041345 fix(providers): add llm7 to provider test support 2026-08-13 11:56:06 +07:00
zmf
8a527fec91 fix(security): SSRF guard on search baseUrl, default-password remote login, and request-details redaction
- resolveBaseUrl() rejects client-supplied non-public baseUrls via assertPublicUrl (SSRF guard on /v1/search)
- fresh-install remote login with default password returns 403 without issuing a JWT
- /api/usage/request-details redacts request/providerRequest/providerResponse/response payloads
- declare chalk and prop-types in package.json (used but previously undeclared)
2026-08-13 11:50:25 +07:00
Nguyen Thanh Dat
70ba0024b0 fix(translator): preserve prompt_cache_key when converting chat to responses 2026-08-13 11:46:49 +07:00
brimob-sowax
80afb59907 fix(qoder): detect billing blocks at stream start, return 403 for failover
Peek the first SSE frame in wrapQoderSSE; if statusCodeValue != 200 and the
body carries a billing signature (code 112/10605 or pricingUrl), return a
synthetic 403 so chatCore marks the connection unavailable and triggers
combo/account fallback instead of leaking the error text into chat.

wrapQoderSSE becomes async; consumed peek bytes are re-processed in the
stream start() seed loop so nothing is dropped.
2026-08-13 11:43:11 +07:00
chisewaguri
10a923da11 fix(responses): don't close message on empty tool_calls array
Some providers (e.g. codebuddy/cbcn) attach an empty tool_calls array to every streaming chunk. An empty array is truthy in JS, so the guard 'if (delta.tool_calls)' closed the message on the first content token and emitted response.output_text.done early, dropping the remaining deltas. Guard on a non-empty array; finish_reason still closes the message and real tool calls still close it before emitting function_call items.

fixes #3234
2026-08-13 11:40:45 +07:00
yusei21
01858feca0 docs(i18n): add Brazilian Portuguese documentation 2026-08-13 11:40:26 +07:00
Moein Arabi
e2a4fe048f fix(hermes): add api_key parameter to model block in YAML configuration 2026-08-13 11:35:02 +07:00
nguyenha935
b44bb09f72 fix(kiro): report real output tokens and stop discarding usable turns 2026-08-13 11:33:41 +07:00
decolua
456f2a2635 feat(usage): wire force flag through client + usage route
Manual refresh (↻) sends ?force=1 so it bypasses the Claude quota cache (dedup + TTL) added in cd4003bc. Auto-refresh and multi-tab stays cached, so Anthropic's usage endpoint is no longer hammered.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-13 11:31:07 +07:00
decolua
cd4003bc8b feat(usage): dedup + cache Claude quota calls to avoid 429
Multiple tabs/accounts/auto-refresh funneled straight to Anthropic and tripped 429. Add a 120s TTL cache keyed by access token with in-flight promise dedup, serve the last good read on soft failure, and thread a force flag through getUsageForProvider for manual refresh. Also lower the dashboard poll cadence (180s to 600s) and stable group-by-provider so connection order stops jumping.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-13 11:27:57 +07:00
decolua
71dcdc1053 fix(headroom): toggle reflects enabled setting even when proxy is down
Toggle was checked={headroomEnabled && headroomRunning} and disabled when the proxy was down, so a downed proxy showed OFF while headroomEnabled stayed true in the DB. The engine only checks headroomEnabled, so it kept calling /v1/compress. Toggle now reflects the user setting; proxy up/down stays visible via the status chip.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-13 11:27:49 +07:00
a3182a7265 merge: integrate origin/master (v0.5.50) into gitea/new_feature
- Resolve conflicts in chatCore handlers: keep apiKey/streamErrorPatterns
  from the details-filters feature, adopt origin's stripContinuityFields,
  customToolNames, cache-inclusive usage accounting, and Responses-API
  SSE→JSON conversion
- Adopt origin's provider usage handlers (codebuddy-intl, qoder creds)
  and modality detection (audio/video inputs)
- Keep requestDetails apiKey column (schema v2) + masked key persistence

Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
2026-08-06 09:54:31 +07:00
386b25ff7f feat(dashboard): add model/status/account/api-key filters to usage details tab
- Add apiKey column to requestDetails (schema v2 + migration 002)
- Persist masked API key via buildRequestDetail across chatCore handlers
- Add getRequestDetails apiKey filter + distinct models/apiKeys/statuses helpers
- New /api/usage/filters endpoint returning grouped connections per provider
- Group account dropdown by provider using <optgroup>; add Status column
  with success/error badge to the details table

Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
2026-08-06 09:41:03 +07:00
decolua
15223724c3 # v0.5.50 (2026-08-05)
## Features
- **Providers**: add TokenRouter (300+ models via OpenAI-compatible gateway) with
  exact per-model pricing for 110 models and `reasoning_effort` thinking config
- **Providers**: add Self-hosted STT / TTS / Embedding — point 9Router at your own
  OpenAI-compatible speech and embedding servers (whisper.cpp, faster-whisper,
  Kokoro-FastAPI, llama-server, vLLM, Infinity). Unlike the named cloud providers
  these read `baseUrl` per connection, so one provider can front several machines
- **Combos**: default-enable vision/audio capacity adapter (auto-routes to a
  vision/audio-capable model when the target lacks that capability, falling back
  to `oc/mimo-v2.5-free`), wired into chat handler routing
- **Endpoint**: auto-provision a "Default Key" for first-time users so `/v1`
  works without a manual dashboard step
- **Codex**: support GPT-5.6 Max/Ultra reasoning-level overrides (cx/ routes only)
- **Qoder**: support PAT (Personal Access Token) connections end-to-end, alongside
  OAuth device flow
- **CLI tools**: add OpenDesign (manalkaff/opendesign) support
- **Headroom**: report effective payload savings (tool schema/history bytes broken
  out, byte-savings % reflects actual outbound reduction)
- **Ollama**: Cloud quota tracker (session + weekly) + proactive background OAuth
  token refresh scheduler for all providers

## Fixes
- **Providers**: remove Qwen (OAuth flow stopped working reliably)
- **Passthrough**: detect codex-tui/Codex Desktop as native Codex client — they
  were falling through to the translator and losing fields like `reasoning.summary`
- **OAuth**: scope antigravity header fixes to loadCodeAssist/onboardUser only
- **OAuth**: keep `open` external in the build so xAI/Grok token refresh works on
  Windows
- **OAuth**: declare missing `searchParams` in register-session handler (was a
  500 instead of JSON on error)
- **DB**: `ENABLE_REQUEST_LOGS` env var now overrides the UI setting correctly;
  observability defaults to off (opt-in)
- **Translator**: preserve Codex Responses Lite tool use across chat-native
  OpenAI-compatible providers
- **Translator**: don't drop image-only user messages in `prepareClaudeRequest`
- **Translator**: drop JSON Schema keywords Gemini rejects (`uniqueItems`,
  `contains`, `multipleOf`, `unevaluatedProperties`, `unevaluatedItems`,
  `contentSchema`)
- **Claude**: remove global header cache that leaked one client's identity
  headers onto another client/account sharing the server; gate `anthropic-beta`
  by model instead
- **Antigravity**: drop retired Gemini 3.0 quota tiers, show Gemini 3.6 Flash
  usage bars
- **Cloudflare AI**: declare API key authentication (dashboard showed "No
  connections" despite an active key)
- **GitHub Copilot**: hold monthly-exhausted accounts until UTC month reset
  instead of only cooling down 120s
- **CodeBuddy**: dodge Tencent CN content filter, add usage tracking, normalize
  codebuddy-intl messages
- **Usage**: stop losing cached prompt tokens in the forced-SSE→JSON path
- **Grok CLI**: display the public subscription tier from the OAuth token claim
- **Providers**: count apikey connections for Ollama free-tier card; free-tier/
  apikey providers without `authModes` now default to apikey (were treated
  oauth-only)
- **Build**: include static/public assets in standalone output (login page hung
  on 404s when run via PM2)
- **Server**: support IntelliJ IDEA OpenAI-compatible clients over HTTP (h2c
  upgrade handling)
- **Auth**: redirect already-logged-in sessions away from `/login`
- **CLI tools**: enable Apply button for dynamic OpenAI/Anthropic-compatible
  provider connections
- **CLI**: include complete API artifacts in the CLI package
- **TTS**: a bare self-hosted model name is the MODEL, not the voice — `kokoro`
  was parsed as a voice against a default model, 404ing or synthesising with the
  wrong one
  endpoint that drops packets never returns headers, so the request previously
  hung indefinitely
2026-08-05 16:49:14 +07:00
decolua
35f86e5828 fix(oauth): scope antigravity header fixes to loadCodeAssist/onboardUser only
Google fingerprints User-Agent/Client-Metadata on loadCodeAssist and
onboardUser, silently refusing to provision a cloudaicompanionProject
when they don't match the real IDE. Split antigravity's headers out of
the shared gemini-cli constants instead of overwriting them, so the fix
doesn't touch gemini-cli or any other provider.

Inspired by #3000 (thanks @stoXmod for flagging the resource-exhausted
issue), rewritten to keep gemini-cli untouched.
2026-08-05 16:40:09 +07:00
Dasep Moch Luay
41588bea01 feat(providers): add TokenRouter accurate pricing + thinking config
Adds exact per-model rates for 110 TokenRouter models (pulled from
TokenRouter's own pricing API) plus a dedicated thinkingFormat case
(reasoning_effort enum low/medium/high/xhigh/max) and the provider
logo. Provider registration itself already landed in a prior commit;
this fills in what PR #3043 added on top.
2026-08-05 16:31:03 +07:00
decolua
03f8487cc7 test(baseline): regenerate provider/alias snapshots
Sync alias-baseline.json and providers-baseline.json with the current
registry (poolside, tokenrouter, selfhosted-* providers already added;
stale claudeOverlay hook and Kiro X-Amz-Target header already removed).
2026-08-05 16:27:17 +07:00
decolua
99639c0540 test(capacity-adapter): remove unit test file 2026-08-05 16:25:47 +07:00
decolua
e41d85037d test(capacity-adapter): add unit coverage for the capacity adapter service
Covers pool flattening, model augmentation for required capabilities,
context-window history stripping, and the withCapacityAdapterStripping
wrapper.
2026-08-05 16:25:12 +07:00
decolua
02c66fe2bd feat(endpoint): auto-provision default API key for first-time users
- Endpoint page auto-creates a "Default Key" when no keys exist yet,
  so /v1 works out of the box without a manual dashboard step
- Show/copy key buttons stay visible instead of opacity-0 by default
2026-08-05 16:22:06 +07:00
decolua
dcdd4628b3 fix(providers): remove Qwen provider support
Qwen OAuth flow (portal.qwen.ai) stopped working reliably; drop the
executor, registry entry, OAuth provider/service, token refresh
profile, usage handler, and related test coverage and baselines.
2026-08-05 16:17:26 +07:00
decolua
6498b3122f feat(combos): wire capacity adapter into chat handler routing
- detectRequiredCapabilities: infer audioInput/videoInput from block
  type and embedded mime, not just vision/pdf
- handleChat / handleSingleModelChat: augment combo and single-model
  routing with capacity-adapter models when the target lacks a
  required capability, wrapped with history stripping for the
  adapter model's context window
2026-08-05 16:12:55 +07:00
decolua
8e59093db7 feat(combos): default-enable vision/audio adapter with mimo fallback
- Enable vision + audioInput capacity-adapter pools by default for new
  and existing users (mergeWithDefaults backward-compat)
- Fall back to oc/mimo-v2.5-free when an enabled pool has no models
  configured, both in the backend resolver and the combos UI (auto
  refill on removing the last model from a pool)
- Hide PDF/Video from the Vision Adapter UI (PDF never implemented,
  Video lacks translator support) while keeping the settings shape
- Exclude combos from the model picker when opened from the Vision
  Adapter section
- mimo-v2.5 registry entry now declares audioInput/videoInput
- Simplify combo strategy and Vision Adapter descriptions
2026-08-05 16:09:42 +07:00
decolua
cd13d904d7 fix(passthrough): detect codex-tui/Codex Desktop as native Codex client
detectClientTool only matched the legacy "codex-cli" User-Agent, so the
current codex-tui CLI and Codex Desktop (UA "Codex Desktop", originator
"codex_work_desktop") fell through to null and lost native passthrough —
their requests got re-translated, stripping/overwriting fields like
reasoning.summary instead of forwarding the client body as-is.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-05 15:53:20 +07:00
omar-nahhas
fe547f4dc0 feat(providers): self-hosted OpenAI-compatible STT, TTS and embedding providers
Add Self-hosted STT/TTS/Embedding providers that read baseUrl per connection
instead of a fixed registry endpoint, so 9Router can point at whisper.cpp,
faster-whisper, Kokoro-FastAPI, llama-server, vLLM, Infinity, and similar
OpenAI-compatible local servers.

Self-hosted Embedding refuses to run without a baseUrl rather than falling
back to api.openai.com like openaiCompatNode does, since that fallback would
silently send input text and the API key to OpenAI under a provider named
"Self-hosted". Also fixes embeddingsCore to catch adapter build errors as a
400 instead of letting them escape uncaught, and bounds the upstream fetch
with FETCH_CONNECT_TIMEOUT_MS to avoid hanging forever on a dead endpoint.

Self-hosted TTS treats a bare model value as the model rather than the voice,
since the generic OpenAI TTS convention (bare = voice) is backwards for a
provider where the model is the variable part.
2026-08-05 13:38:13 +07:00
Ubuntu
b480892952 feat(providers): add TokenRouter provider
OpenAI-compatible gateway exposing 300+ models (OpenAI, Claude, Gemini,
Qwen, DeepSeek, Kimi, GLM, and more). Registered as p116, append-only.
2026-08-05 13:32:46 +07:00
RobertsXML
3fab15ae3e fix(db): implement ENABLE_REQUEST_LOGS env var override
- Add config priority chain: ENABLE_REQUEST_LOGS > UI setting > OBSERVABILITY_ENABLED fallback
- Fix transaction callback syntax from arrow to function
- Update saveRequestDetail guard to early return instead of semicolon
- Default enableObservability to false (opt-in)
2026-08-05 13:30:28 +07:00
nguyenha935
d06e0d26c6 fix(translator): preserve Responses Lite tools across Chat providers
Codex Responses Lite clients routed to a chat-native OpenAI-compatible
provider lost tool use in three places: non-streaming Chat responses
leaked the raw chat.completion envelope instead of Responses output
items, internal reasoning continuity fields leaked into the outbound
Chat body causing some upstreams to reject the request, and the
Responses to Chat request translator ignored additional_tools,
custom_tool_call, and custom_tool_call_output items entirely.

Also fixes apiType (chat vs responses) for openai-compatible nodes
being resolved from the immutable provider ID instead of the stored
node config, so editing a node's API Type had no runtime effect.
2026-08-05 13:27:25 +07:00
decolua
b11be8be0a fix(antigravity): drop retired Gemini 3.0 tiers from quota tracker
gemini-3-flash-agent, gemini-3-flash, and gemini-3-pro-image are retired;
they no longer need a quota bar in the tracker.
2026-08-05 13:24:57 +07:00
whale9820
42c691b3ea feat(antigravity): show Gemini 3.6 Flash usage bars in quota tracker
getAntigravityUsage filtered the fetchAvailableModels response through a
hardcoded importantModels list that only contained 3.5 Flash, silently
dropping the 3.6 Flash quota buckets so no usage bar rendered.
2026-08-05 13:21:38 +07:00
cardinusantara
a7941ddab4 fix(translator): don't drop image-only user messages in prepareClaudeRequest
hasValidContent() only treated text/tool_use/tool_result blocks as valid
content, so a user message containing only an image block was filtered
out as empty. When it was the only non-system message, this left an
empty messages array and Anthropic rejected the request.
2026-08-05 13:20:27 +07:00
Sutarto Jordan Chrisfivo
646b3b9ba3 fix(cloudflare-ai): declare API key authentication
Cloudflare AI registry entry was missing authType/authModes, causing
the dashboard to report "No connections" despite an active API-key
connection. Closes #2969
2026-08-05 13:13:54 +07:00
huuanh20
baebc9a06e docs(i18n): fix port typo and add RTK Token Saver features
Fix port 201281 -> 20128/v1 typo in README.zh-CN.md diagram, add
RTK Token Saver mention to README.vi.md/README.zh-CN.md, update
Tier 3 free providers to 2026 lineup, and drop hardcoded /tmp
node_modules path from tests/package.json and tests/README.md.
2026-08-05 11:55:02 +07:00
Matt Van Horn
25e4bf1c6c fix(cli): include complete API artifacts in CLI package
Merge the complete generated .next-cli-build/server tree into the
packaged CLI after the standalone copy, since Next's standalone output
is trace-pruned and can omit route modules (e.g. /api/v1/messages) or
chunks loaded dynamically. Add a post-copy integrity check for the
required API route artifacts so an incomplete package fails during
pack:cli instead of at runtime.

Fixes #2945
2026-08-05 11:54:16 +07:00
MiQieR
c570fe33ae feat(tts): add Xiaomi MiMo text-to-speech support
Adds mimo-v2.5-tts as a Media Provider TTS through the existing
OpenAI-compatible chat-completions endpoint. Voice is selected via the
top-level audio.voice field, and an optional style/language hint is
threaded through tts.js -> ttsCore.js -> the new adapter.
2026-08-05 11:46:23 +07:00
ryanngit
d0751bcff7 fix(grok-cli): display public subscription tier
Map the Grok OAuth access-token tier claim to the public plan label
and prefer it over internal /user entitlement names. Fails open to
existing plan detection for opaque or malformed tokens; upstream
remains authoritative for access and quota enforcement.
2026-08-05 11:43:40 +07:00
alfep
948dd8f89b fix(oauth): declare searchParams in register-session POST handler
Missing declaration caused a ReferenceError -> 500 HTML response instead
of JSON when clients called POST .../register-session.
2026-08-05 11:41:10 +07:00
seakleang.nhak
86131b9ca4 feat(codex): support GPT-5.6 Max and Ultra overrides
Add "ultra" reasoning level for Codex GPT-5.6 Sol and Terra, and expose
Max for Luna (Luna falls back Ultra to Max since it is not supported
upstream). Scoped to cx/ routes only; Kiro and generic OpenAI routing
unchanged.
2026-08-05 11:39:59 +07:00
Diwak4r
651df2f0e2 feat(cli-tools): add OpenDesign (manalkaff/opendesign) support
Adds a guide-type CLI Tools entry for OpenDesign, the open-sourced
claude.ai/design skills pack. It has no standalone config - it inherits
the host agent's model/provider config - so once the host (Claude Code,
Cursor, Codex, Gemini CLI, OpenCode) points at 9Router, /opendesign
sessions route through automatically.
2026-08-05 11:36:37 +07:00
minhnhat166
da8691f866 feat(headroom): report effective payload savings
Break out tool schema and tool-history bytes in the size snapshot and
add an effective byte-savings percentage so token-saved logs reflect
the actual outbound payload reduction, not just processed content.
2026-08-05 11:32:38 +07:00
decolua
13ed14568d fix(claude): remove global header cache, gate anthropic-beta by model
The global claudeHeaderCache singleton overlaid the last-seen Claude Code
client's identity headers onto every subsequent request, leaking one
client's headers (anthropic-beta, user-agent, x-stainless-*, etc.) onto
another client/account sharing the same server. Removed the singleton and
the claudeOverlay hook entirely, falling back to static per-provider
headers. anthropic-beta is now computed per-request from the requested
model, gating heavy-agent flags (advanced-tool-use, effort) to
opus/sonnet only.
2026-08-05 11:32:14 +07:00
decolua
1eb37db32d refactor(qoder): dedupe PAT exchange logic, validate PAT keys properly
PAT-to-job-token exchange was duplicated between the executor and the model service, each with its own cache. Consolidate into qoderModels.js and have the executor import it.

Also add a qoder case to the API-key validate route - the generic OpenAI-compat probe cannot validate a PAT (needs job-token exchange + COSY signing first), so bulk-add always reported unknown for qoder keys.
2026-08-05 11:26:22 +07:00
mannnrachman
d433c0b295 feat(qoder): support PAT (Personal Access Token) connections end-to-end
Adds pt-... token auth as an alternative to OAuth device flow. A PAT can't
sign COSY requests directly, so it's exchanged for a short-lived job token
(jt-...) plus userId via openapi.qoder.sh, then used for signing.

Also fixes job-token traffic (jt-...) being rejected by api3.qoder.sh with
403 "Login expired" — the official qodercli serves jt- traffic from
api2.qoder.sh instead, so buildUrl/model-list routing now branches on it.

Quota usage and the dashboard add-key modal are updated to resolve PAT
credentials and label the field correctly, and bulk-add now validates
each key so it gets a real testStatus instead of a hardcoded "unknown".
2026-08-05 11:16:50 +07:00
ryanngit
3292dfc102 fix(github): hold monthly-exhausted accounts until reset
Lock GitHub Copilot connections account-wide until 00:00 UTC on the
first of next month when the upstream 402 response indicates the
monthly additional-usage-limit was hit, instead of only cooling down
the requested model for 120s. Other GitHub 402 responses keep the
existing model-scoped cooldown.
2026-08-05 11:00:40 +07:00
Rafi Mahardika
9138c99391 fix(codebuddy): dodge Tencent filter for CN, add usage tracking & normalize messages for INT
Neutralize CLI-agent system prompts that trigger CodeBuddy CN's content filter, add usage/quota tracking for codebuddy-intl sharing CN's logic, and normalize codebuddy-intl request messages to the shape it expects.
2026-08-05 10:52:03 +07:00
omar-nahhas
41606a37a3 fix(usage): don't lose cached tokens in the forced-SSE->JSON path
handleForcedSSEToJson dropped cached prompt tokens in two ways: the
Responses branch summed only input_tokens, which excludes cache_read
and cache_creation on cache-capable upstreams (measured 2012 reported
vs ~5344 actual, 5332 from cache); and the Chat Completions branch
computed usage correctly but it didn't always reach the client (an
Anthropic response with cache_read_input_tokens: 11022 arrived with no
usage field at all). Now folds cache counters into prompt_tokens,
surfaces them via prompt_tokens_details, and re-attaches usage before
serialisation.
2026-08-05 10:45:55 +07:00
Muhammad Usama
2abe8b855c fix(translator): drop JSON Schema keywords Gemini has no field for
Tool schemas carrying uniqueItems, contains, multipleOf,
unevaluatedProperties, unevaluatedItems, or contentSchema get rejected
by the Gemini API with "Unknown name ...: Cannot find field", failing
the whole request. Add them to UNSUPPORTED_SCHEMA_CONSTRAINTS alongside
the existing stripped keywords (minItems, maxItems, format, ...).
2026-08-05 10:42:54 +07:00
Tomauskasz
c06cc08453 fix(oauth): keep open external so xAI/Grok token refresh works on Windows
`open` derives its own directory from import.meta.url at module scope.
Webpack replaces that with the build machine's absolute path as a
string literal, so a release built on macOS ships a file:///Users/...
URL that fileURLToPath rejects on Windows (no drive letter), throwing
on import. refreshXaiToken dynamic-imports the xAI OAuth service, which
imports open eagerly, so every Grok token refresh silently failed and
was swallowed by a catch that only logs a warning.

Add open to serverExternalPackages so it keeps its real import.meta.url
at runtime, and bundle it into the CLI package via ensureModuleInBundle
(same guard already used for sql.js) since externalizing it means
webpack no longer traces/copies it automatically.
2026-08-05 10:38:25 +07:00
Cokky Turnip
d6df6576c5 fix(providers): count apikey connections for ollama freeTier provider
ollama's registry entry lacked authModes, so dualAuthTypes on the providers page defaulted to oauth and its apikey connections showed as No connections on the freeTier card.
2026-08-05 10:34:44 +07:00
DaDecky
786b3013ba fix(build): include assets in standalone output
With output: "standalone", next build writes server.js under
.next/standalone but leaves generated static/public assets in the
project root, so starting the standalone server directly (e.g. via PM2)
404s on JS/CSS/font/favicon requests and /login stays stuck loading.
Add a postbuild step that copies .next/static and public into the
standalone directory, skipping the workspace-traced CLI build which
already copies its own assets.
2026-08-05 10:32:55 +07:00
dajinglingpake
0648e9e420 fix(server): support IntelliJ IDEA OpenAI clients over HTTP
JetBrains Runtime (JBR 25+) sends an h2c upgrade on OpenAI-compatible requests, which the HTTP/1.1 server would otherwise close. Intercept the upgrade, replay the buffered request through the existing handler, and respond over HTTP/1.1.
2026-08-05 10:32:32 +07:00
DaDecky
ae4f76c433 fix(auth): redirect active sessions from /login
/api/auth/status did not expose whether the auth cookie corresponds to a
valid dashboard session, so /login could only detect "auth disabled"
(requireLogin === false) and not "already logged in". Add authenticated
to the status response and redirect from /login when it's true.
2026-08-05 10:27:27 +07:00
lazysaltyfish
918b3c87a1 fix(cli-tools): enable Apply button for dynamic OpenAI/Anthropic-compatible providers
getAllAvailableModels() only consulted the static PROVIDER_MODELS catalog,
which has no entry for dynamically-registered compatible providers
(id like openai-compatible-chat-uuid). Fall back to the connection's
own defaultModel/customModels/placeholder, mirroring ModelSelectModal.js.
2026-08-05 10:23:59 +07:00
techysy
0e5da70cb1 fix: freeTier/apikey providers without authModes default to apikey in dualAuthTypes
Free-tier and apikey providers (e.g. cloudflare-ai, byteplus, ollama, vertex) whose registry entry omits authModes were treated as oauth-only, hiding their apikey connections on the providers grid card.
2026-08-05 10:22:57 +07:00
2a37a4085e chore: normalize formatting (2-space → tabs) in stream-error-patterns files
Re-tab only — no logic changes. Follows the repo's tab-based formatting
for these files, matching the CommandCode executor/translator style.

Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
2026-08-05 09:03:41 +07:00
e2f8323ab1 docs(open-sse): in-stream error handling recipe + CHANGELOG 2026-08-04 23:35:57 +07:00
9bd7adc556 feat(dashboard): Stream Error Patterns editor on provider page
Per-provider textarea (one pattern per line: plain text or /regex/flags)
saved via /api/settings as streamErrorPatterns. Mirrors the existing
providerTimeouts load/save pattern.
2026-08-04 23:35:25 +07:00
34a78f579e feat(open-sse): mark streaming requestDetail as error on stream-error pattern match 2026-08-04 23:34:21 +07:00
c8b96a61e7 feat(open-sse): non-streaming stream-error pattern match → 502 fallback 2026-08-04 23:34:00 +07:00
e438a03f96 feat(open-sse): early-peek stream error detection + fix UTF-8 loss in CommandCode peek
- new open-sse/utils/streamErrorPeek.js: bounded peek of the first bytes of a
  200 stream; configured pattern match → 502 so account/combo fallback can run
  before any byte reaches the client (streaming included). Re-emits RAW bytes
  so split multi-byte UTF-8 sequences survive the peek (never re-encode
  decoded text — TextDecoder flush corrupts a lone leading byte to U+FFFD).
- chatCore: run the peek after executor.execute when the provider has
  streamErrorPatterns configured; chat.js passes the settings through.
- commandcode executor: same raw-bytes fix in peekForUpstreamError + regression
  test that fails against the old flush-based re-encode.
2026-08-04 23:33:12 +07:00
008e0ef311 feat(settings): default streamErrorPatterns key 2026-08-04 23:28:17 +07:00
5058a402f1 feat(open-sse): streamErrorPatterns matching util (text + regex) 2026-08-04 23:27:49 +07:00
9b27ee2611 fix(open-sse): treat CommandCode in-stream error events as request failures
Upstream emits AI SDK v5 {"type":"error"} events inside an HTTP 200 stream.
The translator turned them into fake success content ([CommandCode error: ...]
+ finish_reason stop), so account/model fallback never fired and logs showed
Status: success.

- translator: error events now emit an OpenAI-shaped error chunk (chunk.error)
  instead of content; parseSSEToOpenAIResponse already detects chunk?.error
- executor: peek the first events before committing the response; an early
  error event returns 502 so fallback runs before any byte reaches the client
2026-08-04 23:26:37 +07:00
c5ce1ef140 feat(commandcode): quota usage dashboard, CLI-parity request, and connect timeout fixes
- Add CommandCode usage handler mirroring the official CLI /usage
  (whoami → credits/subscriptions → summary) with 5h/weekly/monthly
  quota rows, registered in services/usage.js and registry config
- Parse commandcode quota rows in ProviderLimits (remainingPercentage + $ unit)
- Match official CLI request shape: x-command-code-version 1.10.0,
  User-Agent cli, and config.environment '${platform}-${arch}, Node.js ${version}'
- Fix connect timeout unit confusion: both profile and provider pages
  now use ms with a 1s minimum guard (prevents 60ms footgun)
- Fix fetchT0 ReferenceError in base.js error path and log fetch
  diagnostics only on upstream failure
- Quota Tracker defaults to the Active account filter
- Ignore .commandcode/ CLI local state
2026-08-04 22:34:13 +07:00
fcd3dcb409 feat(translator): add vision support for commandcode provider
Map OpenAI image_url / Claude-style image blocks to the {type:"image",
image:"<data URI|url>"} shape the command-code CLI sends to /alpha/generate
instead of dropping them to "[image omitted]". Handles data URIs, raw base64
(with media_type / image/png fallback), and remote URLs.

- Promote bugs-gemini-cursor-commandcode "image content is preserved" from
  it.fails to a real assertion (bug fixed)
- Add vision unit tests to openai-to-commandcode.test.js
- Add test-commandcode-vision.sh curl helper for live verification

Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
2026-08-03 17:06:32 +07:00
067f18aaa1 fix(usage): ReferenceError in byProvider lastUsed overlay broke daily-summary periods
The provider lastUsed overlay loop referenced histRows before its const
declaration (temporal dead zone), throwing ReferenceError for 7d/30d/60d/all
which use the usageDaily summary path — leaving the overview cards and tables
empty regardless of the selected period. Merged the provider overlay into the
existing histRows loop.
2026-08-02 22:48:15 +07:00
B1nh M1nh
f260a1817b feat: Ollama Cloud quota tracker + proactive background OAuth refresh
Ollama: replace informational stub with real quota tracker hitting ollama.com/api/usage (session 5h + weekly 7d, 0..1 ratio) and /api/me plan label; bind handler to apiKey + add features.usageApikey so apikey connections work.

Token refresh: add backgroundTokenRefresh scheduler that refreshes OAuth connections within max(provider lead, 30min) of expiry, independent of inbound traffic (10s after boot, then every 5min, unref'd timers, DISABLE_BACKGROUND_TOKEN_REFRESH kill-switch, fail-open per tick/connection). Registered from custom-server.js (listening) and initializeApp.js. checkAndRefreshToken gains opt-in {force} for the scheduler; request path unchanged.
2026-08-02 09:34:27 +07:00
88faba150a feat(dashboard): model test-all with connection selector, usage by provider, combo enable toggle and showOnlyComboModels setting
- provider detail: Test All Models button runs every model (built-in + custom)
  with an optional connection selector; failed models can be disabled in bulk
- bulk selected-model test now runs all connections in parallel (Promise.all)
- usage overview: new 'Usage by Provider' table view (default) and a Provider
  column in Recent Requests; byProvider now tracks lastUsed
- combos: per-combo enable/disable toggle; disabled combos are skipped by the
  routing engine (getComboModels/getComboModelsFromData) and model listing
- settings: 'Only show combo models' toggle filters ModelSelectModal and the
  /v1/models response to models present in enabled combos
2026-07-31 10:37:45 +07:00
0dbae80930 Merge branch 'master' into gitea/new_feature 2026-07-30 23:14:05 +07:00
decolua
6fcd27337a # v0.5.45 (2026-07-30)
## Features
- **Providers**: add Poolside (OpenAI-compatible)
- **Providers**: add api-airforce, baidu, bazaarlink, bluesminds, kilo-gateway, llm7, morph, sambanova, tencent
- **OAuth**: zed / trae / windsurf providers + harden callback proxies
- **CLI tools**: set Claude Code max context tokens
- **Qoder**: PAT auth + refresh model list
- **Gemini**: Gemini 3.6 Flash tier routing + Gemini 3.5 Flash Lite
- **Claude**: bump default Opus to `claude-opus-5`
- **Kiro**: add Claude Opus 5 models
- **Usage**: Kimi and DeepSeek usage handlers
- **Usage**: SuperGrok weekly pool via gRPC-web

## Fixes
- **Refresh**: rotate `refresh_token` between retry attempts
- **Kiro**: canonicalize tool history and route API keys correctly
- **Kiro**: normalize dashboard thinking intensity models
- **Cursor**: stop leaking agent tool errors as text
- **Gemini**: fill empty tool schemas after `$ref` strip
- **Antigravity**: strip `stream_options` from non-stream requests
- **Jina-reader**: recover after transient errors, use JSON POST API
- **Usage**: record exact embedding tokens
- **Tunnel**: preserve successor cloudflared PID
- **Console-log**: initialize capture at server boot + prevent SSE proxy buffering
- **Dashboard**: count dual-auth, free-tier OAuth and API-key connections correctly
- **Dashboard**: flex quota rows, thin global scrollbars, no hidden-row overflow

## Docs
- **i18n**: expand pt-BR translation to 986 terms
- README: Indonesian translation
2026-07-30 09:43:55 +07:00
decolua
9be6588cc8 chore: drop source-attribution comments from provider code
Remove "Ported from OmniRoute" and cockpit-tools attribution comments.
User-Agent strings and README/landing credits are left intact.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 21:05:23 +07:00
decolua
1319dea620 fix(providers): count apikey connections for freeTier providers
openrouter, nvidia, gemini lack authModes, so dualAuthTypes on the
providers page defaulted to "oauth" and their apikey connections showed
as "No connections" on the freeTier card.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 20:21:27 +07:00
whale9820
31df0635aa feat(providers): add Poolside provider (OpenAI-compatible)
Adds Poolside (inference.poolside.ai) as an API-key provider using the default OpenAI transport. Registers three Laguna models with reasoning capabilities (262K context, 32K max output).
2026-07-29 20:17:26 +07:00
decolua
baf3356583 fix(ui): count free-tier oauth connections on providers list
Free-tier cards (e.g. kimchi, oauth-only) hardcoded "apikey" for stats and
toggle, so oauth connections were invisible on /dashboard/providers despite
showing on the detail page. Use dualAuthTypes per provider instead.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 20:06:20 +07:00
ridwan kulu
24fd165b0d docs(readme): add Indonesian translation
Add an Indonesian README and link it from the root README language switcher.
2026-07-29 19:38:30 +07:00
Sutarto Jordan Chrisfivo
a8313cd322 feat(kiro): add Claude Opus 5 models
Register Opus 5 and its thinking/agentic variants with 1M context
and adaptive-thinking capabilities.
2026-07-29 19:34:14 +07:00
Fábio A.
f8e8039446 i18n(pt-BR): expand partial translation to 986 terms
Add ~793 new pt-BR UI strings for the dashboard and settings.
2026-07-29 19:31:16 +07:00
Cokky Turnip
e3e3e235f6 fix(gemini): fill empty tool schemas after $ref strip
Vertex rejects orphan {} left when $ref/$defs are removed from function declarations. Promote empty nodes to object+reason placeholder in addPlaceholders.
2026-07-29 19:31:15 +07:00
Nurwanda Romadhon
0afe949387 fix(antigravity): strip stream_options from non-stream requests
OpenAI clients may send stream_options with stream=false; Google
generateContent rejects that combination. Drop it when not streaming.
2026-07-29 19:30:44 +07:00
Kyle Welsworth
5e59790824 fix(cursor): stop leaking agent tool errors as text
Emit SSE error frame for unsupported Cursor AgentService IDE tools
instead of assistant content, and drop frames after the turn finishes
to avoid double-closing the stream controller.
2026-07-29 19:29:27 +07:00
nguyenha935
16cb40fda1 fix(kiro): canonicalize tool history and route API keys correctly
Route API-key inference through Amazon Q first, enforce adjacent
one-to-one tool use/result pairs after session replay, and treat
payload-invalid HTTP 400 as terminal.
2026-07-29 19:27:41 +07:00
decolua
44c7b34837 fix(ui): count dual-auth provider cards correctly
Share oauth+apikey/api_key stats for dual-auth providers (incl. kiro)
so card totals match the detail page.
2026-07-29 19:27:31 +07:00
decolua
15dfd86416 fix(ui): flex quota rows and thin global scrollbars
Replace the fixed-table quota layout with flex rows that shrink cleanly,
keep the hidden-quota chip row from overflowing, and use thin mac-like
scrollbars app-wide.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:10:33 +07:00
decolua
8b0fcf4b16 feat(cli-tools): allow setting Claude Code max context tokens
Add a context-window selector on ClaudeToolCard that writes
CLAUDE_CODE_MAX_CONTEXT_TOKENS into settings.json (nudged 2K under the
labeled cap), and clear it on reset/default.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:10:31 +07:00
decolua
6d96e24bd9 chore(providers): refresh catalogs, free tiers, and hide stale ones
Update model lists and context lengths across free/apikey providers,
move bazaarlink, kilo-gateway, and kimchi into freeTier, demote llm7 to
apikey, and hide bluesminds, sambanova, zed, and mimo-free.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:10:09 +07:00
decolua
3b14bf4a49 feat(devin-cli): bridge client tools via MCP and use full agent
Default to the full agent with built-in tools, expose client function
tools as an MCP server, surface tool calls as OpenAI tool_use, resolve
workspace cwd from the request, and bump context windows.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:10:03 +07:00
decolua
9c9dd7b191 feat(qoder): support PAT auth and refresh model list
Exchange Personal Access Tokens for short-lived job tokens, close the
SSE stream on terminal frames so non-streaming clients do not hang,
re-enable OAuth plus API-key auth modes, and replace the model catalog
with the current Qoder aliases.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:09:57 +07:00
decolua
f17a68aaee feat(usage): fetch SuperGrok weekly pool via gRPC-web
Decode GetGrokCreditsConfig frames when REST billing returns empty
caps, so SuperGrok weekly quota shows in the usage dashboard.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:09:43 +07:00
decolua
6eaa9f8369 feat(usage): add Kimi and DeepSeek usage handlers
Wire /v1/usages for Kimi (OAuth + API key) and balance API for DeepSeek,
flag both providers with usage/usageApikey, and normalize their quotas
in the dashboard ProviderLimits parser.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-29 18:09:35 +07:00
decolua
65ac9b3cec fix(ui): prevent hidden quota row from overflowing
Use w-full instead of min-w-0/flex-1 + overflow-x-auto so the hidden
quota chips wrap cleanly instead of stretching the row.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-26 10:03:31 +07:00
decolua
72ec06a81d feat(cli-tools): add Devin CLI provider with ACP stdio executor
Wire Devin CLI as a routed provider that spawns the local `devin acp`
binary. Add the DevinCliExecutor, register it in the executor map, expose
its status through the cli-tools batch endpoint and devin-settings route,
and document setup in cliTools constants.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-26 10:03:22 +07:00
decolua
de2da19a9e feat(providers): add api-airforce, baidu, bazaarlink, bluesminds, kilo-gateway, llm7, morph, sambanova, tencent
Register 9 new upstream providers with logos and update the auto-generated
registry index. Refresh providers/alias baselines and extend the alias
token allowlist so verify-alias stays green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-26 10:03:10 +07:00
decolua
aa0448f7e2 fix(refresh): rotate refresh_token between retry attempts
Rotating-RT providers (xAI/grok-cli) issue a new refresh_token on every
refresh; mutate credentials in-place so refreshWithRetry reuses the fresh
RT instead of the already-consumed one.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-25 17:30:11 +07:00
decolua
41c9e6be87 feat(claude): bump default Opus to claude-opus-5
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-25 17:30:02 +07:00
decolua
8e04fe1734 feat(oauth): zed/trae/windsurf providers + harden callback proxies
- zed live model discovery; codebuddy-intl handler; remove duplicate workbuddy
- split oauth providers.js into per-provider files (facade re-export)
- fold 5 standard refresh providers into config-driven generic
- hide trae/windsurf from registry (no tool calling support)
- fix login-CSRF + SSRF on trae/windsurf/zed local callback proxies
  via loopback-origin guard + strict state validation + apiOrigins allowlist
- move zed RSA private key transit to POST body; redact proxy logs

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-25 17:25:19 +07:00
Phuong Lambert
783e271c16 feat(gemini): add Gemini 3.6 Flash tier routing and 3.5 Flash Lite
Add gemini-3.6-flash tiered (high/medium/low) for Antigravity routing
via upstreamModelId "gemini-3.6-flash-tiered(level)" + thinkingLevel,
plus gemini-3.6-flash and gemini-3.5-flash-lite direct API models.

- getModelUpstreamId: split (level) suffix before lookup, re-append after
- Antigravity executor: preserve transformed body.model
- MITM extractModel: parse thinkingLevel for tiered model (default medium)
- Isolate Cloud Code endpoints: discovery (loadCodeAssist/onboardUser/
  quota) on PROD cloudcode-pa, chat transport on daily-cloudcode-pa
  to bypass prod 429
2026-07-23 16:34:24 +07:00
jacardl
3c17d3406b fix(jina-reader): recover after transient errors and use JSON POST API
Clear stale provider error code and account lock after a successful web
fetch (the core fetch handler never consumed the onRequestSuccess
callback), switch Jina Reader to its documented JSON POST request, and
parse the Title: metadata line before falling back to a Markdown heading.
2026-07-23 16:28:37 +07:00
ankit1324
007d372724 fix(kiro): normalize dashboard thinking intensity models
Strip the generic dashboard model(level) suffix before resolving Kiro
synthetic -thinking/-agentic variants so the upstream request no longer
carries an invalid parenthesized model id. Map explicit levels to native
Kiro effort fields only for supported Claude/GPT model families, and stop
advertising native levels for unsupported legacy Kiro models. Applies to
both OpenAI→Kiro and direct Claude→Kiro routes.
2026-07-23 16:27:06 +07:00
ryanngit
e45bd73d6e fix(tunnel): preserve successor cloudflared PID
Make PID cleanup conditional on the exiting child still owning the PID file so a stale exit cannot erase a replacement tunnel's PID. Only null the in-memory process when the exiting child is current. Explicit disable keeps unconditional cleanup.
2026-07-23 16:22:47 +07:00
zie
c85a5c57ba fix(usage): record exact embedding tokens 2026-07-23 16:07:57 +07:00
Duc Nguyen
57b3b2c175 fix(console-log): initialize capture at server boot + prevent SSE proxy buffering
Initialize initConsoleLogCapture() via Next.js instrumentation register()
hook so logs are captured from startup in headless/Docker deployments, and
add X-Accel-Buffering/Cache-Control headers to the SSE stream route to
prevent reverse proxies from buffering the initial payload.
2026-07-23 16:06:21 +07:00
Biuzai OpenClaw Agent
53a8b5ed55 feat(providers): add Gemini 3.6 Flash and Gemini 3.5 Flash Lite models 2026-07-23 15:48:23 +07:00
decolua
039c4dbc72 feat(providers): add trae/windsurf/zed/workbuddy/codebuddy-intl + icons
- New providers: trae, windsurf, zed, workbuddy, codebuddy-intl
  (registry + executor, wired into executors/index.js + registry/index.js)
- zed: port hosted cloud proxy from OmniRoute — RSA access-token → short-lived
  LLM token exchange (shared/zedAuth.js) + NDJSON {event}/{status}/[DONE]
  stream translated back to OpenAI via Claude/Gemini/OpenAI-Responses translators
- Provider icons (128x128 png) for the 5 new providers
- qoder + tokenRefresh provider tweaks

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-23 15:37:24 +07:00
decolua
79918c7830 # v0.5.40 (2026-07-20)
## Features
- **i18n**: add Khmer (km) translations
- **CLI tools**: configure Grok Build subagent models
- **Kimi**: merge OAuth into dual-auth provider, add K3 / K2.7 models
- **Dashboard**: ProviderTopology flow animation

## Fixes
- **DB**: resolve better-sqlite3 parameter binding crash
- **Translator**: pass `service_tier` through OpenAI → Responses conversion
- **Kiro**: map GPT-5.6 reasoning effort fields
- **Kiro**: validate terminal streams before emitting output
- **Kiro**: map GPT reasoning effort fields
- **Codex**: current `client_version` + refresh-aware model sync
- **Alicode-intl**: split into Coding Plan + Model Studio providers
- **Cursor**: HTTP/2 AgentService support + version bump 3.12.17
- **Dashboard**: cut duplicate API/icon spam, lazy-load provider assets
2026-07-20 17:21:41 +07:00
long2ice
6994cd1f70 fix(cursor): HTTP/2 AgentService support + version bump to 3.12.17
Real Cursor IDE now uses AgentService at agent.api5.cursor.sh (HTTP/2-only)
while 9router still spoke the retired ChatService at api2.cursor.sh with
outdated headers, producing HTTP 429 "Update Required". Add an executeAgent
path that builds an agent.v1.RunRequest Connect RPC over a raw http2 stream
and fetches the account-specific usable model catalog via GetUsableModels.

Also implement MCP tool calling over AgentService: encode OpenAI tools as
AgentRunRequest.mcp_tools (McpToolDefinition with google.protobuf.Value
input_schema), decode McpArgs tool calls, and forward them to the client as
OpenAI tool_calls so the client runs the tool and resumes in the next turn.
Reply to request_context_args with a non-empty RequestContext, to server
heartbeats with client_heartbeat, and to KV blob get/set with empty results,
so action queries no longer stall the stream. Fold the client system prompt
into the user message (custom_system_prompt makes the server return an empty
turn). Bump clientVersion to 3.12.17 and add the x-cursor-client-commit
header so the gateway identifies as a current Cursor IDE release.
2026-07-20 15:39:55 +07:00
Lê Tấn Thắng
4f48ab8c7f fix: resolve better-sqlite3 parameter array binding crash
Spread params into better-sqlite3 Statement.run/get/all so positional ? placeholders bind correctly. better-sqlite3 accepts positional args, not an array, so binding crashed whenever a query had parameters. Matches the bun:sqlite and node:sqlite adapters.
2026-07-20 12:08:58 +07:00
Rafli Ahmad Zulfikar
c97963c4fb fix(translator): pass service_tier through OpenAI→Responses conversion
Forward the service_tier field from OpenAI requests into the Responses API payload so clients can select priority/default/flex tiers instead of the field being silently dropped.
2026-07-20 11:24:37 +07:00
Edison42
cef5dd4d61 fix(kiro): map GPT-5.6 reasoning effort fields
Route GPT-5.6 reasoning effort through Kiro's native reasoning.effort field instead of the legacy Claude output_config.effort path. GPT-5.6 models now emit reasoning.effort for low/medium/high/xhigh, with max mapped to the xhigh wire value.

Preserve the Responses API reasoning.effort through the OpenAI intermediate by copying it to reasoning_effort before the field is dropped. Skip legacy thinking_mode prompt tags when a supported native GPT effort is emitted, while keeping the legacy fallback for unsupported values (auto/minimal/ultra) and explicit disable semantics (none/off/disabled). Claude adaptive effort continues to use thinking plus output_config.effort.
2026-07-20 11:11:37 +07:00
Nur Ad-Duja
d587b2a487 fix(codex): current client_version + refresh-aware model sync
Bump client_version to 0.144.6 (above the 0.144.0 gate in codex CLI's
manifest) so /codex/models no longer returns 200 with newest entries
silently filtered out. Add the originator: codex_cli_rs header used by
every other codex call site, and move the entry onto buildOAuthResolver
so token refresh on 401/403 and a warning field on empty results are
wired in, matching gemini-cli and grok-cli.
2026-07-20 10:55:42 +07:00
Edison42
7c7fae3955 fix(kiro): validate terminal streams before emitting output
Validate AWS EventStream framing, header bounds, CRCs, error frames,
and terminal stop metadata before exposing Kiro output. Classify stop
reasons into dispositions (complete / retryable / terminal_incomplete /
refusal) and retry once when the stream ends with a malformed tool call,
ellipsis-only output, or a short future-action sentence.

Fail closed: propagate streaming failures as error SSE (502) instead of
collapsing them into a successful stop, so incomplete responses no longer
leak as final answers.

Detect the observed evidence-prefixed trailing progress final without
broadening the Chinese heuristic to completed findings.
2026-07-20 10:55:33 +07:00
seakleang.nhak
9ba8f37486 feat(i18n): add Khmer language support
Register Khmer (km) locale, add complete 1,394-entry dashboard translation,
and show the Cambodia flag in the language selector. Place Khmer immediately
before Thai in the selector and keep the header locale in sync after switching.
2026-07-19 16:30:37 +07:00
decolua
55628eea02 fix(alicode-intl): split into Coding Plan + Model Studio providers
8b9cac1 swapped alicode-intl to the DashScope compatible-mode endpoint to
fix #2591 for standard DashScope keys, but that broke Coding Plan keys
(sk-sp-...) which only work on coding-intl.dashscope.aliyuncs.com. The two
key types use two different hosts and are not interchangeable.

- alicode-intl: revert to coding-intl endpoint (Coding Plan keys)
- alims-intl: new provider for dashscope-intl/compatible-mode (standard keys)

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-19 16:29:55 +07:00
Don Tanggang
c4a120af8f docs(readme): update free-tier provider status for 2026
Outdated free-tier info was misleading new users. Sync corrections into
README.md and README.zh-CN.md, and force IPv4-first DNS resolution in the
CLI launcher to avoid undici IPv6 connect timeouts (502).

- Kiro AI: free tier now ~50 credits/month (was "Unlimited FREE")
- Qwen Code / Gemini CLI: free tiers discontinued in 2026
- OpenCode Free: note free model list fluctuates
- Vertex AI: Gemini API no longer uses $300 free credits since Mar 2026
- cli/cli.js: spawn server/tray with --dns-result-order=ipv4first
2026-07-19 14:12:58 +07:00
Edison42
eb00222c4f fix(kiro): map GPT reasoning effort fields
GPT-5.6 via Kiro needs reasoning.effort while Claude uses output_config.effort.
Resolve the effort path per-model schema (like Kiro CLI/KAS) so GPT-5.6
receives the correct structured thinking level. Claude path unchanged.

- Add resolveKiroEffortPath returning "reasoning" | "output_config" | null
- buildKiroAdditionalModelRequestFields emits schema-specific shape
- Keep prompt tags for backward compatibility
- Add OpenAI/Claude translator coverage for GPT-5.6 effort mapping
2026-07-19 13:53:30 +07:00
tuanminhhole
43d4abbcf2 docs(README): add Vietnamese OpenClaw Zalo video guide 2026-07-19 13:45:20 +07:00
rixzkiye
e0ba667450 feat(cli-tools): configure Grok Build subagent models
Add separate model selectors for Grok Build main, general-purpose,
explore, and plan agents. Each override gets an independent 9Router
custom-model slot and context_window derived from 9Router model
capabilities. Preserve and restore pre-existing config on reset.
2026-07-19 13:35:38 +07:00
decolua
0513bf393f Flow animatopn 2026-07-19 13:19:13 +07:00
d826e39008 merge origin/master into gitea/new_feature
Bring local branch up to v0.5.35 while keeping xAI image/edit, SuperGrok
quota tracking, per-provider timeouts, and pinned model-test actions.
2026-07-17 15:31:47 +07:00
2897cc3972 feat(providers): pin model tests to a specific account
Add Test action on each connection and Test Selected for multi-select.
Model pings go through /api/models/test with x-connection-id so the call
uses only the chosen account, with no round-robin fallback.
2026-07-17 15:23:35 +07:00
decolua
ccb0842d0a fix(dashboard): cut duplicate API/icon spam, lazy-load provider assets
Share one /api/models fetch via useModelCaps cache, mount ModelSelectModal
only when open, stop double fetchModelAliases on CLI tool cards, and resolve
provider icons through a session 404 cache with missing PNGs + loading=lazy.
Also include Claude Exa MCP toggle (claude-settings + ClaudeToolCard) that
was already in the working tree.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-17 12:12:20 +07:00
decolua
68566f53dc feat(kimi): merge OAuth into dual-auth provider, add K3/K2.7 models
Gộp kimi-coding vào kimi (oauth+apikey), parity CLIProxyAPI device flow/headers/refresh.
Thêm K3 + K2.7 Code (+ Kimi Code ids), pricing/caps vision, cập nhật baseline.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-17 12:09:14 +07:00
decolua
bc252ea802 # v0.5.35 (2026-07-16)
## Features
- **xAI**: Grok Imagine video generation (`/v1/videos`) + CLI
- **CLI tools**: Grok Build setup — writes `[model.9router]` to `~/.grok/config.toml`
- **GitHub Copilot**: route Claude models through Copilot's native `/v1/messages`
- **Kiro**: add GPT-5.6 model family (#2596)
- **RTK**: `X-9Router-Token-Saver` header to bypass token savers per request
- **Providers**: quota visibility settings
- **Translator**: drop temperature for all Claude models
- **i18n**: Thai (th) + Persian (fa) translations / README

## Fixes
- **Providers**: bulk-add API keys no longer overwrite existing keys (gap-fill `Key N`)
- **Anthropic**: lowercase `anthropic-version` header to prevent duplication on `/v1/messages`
- **Alicode-intl**: use DashScope compatible-mode endpoint so standard keys work
- **Grok CLI**: align Grok Build with current subscription protocol (#2590)
- **Grok CLI**: surface `expiresAt` so proactive token refresh fires (#2546)
- **Kiro**: improve direct session cache reuse
- **Models**: populate capabilities for live-catalog LLM models
- **Models**: list compatible provider models in `/v1/models`
- **Thinking**: send explicit `thinking:{type:adaptive}` alongside `output_config.effort`
- **Translator**: strip `client_metadata` when converting openai-responses → openai

## Improvements
- **Perf**: skip inactive background services on startup
2026-07-16 18:13:51 +07:00
asynx6
de680e789f fix(providers): bulk-add API keys no longer overwrite existing keys
Bulk-add named auto-generated keys by paste-line index, blind to existing
connection names. The backend upserts apikey connections by exact name
(connectionsRepo), so a colliding generated name silently replaced an
existing key instead of inserting a new one.

Add a collision-aware planner (src/shared/utils/bulkAdd.js) that gap-fills
the smallest free "<base> <n>" against both existing connection names and
names assigned earlier in the same batch, so a generated name is never
reused and the backend always inserts. Applies to auto-named lines, custom
name|apiKey lines, and Cloudflare name|apiKey|accountId lines.

Wire the planner into AddApiKeyModal and pass existing connection names
from the provider detail page. Add unit tests covering gap-fill, custom
names, Cloudflare 3-part format, and robustness.
2026-07-16 17:20:10 +07:00
Tuan Do
6acc3bb965 fix(anthropic): lowercase anthropic-version header key to prevent duplication on /v1/messages 2026-07-16 16:15:24 +07:00
Ella CEO
8b9cac180e fix(alicode-intl): use DashScope compatible-mode endpoint so standard keys work
Switch baseUrl from coding-intl.dashscope.aliyuncs.com (Coding Plan keys
only) to dashscope-intl.aliyuncs.com/compatible-mode so ordinary DashScope
API keys authenticate. Path /v1/chat/completions and preserveCacheControl
quirk unchanged.

Fixes #2591
2026-07-16 15:56:19 +07:00
YasharSL
30d0f6d3d8 docs(README): Add Persian youtube video tutorial 2026-07-16 15:47:40 +07:00
ryanngit
59b7828237 fix(grok-cli): align Grok Build with current subscription protocol (#2590) 2026-07-16 15:33:19 +07:00
ann
d6761c6fb0 feat(xai): add Grok Imagine video generation (/v1/videos) + CLI
Async video job proxy mirroring the existing image-generation layer split:
Next routes → src/sse/handlers/videoGeneration.js (auth gate, account
fallback loop, refresh persistence) → open-sse/handlers/videoCore.js
(transparent upstream proxy, 401 refresh-once/retry-once, secret sanitization).

- POST /v1/videos/{generations,edits,extensions}: byte-exact body forward
  (JSON + multipart), request_id passthrough, Idempotency-Key forwarded
- GET /v1/videos/{request_id}: status/progress/video.url passthrough
- Register grok-imagine-video (kind: "video"); add "video" to MODEL_TYPE_TO_KIND
  so video models stay out of chat lists (also fixes runwayml leak)
- 9router xai video CLI: submit → poll → atomic MP4 download
- No auto-retry of creation POSTs (billable jobs); rotate accounts only on
  401/403/429; sanitize Bearer tokens + credential values from errors/logs

Closes #1285
2026-07-16 15:29:52 +07:00
M0nt
02ccdc2d22 i18n: add Persian (fa) translations for README and UI
Add full Persian (Farsi) translation of README and sync UI literals
to match zh-CN key set (1389 keys). No runtime code changes.
2026-07-16 15:19:38 +07:00
Edison42
9c58ba645e fix(kiro): improve direct session cache reuse
Reshape Kiro direct requests so resumed client sessions reuse Kiro's
cache-affinity fields instead of starting unrelated CodeWhisperer
conversations.

- keep conversationState.conversationId stable when the client sends an
  explicit session id (x-session-id, session_id, conversation_id, Claude
  Code session metadata)
- add a stable conversationState.agentContinuationId per Kiro session
- send conversationState.agentTaskType: "vibe" and agentMode: "vibe",
  matching the normal Kiro CLI/KAS chat path
- move Kiro thinking instructions into Kiro-compatible systemPrompt /
  additionalModelRequestFields instead of generic top-level thinking
- keep volatile timestamp context out of the top-level systemPrompt; it
  remains only in user content fallback
- suppress additionalModelRequestFields for legacy 4.5-era Claude/Kiro
  models that reject it, while defaulting future Claude/Kiro model ids
  to supported
- preserve Kiro meteringEvent credit usage internally for accounting
  without leaking provider-specific fields into OpenAI-compatible usage
- prevent unrelated headerless Kiro requests from sharing one
  connection-wide continuation
- cap/evict continuation sessions so long-running processes do not grow
  the continuation map unbounded
- treat generated headerless Kiro sessions as one-shot so they do not
  evict real explicit-session continuations
- keep credit-only Kiro metering valid for internal persistence when
  token metrics are unavailable
2026-07-16 15:15:05 +07:00
rixzkiye
70e8dc4974 feat(cli-tools): add Grok Build setup
Add Grok Build to Dashboard → CLI Tools. Apply writes a [model.9router]
custom model to ~/.grok/config.toml and sets [models].default, routing
the xAI Grok TUI through 9Router. Reset removes the slot and restores
the previous default.
2026-07-16 14:47:20 +07:00
Edison42
b94685b80d feat(kiro): add GPT-5.6 model family (#2596)
Add GPT-5.6 Sol/Terra/Luna and their synthetic thinking/agentic/
thinking-agentic variants to the Kiro static catalog with the observed
272k context window and credit multipliers (2.4/1.2/0.6), register MITM
mapping slots for the new base ids, and override runtime capabilities so
the GPT-5.6 family reports the 272k window instead of the generic GPT-5
profile.
2026-07-16 14:38:08 +07:00
decolua
0248dd5348 feat(i18n): add Thai language translation (#2581) 2026-07-16 14:31:23 +07:00
hungtrinh
27b37705b3 perf(startup): skip inactive background services 2026-07-16 12:09:48 +07:00
Ella CEO
7dfb346667 fix(grok-cli): surface expiresAt so proactive token refresh fires (#2546) 2026-07-16 12:00:14 +07:00
decolua
a6a41dfb3c Merge remote-tracking branch 'upstream/master'
# Conflicts:
#	.gitignore
#	open-sse/handlers/chatCore.js
2026-07-16 11:59:46 +07:00
luoyide
2629218b04 fix(models): populate capabilities for live-catalog LLM models 2026-07-16 11:50:15 +07:00
joachimBrindeau
c9926897ba feat(rtk): add X-9Router-Token-Saver header to bypass token savers per request 2026-07-16 11:27:42 +07:00
liamgnc
88a8c72d2d fix(models): list compatible provider models in /v1/models
Replace the overly-broad UPSTREAM_CONNECTION_RE regex (which matched all
provider IDs with UUID suffixes) with an x-9r-internal-models-fetch
header to detect cross-instance recursive /models fetches.

fetchCompatibleModelIds now sends the header when fetching upstream
/models; the GET handler detects it and skips dynamic fetching, breaking
the recursion loop while letting compatible providers (MLX, Ollama, vLLM)
list their models. Fixes #2626.
2026-07-16 11:18:09 +07:00
luoyide
ba508f2506 fix(thinking): send explicit thinking:{type:adaptive} alongside output_config.effort 2026-07-16 11:17:08 +07:00
decolua
a077ee85bd gitignore 2026-07-16 11:16:57 +07:00
qianze
e567ba800f fix(translator): strip client_metadata when converting openai-responses to openai
client_metadata is an OpenAI Responses API-specific field. When translating
openai-responses requests to openai (Chat Completions), it leaked through
to providers like NVIDIA, which rejected it with a 400 "Unsupported
parameter". Strip it in the Responses-specific cleanup block alongside
input, instructions, store, and reasoning.
2026-07-15 17:38:53 +07:00
luoyide
542a088c04 feat(github): route Claude models through Copilot's native /v1/messages
GitHub Copilot's /chat/completions and /responses endpoints never surface
prompt-cache token counts for Claude models. Route Claude models (detected
by name pattern) to Copilot's Anthropic-native /v1/messages shim via a new
executeWithMessagesEndpoint(), translating OpenAI-shape requests to Claude
natively so cache_control gets injected and cached_tokens surface.

Also fixes translateRequest()'s internal _toolNameMap being sent upstream,
which made Anthropic's strict schema reject tool-call requests with a 400 —
now stripped and threaded through response state. Removes the now-dead
response_format Claude JSON-mode workaround.
2026-07-15 17:34:08 +07:00
Moradii.Mohammadreza
9173c29b66 feat(translator): drop temperature for all Claude models
Broaden strip rule from /claude-opus-4/i to /claude/i so temperature is
removed for every Claude model, not just opus-4. Fixes Anthropic 400 on
OpenAI-compatible routes. #1748
2026-07-15 17:08:41 +07:00
decolua
eceac9d7ae gitignore 2026-07-15 16:40:41 +07:00
ab9a3c1d43 feat(xai): track SuperGrok weekly limit + API usage quota
Fetch OAuth quota from cli-chat-proxy billing, GetGrokCreditsConfig weekly
window, and settings plan label so the dashboard matches grok.com usage.
2026-07-13 23:44:36 +07:00
minnyww
837cfec5a9 feat(i18n): complete Thai translation (1389 keys) + README.th.md 2026-07-13 17:08:49 +07:00
b1d368d960 feat: xAI image generate/edit, API key import, and per-provider timeouts
- Add dedicated xAI image adapter with generate + edit (multi-image) via
  /v1/images/generations and /v1/images/edits, plus aspect_ratio/resolution UI
- Support importing existing API keys and exposing connection api-key routes
- Add global/per-provider connect timeout overrides from settings
- Keep unrelated provider UX improvements on this branch; no Grok quota tracking
2026-07-13 16:52:22 +07:00
minnyww
f89ba32d79 feat(i18n): add Thai language translation 2026-07-13 16:50:09 +07:00
decolua
9845a1702f # v0.5.30 (2026-07-10)
## Features
- **Perplexity**: add Agent API provider (#2492)
- **Grok CLI**: add Grok CLI / Grok Build provider with OAuth device-code flow (#2502)
- **Featherless**: add OpenAI-compatible provider presets
- **SearXNG**: configure endpoint via SEARXNG_URL env (#2499)
- **Providers**: add max thinking level for gpt-5.6-sol (#2500)
- **Headroom**: add extras detection and install UI (#2403)
- **Headroom**: activate/uninstall extras + fix interpreter detection
- **PXPipe**: PXPIPE token saver — multimodal prompt compression (#2465)
- **Proxy-Pools**: auto-rotate strategy for no-auth providers (#2409)

## Fixes
- **Cloudflare-AI**: support accountId in bulk key import (#2449)
- **DB**: backup on schema change, MCP child cleanup, codex models, usage providers OOM
- **Codex**: avoid bare-email OAuth dedup (#2477)
- **CLI**: allow staged app bundle builds (#2479)
- **Headroom**: compress Kiro conversation state (#2488)
- **Gemini-CLI**: raise output floor for thinking and add validated toolConfig (#2486)
- **GitHub**: label Copilot profiles by account identity (#2498)
- **OpenAI-to-Claude**: unwrap bare {function:{…}} tools without parent type (#2473)
- **Translator**: clamp thinking effort max->xhigh for OpenAI format (#2466)
- **RTK/find**: detect and group Windows backslash-style find output (#2448)
- **Codex**: handle fast tier and capacity SSE (#2452)
- **Volcengine-ark**: clamp Kimi max_tokens to 32768 endpoint cap
- **Antigravity**: align provider fingerprint with IDE Desktop 2.1.1 (#2389)
- **Pricing**: update Claude/Codex model rates and add new models

## Improvements
- **i18n(zh-CN)**: complete Chinese translations for all UI strings (#2436)
- **API**: caching for tunnel and version status endpoints
- **Perf**: faster dev startup and lighter bundle
2026-07-10 18:12:07 +07:00
decolua
a625ea9fd8 refactor(log): unify request lifecycle logging with session-colored tags
Collapse scattered per-request console lines (request/routing/auth/pending/
usage/stream-usage/stream) into 3 correlated lines: request, transform,
done. Add stable per-session color tag so concurrent request lines are
easy to follow, surface thinking intent, always-on full error logging
for debug, re-enable warn level, and uppercase keyword labels. Also fix
usage overview cards wrapping (5 cards -> grid-cols-5).

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 18:01:20 +07:00
decolua
b61c50cbb7 # v0.5.29 (2026-07-10)
## Features
- **Perplexity**: add Agent API provider (#2492)
- **Grok CLI**: add Grok CLI / Grok Build provider with OAuth device-code flow (#2502)
- **Featherless**: add OpenAI-compatible provider presets
- **SearXNG**: configure endpoint via SEARXNG_URL env (#2499)
- **Providers**: add max thinking level for gpt-5.6-sol (#2500)
- **Headroom**: add extras detection and install UI (#2403)
- **Headroom**: activate/uninstall extras + fix interpreter detection
- **PXPipe**: PXPIPE token saver — multimodal prompt compression (#2465)
- **Proxy-Pools**: auto-rotate strategy for no-auth providers (#2409)

## Fixes
- **Cloudflare-AI**: support accountId in bulk key import (#2449)
- **DB**: backup on schema change, MCP child cleanup, codex models, usage providers OOM
- **Codex**: avoid bare-email OAuth dedup (#2477)
- **CLI**: allow staged app bundle builds (#2479)
- **Headroom**: compress Kiro conversation state (#2488)
- **Gemini-CLI**: raise output floor for thinking and add validated toolConfig (#2486)
- **GitHub**: label Copilot profiles by account identity (#2498)
- **OpenAI-to-Claude**: unwrap bare {function:{…}} tools without parent type (#2473)
- **Translator**: clamp thinking effort max->xhigh for OpenAI format (#2466)
- **RTK/find**: detect and group Windows backslash-style find output (#2448)
- **Codex**: handle fast tier and capacity SSE (#2452)
- **Volcengine-ark**: clamp Kimi max_tokens to 32768 endpoint cap
- **Antigravity**: align provider fingerprint with IDE Desktop 2.1.1 (#2389)
- **Pricing**: update Claude/Codex model rates and add new models

## Improvements
- **i18n(zh-CN)**: complete Chinese translations for all UI strings (#2436)
- **API**: caching for tunnel and version status endpoints
- **Perf**: faster dev startup and lighter bundle
2026-07-10 17:51:48 +07:00
decolua
baafc74c1f Bump version 2026-07-10 17:48:53 +07:00
decolua
bb314118f2 Bump version 2026-07-10 17:38:40 +07:00
decolua
2d515c8abc # v0.5.28 (2026-07-10)
## Features
- **Perplexity**: add Agent API provider (#2492)
- **Grok CLI**: add Grok CLI / Grok Build provider with OAuth device-code flow (#2502)
- **Featherless**: add OpenAI-compatible provider presets
- **SearXNG**: configure endpoint via SEARXNG_URL env (#2499)
- **Providers**: add max thinking level for gpt-5.6-sol (#2500)
- **Headroom**: add extras detection and install UI (#2403)
- **Headroom**: activate/uninstall extras + fix interpreter detection
- **PXPipe**: PXPIPE token saver — multimodal prompt compression (#2465)
- **Proxy-Pools**: auto-rotate strategy for no-auth providers (#2409)

## Fixes
- **Cloudflare-AI**: support accountId in bulk key import (#2449)
- **DB**: backup on schema change, MCP child cleanup, codex models, usage providers OOM
- **Codex**: avoid bare-email OAuth dedup (#2477)
- **CLI**: allow staged app bundle builds (#2479)
- **Headroom**: compress Kiro conversation state (#2488)
- **Gemini-CLI**: raise output floor for thinking and add validated toolConfig (#2486)
- **GitHub**: label Copilot profiles by account identity (#2498)
- **OpenAI-to-Claude**: unwrap bare {function:{…}} tools without parent type (#2473)
- **Translator**: clamp thinking effort max->xhigh for OpenAI format (#2466)
- **RTK/find**: detect and group Windows backslash-style find output (#2448)
- **Codex**: handle fast tier and capacity SSE (#2452)
- **Volcengine-ark**: clamp Kimi max_tokens to 32768 endpoint cap
- **Antigravity**: align provider fingerprint with IDE Desktop 2.1.1 (#2389)
- **Pricing**: update Claude/Codex model rates and add new models

## Improvements
- **i18n(zh-CN)**: complete Chinese translations for all UI strings (#2436)
- **API**: caching for tunnel and version status endpoints
- **Perf**: faster dev startup and lighter bundle
2026-07-10 17:36:21 +07:00
decolua
74d5fedf79 feat(headroom): activate/uninstall extras + fix interpreter detection
- find interpreter next to headroom binary so extras/version read correctly
- add on/off toggle to activate [code]/[ml] via proxy restart
- add uninstall action + live install log progress + ~1GB confirm modal
2026-07-10 17:32:45 +07:00
Elio Bonfim Júnior
dcf1927f22 feat(pxpipe): PXPIPE token saver — multimodal prompt compression (#2465)
Add pxpipe as an experimental fifth Token Saver: Claude-format request
bodies above a configurable size threshold are rendered as dense PNGs
via the pxpipe-proxy library API (transformAnthropicMessages) before
dispatch, cutting estimated input tokens by ~35-60% on token-dense
contexts. Integration follows the Headroom pattern: applied to the final
body in chatCore just before dispatch, fail-open on any error/timeout.

Managed npm install into DATA_DIR/pxpipe, dynamic loader with per-version
cache-bust, JSONL event log with rotation, /api/pxpipe/* endpoints, Token
Saver card (marked experimental) + /dashboard/pxpipe page, and per-request
Activated/Skipped annotation in Request Details. Disabled by default.
2026-07-10 16:10:42 +07:00
Fadjrir Herlambang
e1f3399b73 feat(proxy-pools): auto-rotate strategy for no-auth providers (#2409)
Add round-robin/random proxy pool rotation for no-auth free providers
(e.g. OpenCode Free) to distribute load across all active pools and
avoid per-IP rate limits. Rotation strategy is selectable per provider
in NoAuthProxyCard and persisted to settings.providerStrategies.
2026-07-10 16:05:07 +07:00
KunN-21
f1f9d27061 feat(headroom): add extras detection and install UI (#2403)
- add Headroom extras status + install endpoints
- show Headroom version + code/ml extras in Token Saver UI
- fix Windows interpreter selection to read from env with headroom-ai
2026-07-10 16:05:04 +07:00
decolua
90df008f0c chore(release): v0.5.25
Update CHANGELOG, bump version, trim usage overview cards.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 16:04:37 +07:00
decolua
d2599ebf17 fix(pricing): update Claude/Codex model rates and add new models
Add claude-fable-5, gpt-5.6 family; correct gpt-5/5.1/5.2/5.3-codex rates to official pricing.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 16:02:18 +07:00
Rizkal
5cdcf67484 fix(cloudflare-ai): support accountId in bulk key import (#2449)
Bulk import now parses name|apiKey|accountId lines for Cloudflare AI and
forwards accountId via providerSpecificData, with a provider-specific
placeholder and format hint.
2026-07-10 13:10:29 +07:00
decolua
b25e10160d fix: DB backup on schema change, MCP child cleanup, codex models, usage providers OOM
- Backup DB only on real SCHEMA_VERSION change, not every app version bump
- Kill idle MCP stdio bridge children to prevent orphan process leaks
- Add getDistinctProviders to avoid loading every row JSON blob (OOM fix)
- Update codex model list (gpt-5.6 sol/terra/luna, drop 5.3 codex variants)
- Reorder Claude default models (fable first)

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 13:08:58 +07:00
decolua
0270f6ea70 perf: faster dev startup and lighter bundle
- Switch dev default to Turbopack (5-14x faster compile); keep webpack as dev:webpack
- Tailwind v4 source() base so JIT scans identically under both bundlers
- Lazy-load @xyflow/react via next/dynamic to keep it out of the shared bundle
- optimizePackageImports for heavy barrel imports (xyflow, dnd-kit, material-symbols, marked)
- Replace blind setTimeout waits with TCP health-check (waitServerReady)
- Run checkForUpdate in parallel instead of blocking server spawn
- Background MITM/tunnel/cloudflared kills off the critical path

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 13:08:02 +07:00
newnol
ce6bdf7fc2 feat(perplexity): add Agent API provider (#2492)
Add perplexity-agent provider using OpenAI-compatible Responses API,
routing third-party models (GPT, Claude, Gemini, Grok, GLM, Kimi, Sonar)
through one endpoint. Expose /v1/models discovery, add chat-search wrapper
via web_search tool. Existing Sonar provider unchanged.
2026-07-10 11:57:15 +07:00
Fadjrir Herlambang
a11937cdd6 feat(grok-cli): add Grok CLI / Grok Build provider with OAuth device-code flow (#2502)
New OAuth provider routing through cli-chat-proxy.grok.com (OpenAI Responses
API), distinct from xai (api.x.ai) and grok-web (cookie SSO):

- Registry + GrokCliExecutor: Chat Completions -> Responses transform, CLI
  fingerprint headers, virtual effort models grok-4.5-{low,medium,high}
- OAuth device-code flow (auth.x.ai) with no-PKCE, shared xAI token refresh
- store=false multi-turn continuity via reasoning encrypted_content
- Quota tracker: on-demand window + prepaid balance on dashboard
- Connection test: 402 spending-limit = soft success (auth OK, out of credits)
- Alias/oauth/provider baselines + unit tests
2026-07-10 11:47:08 +07:00
Hermes Hunter
c73c419d09 fix(codex): avoid bare-email OAuth dedup (#2477)
Only update an existing Codex OAuth row when both rows share the same
chatgptAccountId, so a second Codex login no longer overwrites the first
account's rotated token pair. Also fall back to
workspaceId || chatgptAccountId || accountId for the chatgpt-account-id header.
2026-07-10 11:41:43 +07:00
ryanngit
a3b267a5cb fix(cli): allow staged app bundle builds (#2479)
Write the standalone app bundle and MITM bundle to NINEROUTER_CLI_APP_DIR
when set, so staged deploys can build to a separate destination before
swapping. Default output stays at cli/app when the env var is unset.
2026-07-10 11:40:03 +07:00
Edison42
65c65a0f56 fix(headroom): compress Kiro conversation state (#2488)
Project conversationState history/currentMessage into OpenAI-style
messages for /v1/compress, then write compressed text back into the
original Kiro fields while preserving provider payload shape. Fail open
when the proxy returns malformed or reordered messages.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:33:21 +07:00
DOMANHDUC
7610f28f42 fix(gemini-cli): raise output floor for thinking and add validated toolConfig (#2486)
Gemini CLI requests with small max_tokens spend the whole output budget on
thoughts after reasoning_effort maps to thinkingConfig, returning blank
content or finish=length. Raise maxOutputTokens floors per thinking level/
budget (clamped to caps.maxOutput). Also emit toolConfig
functionCallingConfig.mode=VALIDATED for Gemini CLI tool requests to avoid
MALFORMED_FUNCTION_CALL.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:32:11 +07:00
newnol
0d4d4bc261 feat(featherless): add OpenAI-compatible provider presets
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:29:48 +07:00
ryanngit
3a7a878f91 fix(github): label Copilot profiles by account identity (#2498)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:29:12 +07:00
zie
e79f9eddb4 feat(searxng): configure endpoint via SEARXNG_URL env (#2499)
Add SEARXNG_URL runtime override for the built-in SearXNG web-search
provider, defaulting to http://localhost:8888/search. Enables Docker
and remote SearXNG deployments without changing existing behavior.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:29:06 +07:00
Rafli Ahmad Zulfikar
b9e2611045 feat(providers): add max thinking level for gpt-5.6-sol (#2500)
Expose max in the Codex thinking dropdown for gpt-5.6-sol only (maps to
xhigh on wire; live probe rejected ultra). Include custom/kilo models
when computing provider thinking options so manually added gpt-5.6-sol
contributes its max level.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:17:32 +07:00
Samir Abis
ddd5509e97 fix(openai-to-claude): unwrap bare {function:{…}} tools without parent type (#2473)
Translator only unwrapped tool.function when both tool.type==="function"
and tool.function were truthy. Loose/legacy OpenAI clients emit the bare
{ function: { name, parameters } } shape (no parent type), which fell
through and forwarded name: undefined upstream, rejected by strict
Anthropic-compatible gateways (MiniMax M3) as (2013) invalid tool type.

Unwrap tool.function whenever present. Built-in tools stay pass-through.
Adds regression coverage for the 4 tool shapes. See #2435.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-10 11:11:37 +07:00
thienpv
288940960a fix(translator): clamp thinking effort max->xhigh for OpenAI format (#2466)
Claude Code sends reasoning_effort "max" (its top level); OpenAI enum caps
at "xhigh" and rejects "max" with HTTP 400 "max effort not support".
applyFormat case "openai" now clamps "max"->"xhigh" before assigning
body.reasoning_effort; other levels pass through unchanged.

Add regression test covering client output_config.effort, direct
reasoning_effort, passthrough of xhigh/high, and budget_tokens capping.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-09 15:13:24 +07:00
Diwak4r
d75471bbbc fix(rtk/find): detect and group Windows backslash-style find output (#2448)
isPathLike rejected any line with a colon, so Windows absolute paths
(C:\Users\me\a.js) were never recognized and find dumps went uncompacted.
find.js also split only on "/", mis-grouping backslash paths.

- autodetect: treat drive-letter prefix (X:\ or X:/) as path-like before
  the general colon rejection.
- find.js: split on the last "/" or "\" separator and normalize emitted
  directory labels to forward slashes.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-09 15:12:52 +07:00
ryanngit
0c55d49ab6 fix(codex): handle fast tier and capacity SSE (#2452)
- map service_tier=fast to upstream priority; drop unsupported tiers
- normalize reasoning effort max to xhigh (codex-only)
- convert 200-SSE model-capacity errors into 503 so account fallback rotates
- keep normal SSE output intact after peeking

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-09 15:12:46 +07:00
whale9820
cfbdf06047 fix(volcengine-ark): clamp Kimi max_tokens to 32768 endpoint cap
VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the
model's advertised ceiling is far higher (Kimi-K2.7-Code resolves to
maxOutput 262144), so clampToModelMaxOutput alone leaves it uncapped and
the request 400s. Add a Kimi-scoped rule with an explicit maxOutputCap of
32768, combined with the model ceiling via min(). Covers max_tokens,
max_completion_tokens, max_output_tokens.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-09 15:10:05 +07:00
decolua
a4c5fa4e14 refactor(api): implement caching for tunnel and version status endpoints 2026-07-09 15:08:30 +07:00
qianze
20b442b708 i18n(zh-CN): complete Chinese translations for all UI strings (#2436)
Add 551 new translations covering previously untranslated areas
(landing, CLI tools, MITM, skills, combos, token-saver, OIDC,
relay deploy, provider details, proxy pools, quota tracker,
OAuth modals). Total: 838 -> 1389 entries.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-09 15:07:12 +07:00
nguyenha935
71cd5b2f23 fix(antigravity): align provider fingerprint with IDE Desktop 2.1.1 (#2389)
Match captured official Antigravity IDE traffic: cloudcode-pa host,
antigravity/ide/2.1.1 User-Agent, IDE-shaped agent requestId, and drop
router-only stream/usage headers plus the legacy double system prompt.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-08 10:17:47 +07:00
decolua
b10b807063 # v0.5.20 (2026-07-07)
## Features
- **Thinking**: per-model thinking level picker on provider page — appends `(level)` suffix to copied model names for forced reasoning effort across all formats (openai, claude, gemini, deepseek, kimi, qwen, zai, minimax, hunyuan, step)
- **RTK**: add JS-native git-log filter (#2423)
- **Caveman**: add targeted upstream-aligned style rules (#2424)
- **i18n**: add Farsi (fa) language support (#2385)

## Fixes
- **Thinking**: strip `(level)` suffix from upstream `body.model` so providers no longer reject requests
- **Translator**: preserve developer instructions in openai-responses conversion (#2434)
- **count_tokens**: count structured Anthropic blocks (#2419)
- **Volcengine-ark**: clamp GLM-5 max_tokens to model output ceiling (#2428)
- **Kimi**: normalize reasoning_effort to backend enum (#2427)
- **Claude**: reconcile max_tokens vs thinking budget and lift per-model ceiling (#2381)
- **Kiro**: deliver system prompt natively, add Opus 4.5/4.7/4.8, tolerate dash version ids (#2366)
- **Headroom**: proxy dashboard through app (#2372)
- **MITM**: recover from stale lock file on server start
2026-07-07 16:29:11 +07:00
baibiao
081c6f2aff fix(count_tokens): count structured Anthropic blocks (#2419)
Estimate tokens for tool_use, tool_result, thinking, system, and tools
blocks instead of text only, so count_tokens no longer returns 0 for
structured content and breaks Claude Code auto-compaction (#2337).

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 12:06:09 +07:00
KunN-21
19281b5524 feat(rtk): add JS-native git-log filter (#2423)
Compress git log output via dedicated RTK filter: keep commit headers,
Author/Date, subject; drop body padding, decoration, embedded diff lines.
Wire into autodetect (git-log prioritized before git-diff) and registry.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 12:02:56 +07:00
whale
bbae990b92 fix(volcengine-ark): clamp GLM-5 max_tokens to model output ceiling (#2428)
Ark rejects max_tokens above 128000 for GLM-5.2. Add a config-driven STRIP_RULES entry that clamps max_tokens, max_completion_tokens and max_output_tokens down to the model maxOutput before the upstream call.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 11:57:04 +07:00
whale
8c068a1f5c fix(kimi): normalize reasoning_effort to backend enum (#2427)
Map auto→high, minimal→low, xhigh→max and whitelist low/medium/high/max
so Kimi/kimchi SGLang backends no longer receive invalid effort values.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 11:56:39 +07:00
KunN-21
97a6708651 feat(caveman): add targeted upstream-aligned style rules (#2424)
Add four shared Caveman prompt fragments (no invented abbreviations,
preserve user language, no self-reference, no decoration) across all six
levels, and remove ULTRA contradictions around abbreviations/arrow
shorthand. Adds regression tests for the prompt rules.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 11:55:45 +07:00
deranalabs
a3cd7c82bc fix(translator): preserve developer instructions in openai-responses conversion (#2434)
Map role="developer" messages to top-level instructions alongside
role="system" in openaiToOpenAIResponsesRequest. Previously developer
messages matched no branch and were silently dropped from the Responses
request, losing GPT-5/Codex system-level prompts.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 11:54:07 +07:00
decolua
bf7da67859 docs(readme): swap in Vietnamese tutorial video; chore(pricing): minor update
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 11:44:30 +07:00
decolua
da0149de97 fix(mitm): recover from stale lock file on server start
Detect dead PID in lock file and reclaim it instead of failing, and drop unused fs dependency.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 11:44:20 +07:00
MuhammadHamidRaza
1885ad7f64 docs(readme): add English and Urdu/Hindi video tutorials (#2305)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-05 17:47:17 +07:00
Sutarto Jordan Chrisfivo
481e7e467b fix(headroom): proxy dashboard through app (#2372)
Add a 9Router-side proxy so the Headroom dashboard and its data
endpoints (/stats, /health, /stats-history, /transformations/feed)
stay same-origin when opened remotely through the 9Router app, and
add an "Open Headroom Dashboard" link in the Token Saver modal.

Gate /api/headroom/proxy as LOCAL_ONLY (loopback + CLI token) to
match start/stop, and strip cookie/authorization when the Headroom
target is non-loopback to avoid leaking viewer credentials.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-05 17:45:04 +07:00
Mohammed Faheem
008de32c06 docs: add CLAUDE.md guidance for Claude Code (#2354)
Top-level guide for Claude Code / AI coding agents working in this repo,
complementing docs/ARCHITECTURE.md and open-sse/AGENTS.md.

Docs-only: PR's incidental code changes were dropped as they reverted #2366.

Co-Authored-By: Mohammed Faheem <mohammed.faheem@adbsafegate.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-05 17:44:04 +07:00
Arash Kadkhodaei
b6454d84da feat(i18n): add Farsi (fa) language support (#2385)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-05 17:38:17 +07:00
thienpv
46e6c01a01 fix(claude): reconcile max_tokens vs thinking budget and lift per-model ceiling (#2381)
On the translated OpenAI->Claude path, adjustMaxTokens capped max_tokens
before applyThinking set thinking.budget_tokens, so max-effort budget
(128000) could exceed a 64k-clamped max_tokens -> Anthropic 400.
prepareClaudeRequest now reconciles after the budget is known: prefer
raising max_tokens, only shrink budget when it meets/exceeds the ceiling.

Also lift the global 64000 cap: the ceiling is now the model's real
maxOutput, so high-output models (fable/mythos, opus-4.8/sonnet-4.6) get
their full budget. adjustMaxTokens gains an optional ceiling arg (default
unchanged, callers untouched); openai-to-claude passes the model maxOutput.

Native Claude Code passthrough is unaffected.

Co-Authored-By: Claude <noreply@anthropic.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-05 17:38:16 +07:00
VitzS7
5041494e1c fix(kiro): deliver system prompt natively, add Opus 4.5/4.7/4.8, tolerate dash version ids (#2366)
- Pass system prompt via native systemInstruction field (+ <instructions> fallback)
  so Claude models stop treating it as info-only <system-reminder>
- Add Opus 4.5/4.7/4.8 (base/thinking/agentic/thinking+agentic) to Kiro registry
- normalizeModelId(): dash->dot version separator, scoped to Kiro provider only
- Replace <system-reminder> with <instructions> in claude-to-openai/openai-to-kiro

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-05 17:34:04 +07:00
nguyenha935
4dadab9d5f feat: add provider quota visibility settings 2026-07-04 23:26:49 +07:00
decolua
7f436e2792 # v0.5.18 (2026-07-03)
## Features
- **Usage**: track cached tokens + correct input/output/cache cost (#2209) — hodtien
- **Codex**: show reset credit expiry details (#2290) — Rafli Ahmad Zulfikar
- **NVIDIA**: add new models and capabilities — decolua
- **ClinePass**: add provider support — sternelee

## Fixes
- **Usage**: dedupe streaming request-details log entries — Qin Li
- **Claude**: drop foreign thinking signatures in passthrough — decolua
- Prevent non-SSE stream pipe crash and cross-IdP account overwrites (#2244) — KunN-21
- **Kiro**: route IdC auth to regional CodeWhisperer surface (#2297) — Volodymyr Saakian
- **Kiro**: add Claude Sonnet 5 model support (#2264) — Edison42
- **Xiaomi-tokenplan**: region selector, key validation, multi-connection (#2251) — MiQieR
- **Translator**: strict Anthropic content block compliance (#2225) — Sahrul Ramadhan Hardiansyah
- **Kimchi**: strip reasoning_content echo to bound multi-turn input tokens — KunN-21
- **Kimchi**: bump User-Agent to kimchi/0.1.40 (#2256) — Ansh7473
- **Codebuddy-cn**: strip empty tool_calls arrays to preserve reasoning — zmf
- **Antigravity**: preserve Claude tool delta index (#2223) — Sutarto Jordan Chrisfivo
- **MITM**: generate root CA on server startup (#2228) — Sutarto Jordan Chrisfivo
2026-07-03 15:37:17 +07:00
hodtien
54e3245ace feat(usage): track cached tokens + correct input/output/cache cost (#2209)
Normalize every provider to one cache-inclusive convention via
canonicalizeUsage() before persist, and price cached + cache_creation as
subsets of prompt_tokens in calculateCostFromTokens() to stop
double-counting. usageRepo now delegates cost math to a single source.
Surface Cached tokens/cost across dashboard (overview, tokens, cost,
details). Merge Claude message_start cache with message_delta output so
cache counts survive. Compatible LLM nodes now allow multiple API-key
connections (key pool).

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 15:18:27 +07:00
Qin Li
960f8a0379 fix(usage): dedupe streaming request-details log entries
handleStreamingResponse and buildOnStreamComplete each generated their
own streamDetailId for what should be one logical record — the
placeholder row (0 tokens) and the final row (real usage) never shared
an id, so the DB's ON CONFLICT(id) upsert never merged them, leaving a
permanent 0-token stub for every streaming request.

Share the id from buildOnStreamComplete with handleStreamingResponse
so both writes hit the same row.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 15:14:49 +07:00
Rafli Ahmad Zulfikar
5cc4f222f8 feat(codex): show reset credit expiry details (#2290)
Add read-only GET to inspect per-credit reset inventory (status, granted,
expiry, remaining) with a Quota Tracker modal. DRY the route via shared
connection/refresh helpers; keep existing consume POST unchanged.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 15:07:45 +07:00
decolua
cd557a2552 fix(claude): drop foreign thinking signatures in passthrough
Combo mixes models, so non-Claude thinking signatures leak into
conversation history. Native passthrough forwarded them verbatim and
Anthropic rejected the request. Validate signatures and drop invalid
thinking blocks, re-inserting a placeholder when tool_use requires one.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 15:06:19 +07:00
decolua
ced51ed62f feat(nvidia): add new models and capabilities for NVIDIA provider
- Updated capabilities for NVIDIA models to enforce OpenAI-compatible reasoning formats.
- Added new models: MiniMax M3, GLM 5.2, DeepSeek V4 Pro, DeepSeek V4 Flash, Kimi K2.6, and Nemotron 3 Ultra to the NVIDIA registry.

This enhances the provider's functionality and aligns with OpenAI standards.
2026-07-03 12:15:58 +07:00
KunN-21
cb0135b695 fix: prevent non-SSE stream pipe crash and cross-IdP account overwrites (#2244)
- streamingHandler: when upstream returns non-SSE/JSON (e.g. Cloudflare
  5xx HTML), read body, sanitize <title>, notify streamController and
  return a clean JSON error instead of crashing the pipe.
- connectionsRepo: dedup OAuth connections on (email + username) so
  cross-IdP accounts sharing an email no longer overwrite each other;
  workspace providers keep workspace-id matching.
- kimchi: bump User-Agent to 0.1.50, add svg asset + browser-login
  service, and 21 unit tests.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 11:11:07 +07:00
Volodymyr Saakian
abc0add031 fix(kiro): route IdC auth to regional CodeWhisperer surface (#2297)
IAM Identity Center (authMethod=idc) tokens failed every request with 403
"bearer token invalid". Treat idc like api_key/external_idp:

- executors/kiro.js: route idc to *.amazonaws.com CodeWhisperer surface,
  region-aware from credentials.region instead of hardcoded us-east-1.
- openai-to-kiro.js / claude-to-kiro.js: send resolved profileArn or empty
  for idc/external_idp, never the shared builder-id placeholder ARN.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 11:06:40 +07:00
MiQieR
9102c4c6d8 fix(xiaomi-tokenplan): region selector, key validation, multi-connection (#2251)
- Add top-level regions array so Add/Edit modals render region <Select>
- EditConnectionModal: load/persist region generically for region-aware providers
- validate: accept 403 for xiaomi-tokenplan valid keys, add 8s fetch timeout
- Remove single-connection guard for compatible/embedding nodes

Co-authored-by: MiQieR <122154116+MiQieR@users.noreply.github.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 11:03:18 +07:00
Sahrul Ramadhan Hardiansyah
ce6120ce7b fix(translator): strict Anthropic content block compliance (#2225)
Filter empty text blocks from thoughtSignature-only parts, preserve
tool_calls when functionResponse and functionCall coexist in the same
content, and skip empty regular text parts before they reach Claude.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 10:58:39 +07:00
KunN-21
7afaecd617 fix(kimchi): strip reasoning_content echo to bound multi-turn input tokens
Clients echo full message history each turn including reasoning_content,
which the Kimchi OpenAI gateway counts as input tokens. Multi-turn convos
balloon to 100k+ tokens and the model returns empty content.

KimchiExecutor.transformRequest now strips reasoning_content from assistant
messages when it exceeds an 8-char threshold, preserving the 1-char
placeholder injectReasoningContent sets and keeping content intact.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 10:58:31 +07:00
Edison42
a5363b83b5 fix(kiro): add Claude Sonnet 5 model support (#2264)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 10:54:15 +07:00
sternelee
b08751c4ea feat(clinepass): add ClinePass provider support
Register clinepass provider (OAuth + API-key) using Cline's
OpenAI-compatible API with 10 curated models, live /v1/models
resolver, refreshCline-based token refresh with workos: prefix,
and dashboard OAuth login handler.

Reference: https://github.com/jellydn/pi-clinepass-provider
Closes #2261

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 10:53:41 +07:00
Ansh7473
76752a4396 fix(kimchi): bump User-Agent to kimchi/0.1.40 (#2256)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 10:49:53 +07:00
zmf
602ee4054b fix(codebuddy-cn): strip empty tool_calls arrays to preserve reasoning
CodeBuddy CN includes "tool_calls": [] in every SSE streaming delta.
@ai-sdk/openai-compatible checks delta.tool_calls != null — an empty
array passes ([] != null is true in JS), triggering premature
reasoning-end on every reasoning chunk (0/1ms durations in OpenCode).

Strip empty tool_calls arrays in passthrough before hasValuableContent.
Zero side-effect: real tool_calls always have at least one element.

Closes #2176

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-01 09:39:31 +07:00
Sutarto Jordan Chrisfivo
8f81f17b99 fix(antigravity): preserve Claude tool delta index (#2223)
Gemini response translation wrote OpenAI-shaped bookkeeping into the
shared state.toolCalls map, which the downstream openai-to-claude
translator uses for Claude block metadata. That pre-population skipped
blockIndex creation, so Anthropic input_json_delta events lost index.

Track Gemini function calls via state.geminiToolCallCount instead,
leaving state.toolCalls clean for the Claude translator.

Closes #2218

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-01 09:36:43 +07:00
Sutarto Jordan Chrisfivo
182c849979 fix(mitm): generate root ca on server startup (#2228)
Direct MITM server startup read rootCA.key/.crt immediately and exited
when either was missing, bypassing the manager.js CA setup path.

- generate Root CA from server.js when key/cert is missing
- make generateRootCA()/generateCert() synchronous to avoid a startup
  race before readFileSync
- add unit test covering synchronous Root CA creation

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-01 09:30:52 +07:00
decolua
0b3c794075 # v0.5.15 (2026-06-29)
## Features
- Add Kimchi OAuth provider — Nant361
- Refine Qwen vision/video + thinking model patterns — decolua
- Opt-in Codex auto-ping quota keep-alive — Emirhan

## Fixes
- **Responses**: handle response.done terminal events (#2142) — rifuki
- **Headroom**: skip unsafe responses tool history (#2132) — Sutarto Jordan Chrisfivo
- **Translator**: map mid-conversation system message to user (claude→openai) — decolua
- **Gemini**: normalize contents to prevent 400 invalid_argument (#2192) — warelik
- **Gemini**: backfill thoughtSignature + suppress stream done sentinel — WARELIK
- **Alicode**: preserve cache_control for DashScope providers (#2069) — Rex
- **Antigravity**: strip deprecated/readOnly/writeOnly from tool schemas — iletai, Yudhistira-Official
- **CodeBuddy CN**: show bonus packs as one-time, not monthly-replenishing — whale9820
- **Kiro**: strip leaked <thinking> tags from content stream (#2158) — hamsa0x7
- **Tray**: make Windows context menu DPI-aware — Emirhan
- **Kilocode**: expose full gateway catalog in combo model picker — jellylarper
- **OpenCode**: fix Go GLM — decolua
2026-06-29 16:23:33 +07:00
rifuki
a9785a5f70 fix(responses): handle response.done terminal events (#2142)
Treat response.done as a terminal OpenAI Responses stream event so
passthrough streams ending with response.done are not flagged incomplete
and no synthetic response.failed is emitted. Restore the data: [DONE]
sentinel for same-format Responses passthrough streams.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 16:05:29 +07:00
Sutarto Jordan Chrisfivo
373850ee36 fix(headroom): skip unsafe responses tool history (#2132)
Guard openai-responses compression: skip Headroom when body.input
contains non-message items (function_call, function_call_output,
reasoning) to preserve the Responses contract instead of collapsing
them into chat messages.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:59:21 +07:00
decolua
749c2e3f9c fix(translator): map mid-conversation system message to user in claude-to-openai
Claude Code chèn role:system cuối messages[], trước đây bị map thành assistant
khiến hội thoại không kết thúc bằng user → provider OpenAI-compat (LiteLLM)
dịch ngược Anthropic trả 400 "assistant message prefill". Map system -> user
và wrap <system-reminder> để giữ ngữ nghĩa instruction.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:53:26 +07:00
decolua
7fa2e7f029 feat(capabilities): refine Qwen vision/video and thinking model patterns
Add qwen omni (audio/video input), qwen3.5/3.6/3.7 (native vision/video),
and mark qwen coder & max as text-only reasoning models.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:51:56 +07:00
warelik
8d1db46beb fix(gemini): normalize contents to prevent 400 invalid_argument (#2192)
Merge adjacent same-role blocks and strip empty parts before sending to
Gemini, avoiding 400 INVALID_ARGUMENT on consecutive same-role messages.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:38:03 +07:00
Rex
9e3866658a fix(alicode): preserve cache_control for DashScope providers (#2069)
Opt-in quirk preserveCacheControl keeps cache_control on content blocks
for alicode/alicode-intl, enabling DashScope prompt caching. signature
is always stripped; all other providers unchanged.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:29:49 +07:00
Nant361
8a664d619d feat(kimchi): add Kimchi OAuth provider support
Add Kimchi as a browser-token OAuth provider routed through its
OpenAI-compatible gateway. Discover live models for /v1/models and
provider models, normalize Claude-compatible requests, and wire up
provider connection tests.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:29:17 +07:00
WARELIK
2d94fffe3b fix(gemini): backfill thoughtSignature and suppress stream done sentinel
Backfill DEFAULT_THINKING_AG_SIGNATURE onto functionCall parts missing it
(client history replay) and on Claude tool_use blocks, fixing 400
INVALID_ARGUMENT from Gemini-family APIs. Suppress the OpenAI-style
data: [DONE] sentinel for antigravity/gemini/vertex to avoid parser crashes.

Fixes #2193.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:28:03 +07:00
Yudhistira-Official
319caa2d7b fix(antigravity): strip 'deprecated' from tool schemas before Gemini
Gemini rejects the non-standard 'deprecated' keyword in nested tool
schemas with INVALID_ARGUMENT (400). Add it to UNSUPPORTED_SCHEMA_CONSTRAINTS
alongside 'optional' so it gets stripped during translation.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:28:02 +07:00
whale9820
95bfc64f06 fix(codebuddy-cn): show bonus packs as one-time, not monthly-replenishing
CodeBuddy CN bonus packs ("Bonus Pack N") are one-shot credits whose
CycleEndTime equals DeductionEndTime — they expire for good and never
replenish. The dashboard rendered their resetAt as "Reset in Xd",
implying a monthly refill.

Tag bonus packs recurring:false (refill packs recurring:true) in the
usage handler, forward the flag through parseQuotaData, and word the
quota table / progress bar as "Expires in" / "Expires at" for
one-shot packs.
2026-06-29 15:22:47 +07:00
hamsa0x7
eff81b1242 fix(kiro): strip leaked <thinking> tags from content stream (#2158)
CodeWhisperer leaks literal <thinking> blocks into assistantResponseEvent,
duplicating reasoning already routed via reasoningContentEvent. Track
inThinking state to strip these tags during SSE transform, handling split
chunks across tag boundaries.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:14:15 +07:00
Emirhan
b66b5c68ce feat(quota): add opt-in Codex auto-ping
Generalize Claude 5h auto-ping into a provider-generic scheduler and add
opt-in Codex auto-ping that warms the next 5h window via a tiny gpt-5.5
request when session.resetAt slides. Default off, per-connection toggle,
failure cooldown, blocking-quota skip, drains stream before success.

Closes #2107

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:13:17 +07:00
Emirhan
fc8722e897 fix(tray): make Windows context menu DPI-aware
Set process DPI awareness (Per-Monitor V2 with fallbacks) and enable
WinForms visual styles before creating the tray NotifyIcon, so the
Windows context menu renders sharply on DPI-scaled displays.

Closes #2161

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:12:24 +07:00
jellylarper
713c563765 fix(kilocode): expose full gateway catalog in combo model picker
Add modelsFetcher + passthroughModels so the dynamic Kilo Gateway
catalog surfaces in the combo model picker, matching openrouter.js.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:12:18 +07:00
iletai
3d20a4ccd2 fix(antigravity): strip deprecated/readOnly/writeOnly from tool schemas
Gemini/Antigravity generateContent rejects the JSON Schema annotation
keywords deprecated, readOnly, writeOnly with a 400 INVALID_ARGUMENT.
MCP tool schemas (e.g. Claude Code) commonly set deprecated:true, making
every request with such a tool fail. Add them to
UNSUPPORTED_SCHEMA_CONSTRAINTS so cleanJSONSchemaForAntigravity removes
them recursively before the request is sent.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-29 15:12:17 +07:00
decolua
526235872a Fix OpenCode Go GLM 2026-06-29 15:00:03 +07:00
386 changed files with 75336 additions and 5815 deletions

20
.gitignore vendored
View File

@@ -1,4 +1,5 @@
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files. # See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
# dependencies # dependencies
/node_modules /node_modules
/.pnp /.pnp
@@ -8,8 +9,10 @@
!.yarn/plugins !.yarn/plugins
!.yarn/releases !.yarn/releases
!.yarn/versions !.yarn/versions
# testing # testing
/coverage /coverage
# next.js # next.js
/.next/ /.next/
/.next-cli-build/ /.next-cli-build/
@@ -19,22 +22,28 @@ product
# production # production
/build /build
.idea/ .idea/
# misc # misc
.DS_Store .DS_Store
*.pem *.pem
# debug # debug
npm-debug.log* npm-debug.log*
yarn-debug.log* yarn-debug.log*
yarn-error.log* yarn-error.log*
.pnpm-debug.log* .pnpm-debug.log*
# env files (can opt-in for committing if needed) # env files (can opt-in for committing if needed)
.env* .env*
!.env.example !.env.example
# vercel # vercel
.vercel .vercel
# typescript # typescript
*.tsbuildinfo *.tsbuildinfo
next-env.d.ts next-env.d.ts
.bin/* .bin/*
data/ data/
logs/* logs/*
@@ -52,18 +61,23 @@ Thanks.md
PUBLIC.en.md PUBLIC.en.md
PR/* PR/*
package-lock.json package-lock.json
#Ignore vscode AI rules #Ignore vscode AI rules
.github/instructions/codacy.instructions.md .github/instructions/codacy.instructions.md
README1.md README1.md
deploy*.sh deploy*.sh
ecosystem.config.* ecosystem.config.*
scripts/agSniffer/* scripts/agSniffer/*
gitbooks/* gitbooks/*
gitbook/README.md gitbook/README.md
# Refactor backup reference (do not bundle/lint) # Refactor backup reference (do not bundle/lint)
open-sse.old/ open-sse.old/
.graphifyignore .graphifyignore
graphify-out/* graphify-out/*
# Local-only working dirs (notes, vendored repos, scripts, skills) # Local-only working dirs (notes, vendored repos, scripts, skills)
.claude/ .claude/
.docs/ .docs/
@@ -72,8 +86,6 @@ graphify-out/*
.codegraph/ .codegraph/
.PR/ .PR/
.next-analyze/* .next-analyze/*
# CommandCode CLI local state (auth/taste/projects)
.commandcode/
# Pi subagent run artifacts # Kiro local workspace state
.pi-subagents/ .kiro/

View File

@@ -1,3 +1,168 @@
# v0.5.69 (2026-09-05)
## Features
- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities
- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`)
- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys
- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819)
- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through
- **CLI tools**: replace Copilot MITM with VS Code extension setup guide
- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace
## Fixes
- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792)
- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813)
- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797)
- **Dashboard**: dynamic mode label for local/remote detection (#3801)
- **Codex**: format reset credit API errors cleanly (#3778)
- **Security**: guard cowork MCP tools probe against SSRF (#3783)
- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800)
- **Logger**: suppress noisy background token refresh logs
- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory
# v0.5.65 (2026-09-03)
## Features
- **Fetch**: add Ollama Cloud web fetch provider
- **Gemini / Antigravity**: add Gemini 3.8 Flash support and bump IDE fingerprint to 2.11.0
- **Claude**: add Claude Fable 5.1 support (adaptive thinking with `output_config.effort`), bump Claude Code fingerprint to 2.1.258 for new-model access
- **Providers**: add client-side status filter (All / Active / Inactive / No connection) on the Providers dashboard; add max height and scroll for connection list
- **Providers & Models**: streamline tokenrouter model catalog down to 22 flagship/newest models and add missing provider icons; refresh Codebuddy-CN catalog (add hy4-preview/hy3/glm-5.3/kimi-k3-1, drop EOL glm-5.0/glm-4.7)
- **Models**: capability toggles (vision, reasoning) when adding custom models with upsert and live caps refresh
- **CLI tools**: support saving and managing custom API key presets
- **Quota**: add usage and rate-limit tracking for Groq via `x-ratelimit-*` headers
- **i18n**: complete Indonesian translation (1391 keys)
## Fixes
- **Security**: close SSRF guard bypasses in `ssrfGuard.js` (alternate IPv6 encodings, hostname trailing dots, wildcard DNS resolution check, safe redirect handling) (#3714)
- **Model markers**: strip the `[1m]` context marker Claude Code appends to model names (`claude-opus-5[1m]`) preventing model resolution failures (#3690)
- **Claude**: drop `server_tool_use` blocks carrying foreign IDs to avoid Anthropic 400 rejections; never anchor cache breakpoints on `defer_loading` tools (#3567)
- **Antigravity**: strike-break optimistic quota readings that keep 429ing by blocking the connection+model pair for 15m after 3 strikes (#3681); preserve client identity on model catalog requests (#3414)
- **Auth**: protect root `/responses` rewrite requiring API key validation in dashboardGuard
- **Chat & Docker**: return 503 Service Unavailable when all credentials are rate-limited; explicitly bundle `node-machine-id` into standalone Docker runtime image
- **OpenCode**: route Muse Spark models to `/zen/v1/responses` and declare vision support; filter inactive free model
- **Kiro**: preserve inline images as OpenAI-compatible `image_url` parts in OpenAI MITM; remove redundant top-level `systemPrompt` from payload
- **Usage**: read Responses-shape `cached_tokens` in `extractUsageFromResponse` for non-streaming traffic
- **Models**: support single model lookup with provider-prefixed IDs (e.g. `cc/claude-sonnet-5`)
- **Translator**: route Gemini thinking through `reasoning_effort` on OpenAI-compatible wire; convert `prefixItems` and ensure array items in Gemini schema sanitizer
- **UI**: apply persisted theme before first paint to prevent flash on reload; translate combo vision adapter label
# v0.5.59 (2026-08-29)
## Features
- **Search**: new web search providers — Antigravity (Google Search grounding
on the existing OAuth account pool, citations keyed and merged by URL) and
Xquik (X search with `x-api-key` auth, cursor pagination, credit-based
usage), both on `POST /v1/search`. Based on #3437 by @Nautilaceae
- **Search**: ollama-search and zai-search borrow a chat provider's API key
instead of requiring their own connection, driven by a new
`credentialFallback` registry field. zai-search later folded into the `glm`
provider itself so the web search page shows the shared connection
- **Models**: daily background sync of model capabilities from models.dev —
modalities keyed by model id (majority of sources must declare one),
context/output limits keyed by provider + model, strictly additive and
sitting below the hand-written tables. ETag + mtime cache, 60s startup
delay, `MODEL_CATALOG_SYNC=off` to disable
- **Models**: add GLM-5.3-Flash (1M context, natively multimodal), DeepSeek
V4 Vision, Grok 4.5/4.6 (500k context); correct glm-4.6v/4.5v video input
and output limits, backfill glm-4.6v on glm-cn
- **Usage**: show the Zed plan quota on the dashboard — plan, edit
predictions, hosted model requests and billing-cycle reset; unlimited rows
render as "N used · Unlimited"
- **Usage**: track GPT-5.3-Codex-Spark quota windows (spark_session /
spark_weekly) from the Codex usage response (#3431)
- **Antigravity**: quota-aware routing — on 409/429 fetch live quota for the
exact per-model resetAt and skip only the exhausted account/model pair;
report the earliest reset when every account is blocked (#3561)
- **Antigravity**: map image `size` to the aspect-ratio model suffix (-WxH);
add the Gemini 3.7 Flash tiers to MITM defaultModels so they show up in
the dashboard model-mapping table
- **Dashboard**: bulk import Grok CLI accounts from JSON — paste an array or
drag-drop multiple .json files, all OAuth connections created in a single
call, mirroring the codex flow
- **CLI tools**: endpoint presets shared across every tool card through one
live-resyncing store, instead of per-card localStorage copies that never
saw each other's saved endpoints
- **Token Saver**: configurable compression timeout (`headroomTimeoutMs`) —
the fixed 3000 ms made busy machines time out and send inconsistently
compressed bodies, hurting prompt caching
- **i18n**: pt-BR expanded to 1132 terms
## Fixes
- **Claude Code**: add Claude Fable 5.1 and advertise Claude Code 2.1.258 in
both the request header and billing identity; use its permanent adaptive-thinking
mode with `output_config.effort`
- **Stream**: record usage when a client closes on the terminal event — the
Responses API has no [DONE] sentinel, so codex closed the socket on
`response.completed` and cancelled the reader before flush() ran its usage
side effects; the tail now lives in a once-guarded finalizeStream(). Also
stop logging a disconnect for every completed Responses call
- **Stream**: parse the trailing NDJSON line an Ollama stream leaves behind
without a closing newline — the final chunk carrying `done_reason` and the
token counts was dropped
- **Session**: read the Claude Code session id from the
`x-claude-code-session-id` header — `metadata.user_id` is dropped by
Responses translation, splitting one conversation across several
`prompt_cache_key` values and missing the upstream prefix cache
- **Usage**: preserve nested `cached_tokens` — the top-level-only read
persisted `cached_tokens: 0` for every Responses-format provider (codex,
grok-cli, …), billing cache hits at the full input rate
- **Usage**: GLM quotas accept CREDIT_LIMIT plans and multi-interval windows
(5h session / 7d weekly) instead of overwriting a single "session" key
- **Models**: the catalog sync no longer erases its own output — deltas were
measured against the previous run's writes (the second run cut `providers`
from 20 entries to 5); one vote per provider in the modality tally, ETag
restored from file on startup, and the worker thread dropped after the
bundler rewrote its path into a module-not-found error
- **Executor**: CommandCode returns errors as a `type:"error"` event inside
an HTTP 200 NDJSON stream — peek the first events before committing, abort
and return a real 4xx/5xx so combo/account fallback triggers instead of
streaming the error text as content
- **Search**: scope failure locks on the credential-fallback path — a failing
search locked `modelLock___all` and took the shared glm key offline for
chat as well; locks are now attributed to the connection's owner and
scoped to `websearch:<provider>`
- **Providers**: connection tests get a 15s AbortSignal timeout instead of
hanging and exhausting the browser socket pool; guard undefined provider
names on the providers page
- **Antigravity**: sanitize competing-client branding via a config-driven
rule table (Zed's Claude-agent prompt, opencode → antigravity) — upstream
answers 429 Quota Exhausted. Applied in the executor so the shared
openai-to-gemini translator leaves gemini/vertex/zed untouched
- **MiniMax**: preserve images on the sourceFormat-matched OpenAI transport
— MiniMax-M3 resolved a Claude-shaped body posted to the OpenAI endpoint,
silently dropping `image_url` blocks (#3418)
- **Claude**: decloak tool names in same-format streaming passthrough —
OAuth-cloaked names (CLAUDE_TOOL_SUFFIX) leaked to the client and every
tool call was rejected as unknown
- **Tools**: default a missing `tools[].type` to "custom" on Claude-format
requests — strict Anthropic-compatible gateways (MiniMax) reject the
request with 400 otherwise
- **Translator**: zai thinkingFormat sends the top-level `reasoning_effort`
object GLM-5.2+ requires — every GLM-5.x request ran at the model default
(max); gated on GLM-5.2+ since older GLM does not read it (#2721)
- **RTK**: system prompt injection matches each target wire format
(Chat/Responses/Claude/Gemini/Kiro) and is exact-idempotent across retries,
so distinct prompts sharing a long prefix are no longer collapsed (#3202).
Also set the diagnostic before the silent null return on Responses
translation failure so the panel is no longer blank
- **OpenCode**: route muse-spark through /zen/v1/responses (it 500s on
chat/completions), normalizing the Chat fields the Responses API rejects
and clamping max/ultra effort to xhigh
- **CLI**: install better-sqlite3 without build tools on Node 22+ (N-API
13.0.3 ships per-platform prebuilds, `--ignore-scripts` skips the implicit
node-gyp build); Node < 22 stays on 12.6.2, working installs untouched
- **CLI tools**: send the API key Codex actually reads —
`[model_providers.9router.http_headers]` instead of auth.json (which left
every request 401 and clobbered an existing ChatGPT login); subagent model
moved to `agents.default_subagent_model`
- **OAuth**: refresh Cline tokens with the extension JSON contract
- **Dashboard**: clamp the API key mask length — keys shorter than 8 chars
threw RangeError and crashed the media-provider detail page
- **UI**: wait for the Material Symbols font itself before revealing icons —
`document.fonts.ready` resolved before the 4MB woff2 even started loading,
leaving icons blank until a second load
# v0.5.55 (2026-08-14) # v0.5.55 (2026-08-14)
## Features ## Features
@@ -625,4 +790,4 @@
# v0.4.46 (2026-05-15) # v0.4.46 (2026-05-15)
## Breaking Changes ## Breaking Changes
- Tunnel public URL changed — old tunnel links no longer work, please reconnect to get the new URL - Tunnel public URL changed — old tunnel links no longer work, please reconnect to get the new URL

View File

@@ -89,3 +89,13 @@ Pre-translate hooks that compress `tool_result` content in-place to cut tokens.
- Security-sensitive env: `JWT_SECRET` (session cookie), `INITIAL_PASSWORD` (default `123456` — must override), `API_KEY_SECRET`, `MACHINE_ID_SALT`. Full env contract in `.env.example` and ARCHITECTURE.md's env matrix. - Security-sensitive env: `JWT_SECRET` (session cookie), `INITIAL_PASSWORD` (default `123456` — must override), `API_KEY_SECRET`, `MACHINE_ID_SALT`. Full env contract in `.env.example` and ARCHITECTURE.md's env matrix.
- Binary/protobuf upstreams (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — they're handled inside their own executor, not the translator. - Binary/protobuf upstreams (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — they're handled inside their own executor, not the translator.
- Versioning: root and `cli/` are versioned independently; changes are logged in `CHANGELOG.md`. Commit style is Conventional Commits (`fix(translator): …`, `feat(...)`). - Versioning: root and `cli/` are versioned independently; changes are logged in `CHANGELOG.md`. Commit style is Conventional Commits (`fix(translator): …`, `feat(...)`).
<!-- BEGIN:nextjs-agent-rules -->
# This is NOT the Next.js you know
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` (resolved from this file's directory; in monorepos the `next` package may not be visible from the repo root) before writing any code. Heed deprecation notices.
This block is written and re-added by `next dev` — verify at `node_modules/next/dist/server/lib/generate-agent-files.js`. Removing it from a diff only re-creates the uncommitted change; committing it with your work keeps the tree clean.
<!-- END:nextjs-agent-rules -->

View File

@@ -2,14 +2,15 @@
ARG NODE_IMAGE=node:22-alpine ARG NODE_IMAGE=node:22-alpine
FROM ${NODE_IMAGE} AS base FROM ${NODE_IMAGE} AS base
WORKDIR /app WORKDIR /app
# CN mirror for apk (used by builder and runner stages)
RUN sed -i 's|dl-cdn.alpinelinux.org|mirrors.aliyun.com|g' /etc/apk/repositories
FROM base AS builder FROM base AS builder
RUN apk --no-cache upgrade && apk --no-cache add python3 make g++ linux-headers RUN apk --no-cache upgrade && apk --no-cache add python3 make g++ linux-headers
COPY package.json ./ COPY package.json ./
RUN --mount=type=cache,target=/root/.npm \ RUN npm install --registry=https://registry.npmmirror.com
npm install
COPY . ./ COPY . ./
ENV NEXT_TELEMETRY_DISABLED=1 ENV NEXT_TELEMETRY_DISABLED=1
@@ -40,6 +41,8 @@ COPY --from=builder /app/node_modules/next ./node_modules/next
# sql.js loads dist/sql-wasm.wasm by path at runtime; tracing only follows JS imports, # sql.js loads dist/sql-wasm.wasm by path at runtime; tracing only follows JS imports,
# so the last-resort DB driver would abort with ENOENT on the missing binary. # so the last-resort DB driver would abort with ENOENT on the missing binary.
COPY --from=builder /app/node_modules/sql.js ./node_modules/sql.js COPY --from=builder /app/node_modules/sql.js ./node_modules/sql.js
# node-machine-id is createRequire-loaded at runtime; tracing omits it.
COPY --from=builder /app/node_modules/node-machine-id ./node_modules/node-machine-id
RUN mkdir -p /app/data && chown -R node:node /app && \ RUN mkdir -p /app/data && chown -R node:node /app && \
mkdir -p /app/data-home && chown node:node /app/data-home && \ mkdir -p /app/data-home && chown node:node /app/data-home && \

1407
bun.lock Normal file

File diff suppressed because it is too large Load Diff

View File

@@ -6,7 +6,13 @@ const fs = require("fs");
const os = require("os"); const os = require("os");
const path = require("path"); const path = require("path");
const BETTER_SQLITE3_VERSION = "12.6.2"; // Gate the pinned version by Node major, mirroring src/lib/db/driver.js gating
// style: 13.x is N-API and ships per-platform prebuilds inside the package, so
// it needs no ABI-specific download. It requires Node >= 22; older runtimes stay
// on 12.6.2, which fetches an ABI-specific binary via prebuild-install.
const [NODE_MAJOR] = process.versions.node.split(".").map(Number);
const USE_NAPI_BUILD = NODE_MAJOR >= 22;
const BETTER_SQLITE3_VERSION = USE_NAPI_BUILD ? "13.0.3" : "12.6.2";
const SQL_JS_VERSION = "1.14.1"; const SQL_JS_VERSION = "1.14.1";
function getDataDir() { function getDataDir() {
@@ -45,9 +51,23 @@ function hasModule(name) {
return fs.existsSync(path.join(getRuntimeNodeModules(), name, "package.json")); return fs.existsSync(path.join(getRuntimeNodeModules(), name, "package.json"));
} }
function isGlibcRuntime() {
try { return Boolean(process.report?.getReport()?.header?.glibcVersionRuntime); } catch { return true; }
}
// 12.x compiles/downloads into build/Release; 13.x ships prebuilds/<platform>-<arch>.node.
function getBetterSqliteBinary() {
const root = path.join(getRuntimeNodeModules(), "better-sqlite3");
const platform = process.platform === "linux" && !isGlibcRuntime() ? "linuxmusl" : process.platform;
return [
path.join(root, "build", "Release", "better_sqlite3.node"),
path.join(root, "prebuilds", `${platform}-${process.arch}.node`),
].find((file) => fs.existsSync(file));
}
function isBetterSqliteBinaryValid() { function isBetterSqliteBinaryValid() {
const binary = path.join(getRuntimeNodeModules(), "better-sqlite3", "build", "Release", "better_sqlite3.node"); const binary = getBetterSqliteBinary();
if (!fs.existsSync(binary)) return false; if (!binary) return false;
try { try {
const fd = fs.openSync(binary, "r"); const fd = fs.openSync(binary, "r");
const buf = Buffer.alloc(4); const buf = Buffer.alloc(4);
@@ -91,6 +111,7 @@ function runNpmInstall({ cwd, pkgs, extraArgs = [], timeout = 180000 }) {
function npmInstall(pkgs, opts = {}) { function npmInstall(pkgs, opts = {}) {
const cwd = ensureRuntimeDir(); const cwd = ensureRuntimeDir();
const extra = opts.optional ? ["--no-save"] : []; const extra = opts.optional ? ["--no-save"] : [];
if (opts.ignoreScripts) extra.push("--ignore-scripts");
if (!opts.silent) console.log("⏳ Installing SQLite engine (first run)..."); if (!opts.silent) console.log("⏳ Installing SQLite engine (first run)...");
const res = runNpmInstall({ cwd, pkgs, extraArgs: extra, timeout: opts.timeout || 180000 }); const res = runNpmInstall({ cwd, pkgs, extraArgs: extra, timeout: opts.timeout || 180000 });
if (!res.ok && !opts.silent) { if (!res.ok && !opts.silent) {
@@ -129,7 +150,10 @@ function ensureSqliteRuntime({ silent = false } = {}) {
return { betterSqlite: true, sqlJs: sqlJsOk }; return { betterSqlite: true, sqlJs: sqlJsOk };
} }
const ok = npmInstall([`better-sqlite3@${BETTER_SQLITE3_VERSION}`], { optional: true, silent }); // npm injects an implicit `node-gyp rebuild` for any package carrying a
// binding.gyp, which would demand build tools even though 13.x already bundles
// the binary — skip scripts so the bundled prebuild is used as-is.
const ok = npmInstall([`better-sqlite3@${BETTER_SQLITE3_VERSION}`], { optional: true, silent, ignoreScripts: USE_NAPI_BUILD });
return { return {
betterSqlite: ok && hasModule("better-sqlite3") && isBetterSqliteBinaryValid(), betterSqlite: ok && hasModule("better-sqlite3") && isBetterSqliteBinaryValid(),
sqlJs: sqlJsOk, sqlJs: sqlJsOk,

View File

@@ -1,6 +1,6 @@
{ {
"name": "9router", "name": "9router",
"version": "0.5.55", "version": "0.5.69",
"description": "9Router CLI - Start and manage 9Router server", "description": "9Router CLI - Start and manage 9Router server",
"bin": { "bin": {
"9router": "./cli.js" "9router": "./cli.js"
@@ -16,7 +16,7 @@
"scripts": { "scripts": {
"dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js", "dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js",
"build": "node scripts/build-cli.js", "build": "node scripts/build-cli.js",
"pack:cli": "npm run build && npm pack --pack-destination ../..", "pack:cli": "npm run build && npm pack --pack-destination ..",
"publish:cli": "npm run build && npm publish", "publish:cli": "npm run build && npm publish",
"postinstall": "node hooks/postinstall.js", "postinstall": "node hooks/postinstall.js",
"prepublishOnly": "npm run build" "prepublishOnly": "npm run build"

View File

@@ -53,6 +53,9 @@ const PROVIDER_MODELS = {
{ id: "glm-4.7" }, { id: "glm-4.7" },
], ],
ag: [ ag: [
{ id: "gemini-3.8-flash-high" },
{ id: "gemini-3.8-flash-medium" },
{ id: "gemini-3.8-flash-low" },
{ id: "gemini-3.7-flash-high" }, { id: "gemini-3.7-flash-high" },
{ id: "gemini-3.7-flash-medium" }, { id: "gemini-3.7-flash-medium" },
{ id: "gemini-3.7-flash-low" }, { id: "gemini-3.7-flash-low" },
@@ -101,6 +104,8 @@ const PROVIDER_MODELS = {
{ id: "claude-3-5-sonnet-20241022" }, { id: "claude-3-5-sonnet-20241022" },
], ],
gemini: [ gemini: [
{ id: "gemini-3.8-flash" },
{ id: "gemini-3.7-flash" },
{ id: "gemini-3.6-flash" }, { id: "gemini-3.6-flash" },
{ id: "gemini-3.5-flash-lite" }, { id: "gemini-3.5-flash-lite" },
{ id: "gemini-3-pro-preview" }, { id: "gemini-3-pro-preview" },

View File

@@ -0,0 +1,261 @@
# OpenCode Go Session Header Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** Send a stable, conversation-scoped `x-opencode-session` header on every OpenCode Go request and install the patched CLI locally.
**Architecture:** Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. `chatCore` passes the provider-scoped session resolved from the original request plus the detected client tool; the executor derives a request-local upstream session and delegates all existing transport, authentication, retry, and proxy behavior to `DefaultExecutor`.
**Tech Stack:** Node.js ESM, Vitest, Next.js, npm CLI packaging, GitHub CLI.
## Global Constraints
- Apply the header to OpenCode Go chat completions, Claude Messages, and OpenAI Responses transports.
- Preserve a valid native `x-opencode-session`; hash all translated non-OpenCode identities to `ses_<32 lowercase hex>`.
- Namespace translated identities by detected client tool, using `generic` when unknown.
- Do not keep mutable per-request session state on the executor singleton or mutate the caller's credentials object.
- Do not change OpenCode Go models, routing, reasoning, tool behavior, dependencies, or unrelated providers.
- Reuse upstream issue #3759 instead of creating a duplicate issue.
---
### Task 1: Add Failing OpenCode Go Session Tests
**Files:**
- Create: `tests/unit/opencode-go-session.test.js`
**Interfaces:**
- Consumes: `getExecutor(provider)` and `DefaultExecutor.buildHeaders(credentials, stream, url, model)`.
- Produces: the required public behavior for `OpenCodeGoExecutor.prepareRequestCredentials({ body, credentials, providerSessionId, clientTool })` and `OpenCodeGoExecutor.execute(args)`.
- [ ] **Step 1: Write the failing tests**
Create a Vitest suite that mocks `proxyAwareFetch`, obtains `getExecutor("opencode-go")`, and asserts:
```js
const prepared = executor.prepareRequestCredentials({
body: { messages: [{ role: "user", content: "hello" }] },
credentials: { apiKey: "test-key", connectionId: "conn-a", rawHeaders: {} },
providerSessionId: "conversation-a",
clientTool: "claude",
});
expect(prepared).not.toBe(credentials);
expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/);
expect(credentials).not.toHaveProperty("_opencodeGoSession");
```
Cover native header preservation, stable values across all three runtime transports, different conversation IDs, different client tools using the same ID, connection fallback, no singleton state, no header on `DefaultExecutor("openai")`, and the final fetch headers returned by `execute()`.
- [ ] **Step 2: Run the focused test and verify RED**
Run:
```bash
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
```
Expected: FAIL because `getExecutor("opencode-go")` still returns `DefaultExecutor` and `prepareRequestCredentials` does not exist.
- [ ] **Step 3: Commit the failing test**
```bash
git add tests/unit/opencode-go-session.test.js
git commit -m "test: cover OpenCode Go session headers"
```
### Task 2: Implement the Dedicated Executor
**Files:**
- Create: `open-sse/executors/opencode-go.js`
- Modify: `open-sse/executors/index.js`
**Interfaces:**
- Consumes: `DefaultExecutor`, `resolveSessionId()`, request `credentials.rawHeaders`, `providerSessionId`, and `clientTool`.
- Produces: `OpenCodeGoExecutor`, `prepareRequestCredentials()`, and an `execute()` override that delegates with cloned credentials.
- [ ] **Step 1: Add the minimal executor implementation**
Implement these rules:
```js
function translatedSessionId(sessionId, clientTool) {
const digest = crypto
.createHash("sha256")
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
.digest("hex")
.slice(0, 32);
return `ses_${digest}`;
}
```
`prepareRequestCredentials()` must read a case-insensitive native
`x-opencode-session` with the same non-empty, 256-character cap used by the
session manager. Otherwise it uses `providerSessionId` or calls
`resolveSessionId({ headers, body, connectionId, scope: "opencode-go" })`, then
returns `{ ...credentials, _opencodeGoSession: value }`.
`execute(args)` must call `prepareRequestCredentials(args)` and delegate using
`super.execute({ ...args, credentials: prepared })`. `buildHeaders()` must call
`super.buildHeaders()` and add the prepared session, with a connection-scoped
fallback for direct callers.
Register `new OpenCodeGoExecutor()` under `"opencode-go"` and export the class.
- [ ] **Step 2: Run the focused test and verify partial GREEN**
Run:
```bash
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
```
Expected: executor-level tests pass; any chatCore-context assertion remains failing until Task 3.
- [ ] **Step 3: Commit the executor**
```bash
git add open-sse/executors/opencode-go.js open-sse/executors/index.js tests/unit/opencode-go-session.test.js
git commit -m "fix(opencode-go): add stable session header executor"
```
### Task 3: Pass Original Request Session Context
**Files:**
- Modify: `open-sse/handlers/chatCore.js`
- Modify: `tests/unit/opencode-go-session.test.js`
**Interfaces:**
- Consumes: existing `sessionSeed` and `clientTool` variables in `handleChatCore()`.
- Produces: `providerSessionId` and `clientTool` fields on both initial and refreshed-credential calls to `executor.execute()`.
- [ ] **Step 1: Add or enable the failing integration assertion**
Use a mocked executor or source request containing a body-only `session_id` and
assert the executor receives the provider-scoped session resolved before
translation.
- [ ] **Step 2: Run the focused test and verify RED**
Run:
```bash
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
```
Expected: FAIL because `handleChatCore()` does not pass `providerSessionId` or
`clientTool` to `executor.execute()`.
- [ ] **Step 3: Pass the request context**
Add the same fields to both executor calls:
```js
executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
```
- [ ] **Step 4: Run focused and neighboring tests**
Run:
```bash
npx vitest run --config tests/vitest.config.js \
tests/unit/opencode-go-session.test.js \
tests/unit/opencode-go-models.test.js \
tests/unit/session-manager.test.js \
tests/unit/executor-const-guard.test.js
```
Expected: PASS with zero failed tests.
- [ ] **Step 5: Commit the context wiring**
```bash
git add open-sse/handlers/chatCore.js tests/unit/opencode-go-session.test.js
git commit -m "fix(chat): forward provider session context"
```
### Task 4: Verify and Install the Local CLI Package
**Files:**
- Generated: `9router-0.5.65.tgz`
- Packaged output: `cli/app/server.js`
**Interfaces:**
- Consumes: completed source changes and existing CLI build scripts.
- Produces: a globally installed patched `9router@0.5.65`.
- [ ] **Step 1: Run source verification**
```bash
git diff --check origin/master...HEAD
npx vitest run --config tests/vitest.config.js tests/unit/
npm run build
```
Expected: every command exits zero. Record any pre-existing full-suite failures
separately rather than hiding them.
- [ ] **Step 2: Build and package the CLI**
```bash
npm --prefix cli run build
npm --prefix cli pack -- --pack-destination ..
```
Expected: `9router-0.5.65.tgz` exists and contains the patched bundled server.
- [ ] **Step 3: Replace the global npm installation**
```bash
npm install -g ./9router-0.5.65.tgz
```
Expected: `/opt/homebrew/lib/node_modules/9router/package.json` reports `0.5.65`
and the installed bundle contains `x-opencode-session` plus the new executor.
- [ ] **Step 4: Commit any required package-source adjustment**
Do not commit generated tarballs or CLI build artifacts unless the repository
already tracks and requires them.
### Task 5: Publish the Upstream Pull Request
**Files:**
- No additional source files unless verification finds a required correction.
**Interfaces:**
- Consumes: verified branch commits and GitHub issue #3759.
- Produces: a fork branch and a PR against `decolua/9router:master`.
- [ ] **Step 1: Create or repair the GitHub fork remote**
Use `gh repo fork decolua/9router --remote` if the current `fork` remote remains
missing, then push `fix/opencode-go-session-header`.
- [ ] **Step 2: Create the PR**
Use title:
```text
fix(opencode-go): send stable session header
```
The body must include the root cause, downstream-session translation policy,
three covered transports, concurrency behavior, verification evidence,
`Fixes #3759`, and a note that this PR is intentionally narrower than #3780.
- [ ] **Step 3: Verify the published PR**
Run `gh pr view --json number,title,state,url,headRefName,baseRefName` and report
the issue and PR URLs.

View File

@@ -0,0 +1,114 @@
# OpenCode Go Session Header Design
## Problem
OpenCode Go will begin rejecting some requests without an
`x-opencode-session` header on September 6, 2026. In 9Router v0.5.65,
`opencode-go` uses `DefaultExecutor`, whose generic header builder does not add
that header. The specialized OpenCode Free executor already sends it, but that
logic does not apply to the paid OpenCode Go provider or its three transports.
## Goals
- Add `x-opencode-session` to every OpenCode Go chat, Claude Messages, and
OpenAI Responses request.
- Translate a downstream conversation identity into a stable upstream identity.
- Keep identities isolated across different downstream agents and conversations.
- Avoid exposing non-OpenCode downstream session identifiers to OpenCode Go.
- Avoid mutable session state on the shared executor singleton.
- Leave OpenCode Free and all unrelated providers unchanged.
## Non-Goals
- Inferring an exact conversation boundary when a downstream client provides no
session or conversation identifier.
- Adding or changing OpenCode Go models, routing, reasoning, or tool behavior.
- Changing the general session-resolution policy for other providers.
## Architecture
Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. The executor
keeps the existing generic URL, authentication, translation, retry, and proxy
behavior, and overrides only the OpenCode Go session-header concern.
`handleChatCore` already resolves a provider-scoped session from the original
request before translation. It will pass that value and the detected client
tool to `executor.execute()` as request context. `OpenCodeGoExecutor.execute()`
will create a shallow request-local credentials object containing the resolved
OpenCode Go session. It will then delegate to `DefaultExecutor.execute()`.
This avoids storing request state on the executor singleton or mutating shared
provider credentials.
## Session Resolution
The original downstream request remains the source of truth. Existing
`resolveSessionId()` behavior recognizes Claude Code, Antigravity, generic
session headers, and common body fields before request translation can discard
them.
Resolution rules:
1. If the downstream request supplies `x-opencode-session`, treat it as an
authoritative OpenCode identity after trimming and length validation.
2. Otherwise use the provider-scoped session resolved from the original request.
3. Namespace the resolved value with the detected downstream agent, falling back
to `generic` when the agent is unknown.
4. Convert the namespaced value to an opaque deterministic identifier:
`ses_` plus the first 32 hexadecimal characters of SHA-256.
5. If no explicit downstream identity exists, the existing provider connection
fallback guarantees that a header is still sent. It is stable but cannot
distinguish multiple conversations sharing that connection.
The same input conversation produces the same upstream identifier for all three
OpenCode Go transports. Different agents using the same raw session value
produce different identifiers.
## Header Injection
`OpenCodeGoExecutor.buildHeaders()` delegates to
`DefaultExecutor.buildHeaders()` and adds only:
```text
x-opencode-session: <stable-session-id>
```
The implementation applies to:
- `https://opencode.ai/zen/go/v1/chat/completions`
- `https://opencode.ai/zen/go/v1/messages`
- `https://opencode.ai/zen/go/v1/responses`
## Error Handling
Session derivation must not make requests fail. Invalid or oversized native
header values are ignored and the normal resolved-session fallback is used.
Hashing uses Node's built-in `crypto` module and requires no new dependency.
## Testing
Add a focused unit suite that proves:
- all three OpenCode Go transports receive the header;
- the same conversation remains stable across requests and transports;
- different conversations produce different values;
- different agents using the same raw ID remain isolated;
- non-OpenCode session IDs are represented as opaque `ses_<32 hex>` values;
- a valid native `x-opencode-session` remains stable;
- headerless requests still receive a stable fallback;
- OpenCode Free behavior is unchanged;
- unrelated `DefaultExecutor` providers do not receive the header;
- no request state is retained on the shared executor instance.
Run the focused unit tests first, then the neighboring executor/session tests,
the full offline test suite, the application build, and the CLI package build.
## Delivery
Build the CLI with `npm --prefix cli run build`, create a package with
`npm --prefix cli pack`, and install the generated tarball globally to replace
the current npm-installed `9router@0.5.65`. Verify the installed package version
and packaged source contains the new executor.
Upstream issue #3759 already tracks the problem, so no duplicate issue will be
created. The pull request will be narrowly scoped to this fix, reference
`Fixes #3759`, and explain how it differs from the broader open PR #3780.

View File

@@ -111,6 +111,27 @@ Model: cx/gpt-5.2-codex
| `cx/gpt-5.2` | GPT 5.2 | General tasks | | `cx/gpt-5.2` | GPT 5.2 | General tasks |
| `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding | | `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding |
### Image Generation
The Codex image catalog includes `cx/gpt-5.6-sol-image`,
`cx/gpt-5.6-terra-image`, and `cx/gpt-5.6-luna-image`, alongside the existing
GPT 5.5, 5.4, and 5.3 image aliases. Select them under **Image → OpenAI Codex**
in the dashboard, or discover them with `GET /v1/models/image` after connecting
a Codex account.
```bash
curl http://localhost:20128/v1/images/generations \
-H "Authorization: Bearer $NINE_ROUTER_API_KEY" \
-H "Content-Type: application/json" \
-d '{"model":"cx/gpt-5.6-sol-image","prompt":"A blue square","size":"1024x1024"}'
```
These are 9Router aliases: the image adapter removes `-image` and sends the
underlying model an `image_generation` tool through the Codex Responses API.
The same endpoint accepts an `image` reference for edits. Image generation
requires an eligible ChatGPT Plus or higher account; availability of each
underlying model and its image tool depends on the connected account.
### Pro Tips ### Pro Tips
- **5-hour rolling quota** - Fresh quota every 5 hours - **5-hour rolling quota** - Fresh quota every 5 hours

View File

@@ -16,7 +16,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
- `rtk/` — request token-killer. `index.js` compresses `tool_result` content in-place (OpenAI/Claude/Kiro shapes); `filters/` per-tool compressors + `autodetect.js`; `headroom.js` external compress proxy; `caveman.js` system-prompt injector. - `rtk/` — request token-killer. `index.js` compresses `tool_result` content in-place (OpenAI/Claude/Kiro shapes); `filters/` per-tool compressors + `autodetect.js`; `headroom.js` external compress proxy; `caveman.js` system-prompt injector.
- `transformer/` — `responsesTransformer.js` (Chat Completions SSE → Codex Responses API SSE), `streamToJsonConverter.js`. - `transformer/` — `responsesTransformer.js` (Chat Completions SSE → Codex Responses API SSE), `streamToJsonConverter.js`.
- `shared/` — cross-provider auth/identity: `clineAuth.js`, `machineId.js`, `qoder/`. - `shared/` — cross-provider auth/identity: `clineAuth.js`, `machineId.js`, `qoder/`.
- `services/` — `model.js`, `provider.js`, `accountFallback.js`, `combo.js`, `compact.js`, `tokenRefresh/`+`tokenRefresh.js`, `oauthCredentialManager.js`, `usage/`, `projectId.js`, `kiroModels.js`/`qoderModels.js`. - `services/` — `model.js`, `provider.js`, `accountFallback.js`, `combo.js`, `tokenRefresh/`+`tokenRefresh.js`, `oauthCredentialManager.js`, `usage/`, `projectId.js`, `kiroModels.js`/`qoderModels.js`.
- `utils/` — streamHandler, stream, sse, error, sessionManager, claudeCloaking, clientDetector, proxyFetch (patches global fetch), cursorProtobuf/cursorChecksum, ollamaTransform. - `utils/` — streamHandler, stream, sse, error, sessionManager, claudeCloaking, clientDetector, proxyFetch (patches global fetch), cursorProtobuf/cursorChecksum, ollamaTransform.
## Conventions ## Conventions
@@ -37,3 +37,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
- `registry/index.js` is an auto-generated static import list; regenerate it (don't hand-edit) after adding a `registry/{id}.js`. REGISTRY_TEMPLATE is excluded by design. - `registry/index.js` is an auto-generated static import list; regenerate it (don't hand-edit) after adding a `registry/{id}.js`. REGISTRY_TEMPLATE is excluded by design.
- Special binary/protobuf formats (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — handle in their executor. - Special binary/protobuf formats (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — handle in their executor.
- `rtk/` + `headroom.js` mutate the request body in-place and are **fail-open**: any error returns null and leaves the body untouched — never throw out of them. RTK skips `is_error`/`status:"error"` tool results to preserve traces. - `rtk/` + `headroom.js` mutate the request body in-place and are **fail-open**: any error returns null and leaves the body untouched — never throw out of them. RTK skips `is_error`/`status:"error"` tool results to preserve traces.
- **HTTP 200 in-stream errors**: some upstreams signal failure INSIDE a 200 stream (AI SDK v5 `{"type":"error"}` events, error text in content). HTTP-level success checks miss these → no fallback, `Status: success` in logs. Three hook points + one config escape hatch:
1. **Translator** — never map an error event to content. Emit an OpenAI-shaped `chunk.error = { message, type }` + terminal chunk (`translator/response/commandcode-to-openai.js` is the worked example). Downstream `parseSSEToOpenAIResponse` already detects `chunk?.error`.
2. **Executor early-peek** — for streaming fallback, read the first events BEFORE returning the response; an error → non-ok Response (`executors/commandcode.js` `peekForUpstreamError`).
3. **Config escape hatch (no code)** — per-provider `streamErrorPatterns` setting (UI: provider page → Stream Error Patterns). Patterns matched against the first ~8KB of the stream and the assembled non-streaming content; see `utils/streamErrorPeek.js` + `utils/streamErrorPatterns.js`.

View File

@@ -171,6 +171,13 @@ export const LOAD_CODE_ASSIST_METADATA = {
// System prompts // System prompts
export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official CLI for Claude."; export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official CLI for Claude.";
// Rewrite rules applied to Antigravity system prompts: competing-client branding
// makes the backend flag the request and answer 429 Quota Exhausted.
export const ANTIGRAVITY_PROMPT_REWRITES = [
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
];
export const ANTIGRAVITY_DEFAULT_SYSTEM = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.**Absolute paths only****Proactiveness**"; export const ANTIGRAVITY_DEFAULT_SYSTEM = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.**Absolute paths only****Proactiveness**";
// Derive từ registry oauth.refreshLeadMs // Derive từ registry oauth.refreshLeadMs

View File

@@ -73,6 +73,17 @@ export const ERROR_RULES = [
{ status: 403, cooldownMs: COOLDOWN.long }, { status: 403, cooldownMs: COOLDOWN.long },
{ status: 404, cooldownMs: COOLDOWN.long }, { status: 404, cooldownMs: COOLDOWN.long },
{ status: 429, backoff: true }, { status: 429, backoff: true },
// --- Request-scoped errors: the request itself is broken — retrying the same
// body on another account/model can never succeed, and locking the account
// would punish a healthy credential for our own bad request. Callers use this
// to fail fast (no account rotation, no model lock).
{ text: "context_length_exceeded", requestScoped: true },
{ text: "context window", requestScoped: true },
{ text: "maximum context length", requestScoped: true },
{ text: "prompt is too long", requestScoped: true },
{ text: "input is too long", requestScoped: true },
{ text: "max_tokens exceed", requestScoped: true },
{ text: "reduce the length", requestScoped: true },
]; ];
// Backward compat: COOLDOWN_MS object (used by index.js re-export) // Backward compat: COOLDOWN_MS object (used by index.js re-export)

View File

@@ -3,7 +3,8 @@ import REGISTRY from "../providers/registry/index.js";
// PROVIDER_MODELS now built from providers/registry (transport + models co-located) // PROVIDER_MODELS now built from providers/registry (transport + models co-located)
import { PROVIDER_MODELS } from "../providers/index.js"; import { PROVIDER_MODELS } from "../providers/index.js";
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js"; import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js"; import { CODEX_REVIEW_SUFFIX, isMuseSparkModel } from "../providers/models/helpers.js";
import { FORMATS } from "../translator/formats.js";
export { PROVIDER_MODELS }; export { PROVIDER_MODELS };
@@ -49,6 +50,9 @@ export function findModelName(aliasOrId, modelId) {
} }
export function getModelTargetFormat(aliasOrId, modelId) { export function getModelTargetFormat(aliasOrId, modelId) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) {
return FORMATS.OPENAI_RESPONSES;
}
const models = PROVIDER_MODELS[aliasOrId]; const models = PROVIDER_MODELS[aliasOrId];
if (!models) return null; if (!models) return null;
return modelTargetFormat(findModel(models, modelId, aliasOrId)); return modelTargetFormat(findModel(models, modelId, aliasOrId));

View File

@@ -1,12 +1,13 @@
import crypto from "crypto"; import crypto from "crypto";
import { BaseExecutor } from "./base.js"; import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js"; import { PROVIDERS } from "../config/providers.js";
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js"; import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js"; import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveSessionId } from "../utils/sessionManager.js"; import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js"; import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js"; import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63} // Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
function sanitizeFunctionName(name) { function sanitizeFunctionName(name) {
@@ -187,6 +188,9 @@ export class AntigravityExecutor extends BaseExecutor {
}; };
} }
const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" });
const sessionId = toNumericSessionId(rawSessionId) || rawSessionId;
// ─── Standard (non-image) request ─── // ─── Standard (non-image) request ───
// Fix contents for Claude models via Antigravity // Fix contents for Claude models via Antigravity
const contents = body.request?.contents?.map(c => { const contents = body.request?.contents?.map(c => {
@@ -202,17 +206,31 @@ export class AntigravityExecutor extends BaseExecutor {
return true; return true;
}); });
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE) // Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
// don't persist thoughtSignature in their history, so backfill the default signature on any // don't persist thoughtSignature in their history, so backfill from cache or default signature.
// functionCall part that arrives without one. // In parallel function calls, only the first call needs a signature; siblings stay unsigned.
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false; let firstFunctionCallSeen = false;
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) { const modifiedParts = parts?.map(p => {
if (!p.functionCall) return p;
const callId = p.functionCall.id;
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null;
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
firstFunctionCallSeen = true;
if (callSig) {
return { ...p, thoughtSignature: callSig };
}
if (p.thoughtSignature && !cachedSig) {
// Unsigned sibling call
const { thoughtSignature: _, ...rest } = p;
return rest;
}
return p;
});
const partsChanged = parts?.length !== c.parts?.length || modifiedParts?.some((p, idx) => p !== c.parts[idx]);
if (role !== c.role || partsChanged) {
return { return {
...c, role, ...c, role,
parts: needsBackfill parts: modifiedParts || parts,
? parts.map(p => (p.functionCall && !p.thoughtSignature)
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
: p)
: parts,
}; };
} }
return c; return c;
@@ -246,13 +264,13 @@ export class AntigravityExecutor extends BaseExecutor {
const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {}; const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {};
stripBlacklisted(requestWithoutTools); stripBlacklisted(requestWithoutTools);
// Rewrite competitive system prompts (e.g. Zed IDE's Claude prompt) to prevent Antigravity from // Rewrite competing-client branding in system prompts (e.g. Zed's Claude prompt,
// flagging the request and immediately blocking it with a 429 Quota Exhausted response. // OpenCode naming) so Antigravity doesn't flag the request with a 429 Quota Exhausted.
if (requestWithoutTools.systemInstruction?.parts) { if (requestWithoutTools.systemInstruction?.parts) {
const oldText = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
for (const part of requestWithoutTools.systemInstruction.parts) { for (const part of requestWithoutTools.systemInstruction.parts) {
if (typeof part.text === "string" && part.text.includes(oldText)) { if (typeof part.text !== "string") continue;
part.text = part.text.split(oldText).join(""); for (const { from, to } of ANTIGRAVITY_PROMPT_REWRITES) {
part.text = part.text.replaceAll(from, to);
} }
} }
} }
@@ -267,7 +285,7 @@ export class AntigravityExecutor extends BaseExecutor {
generationConfig, generationConfig,
...(contents && { contents }), ...(contents && { contents }),
...(tools && { tools }), ...(tools && { tools }),
sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }), sessionId,
safetySettings: undefined, safetySettings: undefined,
...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } }) ...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } })
}; };

View File

@@ -2,6 +2,7 @@ import { HTTP_STATUS, RETRY_CONFIG, DEFAULT_RETRY_CONFIG, resolveRetryEntry, FET
import { shouldRefreshCredentials } from "../services/oauthCredentialManager.js"; import { shouldRefreshCredentials } from "../services/oauthCredentialManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { dbg } from "../utils/debugLog.js"; import { dbg } from "../utils/debugLog.js";
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js"; import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
import { resolveOpenAICompatibleApiType } from "../services/provider.js"; import { resolveOpenAICompatibleApiType } from "../services/provider.js";
@@ -133,13 +134,14 @@ export class BaseExecutor {
// Abort if upstream doesn't return response headers within connection timeout // Abort if upstream doesn't return response headers within connection timeout
const connectCtrl = new AbortController(); const connectCtrl = new AbortController();
const timeoutMs = this.config?.timeoutMs || FETCH_CONNECT_TIMEOUT_MS; const timeoutMs = await resolveProviderTimeoutMs(this.provider, this.config?.timeoutMs, FETCH_CONNECT_TIMEOUT_MS);
const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs); const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs);
const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal; const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal;
let fetchT0 = 0;
try { try {
const bodyStr = JSON.stringify(transformedBody); const bodyStr = JSON.stringify(transformedBody);
const fetchT0 = Date.now(); fetchT0 = Date.now();
dbg("FETCH", `${this.provider.toUpperCase()} → ${url} | body=${bodyStr.length}B | connectTimeout=${timeoutMs}ms`); dbg("FETCH", `${this.provider.toUpperCase()} → ${url} | body=${bodyStr.length}B | connectTimeout=${timeoutMs}ms`);
const response = await proxyAwareFetch(url, { const response = await proxyAwareFetch(url, {
method: "POST", method: "POST",
@@ -165,6 +167,11 @@ export class BaseExecutor {
clearTimeout(connectTimer); clearTimeout(connectTimer);
lastError = error; lastError = error;
const isConnectTimeout = connectCtrl.signal.aborted && error.name === "AbortError"; const isConnectTimeout = connectCtrl.signal.aborted && error.name === "AbortError";
// Error diagnostic — only logs on actual upstream failure. Distinguishes
// undici connect timeout (UND_ERR_CONNECT_TIMEOUT), DNS (ENOTFOUND),
// refused (ECONNREFUSED) vs our own connectCtrl abort (AbortError).
const cause = error?.cause || {};
console.log(`[FETCH-DIAG] ${this.provider} fetch error | name=${error.name} | code=${error.code ?? cause?.code ?? "none"} | msg=${String(error.message).slice(0, 120)} | connectTimeout=${timeoutMs}ms | elapsed=${Date.now() - fetchT0}ms`);
dbg("FETCH", `${this.provider.toUpperCase()} ✖ ${error.name}: ${error.message}${isConnectTimeout ? " (connect timeout)" : ""}`); dbg("FETCH", `${this.provider.toUpperCase()} ✖ ${error.name}: ${error.message}${isConnectTimeout ? " (connect timeout)" : ""}`);
// Connect timeout is internal — convert to retryable network error, don't propagate AbortError // Connect timeout is internal — convert to retryable network error, don't propagate AbortError
if (error.name === "AbortError" && !isConnectTimeout) throw error; if (error.name === "AbortError" && !isConnectTimeout) throw error;

View File

@@ -1,6 +1,7 @@
import { randomUUID } from "crypto"; import { randomUUID } from "crypto";
import { BaseExecutor } from "./base.js"; import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js"; import { PROVIDERS } from "../config/providers.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { commandCodeToOpenAIResponse } from "../translator/response/commandcode-to-openai.js"; import { commandCodeToOpenAIResponse } from "../translator/response/commandcode-to-openai.js";
import { SSE_DONE } from "../utils/sseConstants.js"; import { SSE_DONE } from "../utils/sseConstants.js";
@@ -14,53 +15,282 @@ import { SSE_DONE } from "../utils/sseConstants.js";
* We translate each event to an OpenAI chat.completion.chunk and emit it as SSE so * We translate each event to an OpenAI chat.completion.chunk and emit it as SSE so
* both the streaming and non-streaming (forced SSE → JSON) downstream handlers in * both the streaming and non-streaming (forced SSE → JSON) downstream handlers in
* 9router can consume it without further format translation. * 9router can consume it without further format translation.
*
* Terminal upstream failures arrive as `{"type":"error"}` events inside the HTTP
* 200 stream, so a plain `response.ok` check cannot see them. We peek the first
* events before committing the response (see peekForUpstreamError) so a stream
* that starts with an error fails fast — the normal `!response.ok` path then
* triggers account/model fallback instead of streaming fake success content.
*/ */
export class CommandCodeExecutor extends BaseExecutor { export class CommandCodeExecutor extends BaseExecutor {
constructor() { constructor() {
super("commandcode", PROVIDERS.commandcode); super("commandcode", PROVIDERS.commandcode);
} }
transformRequest(model, body, stream, credentials) { transformRequest(_model, body, _stream, _credentials) {
body.stream = true; body.stream = true;
return body; return body;
} }
buildHeaders(credentials, stream = true) { buildHeaders(credentials, stream = true) {
const headers = { const headers = {
"Content-Type": "application/json", "Content-Type": "application/json",
...(this.config.headers || {}), ...(this.config.headers || {}),
"x-session-id": randomUUID(), "x-session-id": randomUUID(),
}; };
const token = credentials?.apiKey || credentials?.accessToken; const token = credentials?.apiKey || credentials?.accessToken;
if (token) headers["Authorization"] = `Bearer ${token}`; if (token) headers["Authorization"] = `Bearer ${token}`;
if (stream) headers["Accept"] = "text/event-stream"; if (stream) headers["Accept"] = "text/event-stream";
return headers; return headers;
} }
async execute(opts) { async execute(opts) {
const result = await super.execute(opts); const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result; if (!result?.response?.ok || !result.response.body) return result;
result.response = wrapNdjsonAsOpenAISse(result.response, opts.model); result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
return result; return result;
} }
parseError(response, bodyText) {
let parsed = null;
try {
parsed = JSON.parse(bodyText || "{}");
} catch {
parsed = null;
}
const errObj = parsed?.error || parsed;
const msg = errObj?.message || parsed?.message || bodyText || response.statusText;
const status = Number(errObj?.code || errObj?.statusCode || response.status) || response.status;
return {
status,
message: msg || `CommandCode upstream error: ${response.status}`,
};
}
} }
function wrapNdjsonAsOpenAISse(originalResponse, model) { export function parseCommandCodeError(event) {
if (!event || typeof event !== "object") {
return {
statusCode: 503,
message: "CommandCode upstream error",
type: "server_error",
};
}
const errVal = event.error ?? event.message ?? "unknown";
let message = "";
let statusCode = null;
let type = "server_error";
if (typeof errVal === "object" && errVal !== null) {
message = errVal.message || errVal.error || JSON.stringify(errVal);
if (errVal.statusCode && Number.isInteger(Number(errVal.statusCode))) {
statusCode = Number(errVal.statusCode);
} else if (errVal.status && Number.isInteger(Number(errVal.status))) {
statusCode = Number(errVal.status);
}
if (errVal.type) type = errVal.type;
} else if (typeof errVal === "string") {
message = errVal;
} else {
message = JSON.stringify(errVal);
}
if (event.statusCode && Number.isInteger(Number(event.statusCode))) {
statusCode = Number(event.statusCode);
}
if (!statusCode || statusCode < 400 || statusCode > 599) {
const lower = message.toLowerCase();
if (lower.includes("rate limit") || lower.includes("too many requests")) {
statusCode = 429;
type = "rate_limit_error";
} else if (lower.includes("unauthorized") || lower.includes("invalid api key") || lower.includes("authentication")) {
statusCode = 401;
type = "authentication_error";
} else if (lower.includes("payment required") || lower.includes("billing")) {
statusCode = 402;
type = "billing_error";
} else if (lower.includes("quota") || lower.includes("forbidden") || lower.includes("permission")) {
statusCode = 403;
type = "permission_error";
} else if (lower.includes("not found")) {
statusCode = 404;
type = "invalid_request_error";
} else if (lower.includes("unavailable") || lower.includes("overloaded") || lower.includes("server error")) {
statusCode = 503;
type = "server_error";
} else {
statusCode = 503;
}
}
return { statusCode, message, type };
}
export async function inspectAndWrapCommandCodeResponse(originalResponse, model) {
const reader = originalResponse.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
const bufferedLines = [];
let detectedError = null;
try {
while (true) {
const { value, done } = await reader.read();
if (done) {
const trimmed = buffer.trim();
if (trimmed) {
try {
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
const parsed = JSON.parse(jsonStr);
if (parsed?.type === "error") {
detectedError = parsed;
} else {
bufferedLines.push(trimmed);
}
} catch {
bufferedLines.push(trimmed);
}
}
break;
}
buffer += decoder.decode(value, { stream: true });
const lines = buffer.split("\n");
buffer = lines.pop() || "";
let stopLoop = false;
for (const line of lines) {
const trimmed = line.trim();
if (!trimmed) continue;
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
if (!jsonStr || jsonStr === "[DONE]") {
bufferedLines.push(trimmed);
stopLoop = true;
break;
}
let event;
try {
event = JSON.parse(jsonStr);
} catch {
bufferedLines.push(trimmed);
continue;
}
if (event?.type === "error") {
detectedError = event;
stopLoop = true;
break;
}
bufferedLines.push(trimmed);
if (
event?.type === "text-delta" ||
event?.type === "reasoning-delta" ||
event?.type === "tool-input-start" ||
event?.type === "tool-call" ||
event?.type === "finish" ||
event?.type === "finish-step"
) {
stopLoop = true;
break;
}
}
if (stopLoop) break;
}
} catch {
try { reader.releaseLock(); } catch { /* ignore */ }
return originalResponse;
}
if (detectedError) {
try { await reader.cancel(); } catch { /* ignore */ }
const { statusCode, message, type } = parseCommandCodeError(detectedError);
return new Response(
JSON.stringify({
error: {
message: `[CommandCode error: ${message}]`,
type,
code: statusCode,
},
}),
{
status: statusCode,
statusText: statusCode === 503 ? "Service Unavailable" : (statusCode === 429 ? "Too Many Requests" : "Bad Gateway"),
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
},
}
);
}
const combinedStream = createReplayedStream(bufferedLines, buffer, reader);
return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse);
}
function createReplayedStream(bufferedLines, remainingBuffer, reader) {
const encoder = new TextEncoder();
let replayed = false;
return new ReadableStream({
async pull(controller) {
if (!replayed) {
replayed = true;
let prefix = bufferedLines.join("\n");
if (prefix && remainingBuffer) {
prefix += "\n" + remainingBuffer;
} else if (remainingBuffer) {
prefix = remainingBuffer;
} else if (prefix) {
prefix += "\n";
}
if (prefix) {
controller.enqueue(encoder.encode(prefix));
}
}
try {
const { value, done } = await reader.read();
if (done) {
controller.close();
} else {
controller.enqueue(value);
}
} catch (err) {
controller.error(err);
}
},
async cancel(reason) {
try {
await reader.cancel(reason);
} catch {
/* ignore */
}
},
});
}
function wrapNdjsonAsOpenAISse(streamBody, model, originalResponse = null) {
const decoder = new TextDecoder(); const decoder = new TextDecoder();
const encoder = new TextEncoder(); const encoder = new TextEncoder();
let buffer = ""; let buffer = "";
const state = { model }; const state = { model };
const emitChunks = (chunks, controller) => { const emitChunks = (chunks, controller) => {
if (!chunks) return; if (!chunks) return;
const list = Array.isArray(chunks) ? chunks : [chunks]; const list = Array.isArray(chunks) ? chunks : [chunks];
for (const c of list) { for (const c of list) {
if (c == null) continue; if (c == null) continue;
controller.enqueue(encoder.encode(`data: ${JSON.stringify(c)}\n\n`)); controller.enqueue(encoder.encode(`data: ${JSON.stringify(c)}\n\n`));
} }
}; };
const transform = new TransformStream({ const transform = new TransformStream({
transform(chunk, controller) { transform(chunk, controller) {
@@ -70,7 +300,6 @@ function wrapNdjsonAsOpenAISse(originalResponse, model) {
for (const line of lines) { for (const line of lines) {
const trimmed = line.trim(); const trimmed = line.trim();
if (!trimmed) continue; if (!trimmed) continue;
// Translate AI SDK v5 NDJSON line to one or more OpenAI chunks
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller); emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
} }
}, },
@@ -83,11 +312,17 @@ function wrapNdjsonAsOpenAISse(originalResponse, model) {
}, },
}); });
const newBody = originalResponse.body.pipeThrough(transform); const newBody = streamBody.pipeThrough(transform);
return new Response(newBody, { return new Response(newBody, {
status: originalResponse.status, status: originalResponse?.status || 200,
statusText: originalResponse.statusText, statusText: originalResponse?.statusText || "OK",
headers: originalResponse.headers, headers: {
"Content-Type": "text/event-stream",
"Cache-Control": "no-cache",
"Connection": "keep-alive",
...(originalResponse?.headers ? Object.fromEntries(originalResponse.headers.entries()) : {}),
"content-type": "text/event-stream",
},
}); });
} }

View File

@@ -154,7 +154,18 @@ export class DefaultExecutor extends BaseExecutor {
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials); for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
applyAuth(headers, desc, credentials); applyAuth(headers, desc, credentials);
if (this.provider === "claude" && model) { // anthropic-compatible-* nodes serving a real Claude model sit in front of
// Anthropic itself (a rotating multi-account proxy, a corporate gateway),
// so the request needs the same beta flags the `claude` provider sends:
// without `context-management-2025-06-27` upstream rejects the
// `context_management` block Claude Code puts in every request with
// "context_management: Extra inputs are not permitted" (HTTP 400), and the
// combo silently falls through to the next model. The model id gates this:
// a node fronting Kimi or GLM answers on its own ids and never matches, so
// gateways that would choke on unknown beta flags are left untouched.
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
if (model && (this.provider === "claude"
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
headers["Anthropic-Beta"] = selectAnthropicBeta(model); headers["Anthropic-Beta"] = selectAnthropicBeta(model);
} }

View File

@@ -10,6 +10,7 @@ import { CodexExecutor } from "./codex.js";
import { CursorExecutor } from "./cursor.js"; import { CursorExecutor } from "./cursor.js";
import { VertexExecutor } from "./vertex.js"; import { VertexExecutor } from "./vertex.js";
import { OpenCodeExecutor } from "./opencode.js"; import { OpenCodeExecutor } from "./opencode.js";
import { OpenCodeGoExecutor } from "./opencode-go.js";
import { GrokWebExecutor } from "./grok-web.js"; import { GrokWebExecutor } from "./grok-web.js";
import { GrokCliExecutor } from "./grok-cli.js"; import { GrokCliExecutor } from "./grok-cli.js";
import { PerplexityWebExecutor } from "./perplexity-web.js"; import { PerplexityWebExecutor } from "./perplexity-web.js";
@@ -40,6 +41,7 @@ const executors = {
vertex: new VertexExecutor("vertex"), vertex: new VertexExecutor("vertex"),
"vertex-partner": new VertexExecutor("vertex-partner"), "vertex-partner": new VertexExecutor("vertex-partner"),
opencode: new OpenCodeExecutor(), opencode: new OpenCodeExecutor(),
"opencode-go": new OpenCodeGoExecutor(),
"grok-web": new GrokWebExecutor(), "grok-web": new GrokWebExecutor(),
"grok-cli": new GrokCliExecutor(), "grok-cli": new GrokCliExecutor(),
gcli: new GrokCliExecutor(), // Alias gcli: new GrokCliExecutor(), // Alias
@@ -84,6 +86,7 @@ export { CursorExecutor } from "./cursor.js";
export { VertexExecutor } from "./vertex.js"; export { VertexExecutor } from "./vertex.js";
export { DefaultExecutor } from "./default.js"; export { DefaultExecutor } from "./default.js";
export { OpenCodeExecutor } from "./opencode.js"; export { OpenCodeExecutor } from "./opencode.js";
export { OpenCodeGoExecutor } from "./opencode-go.js";
export { GrokWebExecutor } from "./grok-web.js"; export { GrokWebExecutor } from "./grok-web.js";
export { GrokCliExecutor } from "./grok-cli.js"; export { GrokCliExecutor } from "./grok-cli.js";
export { PerplexityWebExecutor } from "./perplexity-web.js"; export { PerplexityWebExecutor } from "./perplexity-web.js";

View File

@@ -0,0 +1,182 @@
import crypto from "node:crypto";
import { DefaultExecutor } from "./default.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../translator/formats/responsesApi.js";
const SESSION_HEADER = "x-opencode-session";
const SESSION_FIELD = "_opencodeGoSession";
const MAX_SESSION_LENGTH = 256;
const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses";
const MAX_TOOL_NAME_LEN = 128;
function normalizeSession(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return normalized;
}
function nativeSession(headers) {
if (!headers || typeof headers !== "object") return null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value);
}
return null;
}
function translatedSession(sessionId, clientTool) {
const digest = crypto
.createHash("sha256")
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
.digest("hex")
.slice(0, 32);
return `ses_${digest}`;
}
// Strip the thinking suffix "model(level)" so checks hit the base id.
function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
function isResponsesModel(model) {
return isMuseSparkModel(baseModelId(model));
}
// Flatten Chat Completions tool declarations into the Responses flat shape and
// drop hosted/nameless tools the /responses endpoint rejects.
function normalizeResponsesTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
const name = rawName.trim();
if (!name) return false;
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
? tool.parameters
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
// Mirror the request translator: {type:"object"} without properties is rejected
// by strict Responses backends, so fill in the empty properties map.
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
for (const k of Object.keys(tool)) delete tool[k];
tool.type = "function";
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
if (description) tool.description = description;
tool.parameters = parameters;
validNames.add(tool.name);
return true;
});
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
if (!n || !validNames.has(n)) delete body.tool_choice;
}
}
}
// Last line of defense for native Responses clients (sourceFormat === targetFormat
// skips translation): coerce items in place so malformed tool payloads 400 here
// with a clear shape instead of upstream as InputValidationError.
function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
item.call_id = clampResponsesCallId(item.call_id);
item.arguments = coerceResponsesArguments(item.arguments);
return true;
}
if (item.type === "function_call_output") {
item.call_id = clampResponsesCallId(item.call_id);
item.output = coerceResponsesOutput(item.output);
return true;
}
return true;
});
}
export class OpenCodeGoExecutor extends DefaultExecutor {
constructor() {
super("opencode-go");
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
return super.buildUrl(model, stream, urlIndex, credentials);
}
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
const sourceCredentials = credentials || {};
const native = nativeSession(sourceCredentials.rawHeaders);
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
headers: sourceCredentials.rawHeaders,
body,
connectionId: sourceCredentials.connectionId,
scope: "opencode-go",
});
return {
...sourceCredentials,
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
};
}
async execute(args) {
const credentials = this.prepareRequestCredentials(args);
return super.execute({ ...args, credentials });
}
buildHeaders(credentials, stream = true, url, model) {
const headers = super.buildHeaders(credentials || {}, stream, url, model);
const prepared = credentials?.[SESSION_FIELD];
if (prepared) {
headers[SESSION_HEADER] = prepared;
return headers;
}
const fallback = this.prepareRequestCredentials({ credentials });
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
return headers;
}
transformRequest(model, body, stream, credentials) {
const out = super.transformRequest(model, body);
if (!isResponsesModel(model || body?.model)) return out;
const normalized = normalizeResponsesInput(out.input);
if (normalized) out.input = normalized;
if (!Array.isArray(out.input) || out.input.length === 0) {
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
}
// Responses names the output cap max_output_tokens, not max_tokens.
if (out.max_output_tokens === undefined) {
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
}
delete out.max_tokens;
delete out.max_completion_tokens;
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
}
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
if (!out.reasoning.summary) out.reasoning.summary = "auto";
}
delete out.reasoning_effort;
out.stream = true;
out.store = false;
normalizeResponsesTools(out);
sanitizeResponsesItems(out);
return out;
}
}

View File

@@ -1,11 +1,17 @@
import crypto from "crypto"; import crypto from "crypto";
import { BaseExecutor } from "./base.js"; import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js"; import { PROVIDERS } from "../config/providers.js";
import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { injectReasoningContent } from "../utils/reasoningContentInjector.js"; import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
import { resolveSessionId } from "../utils/sessionManager.js"; import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
const OPENCODE_UA = "opencode"; const OPENCODE_UA = "opencode";
const MESSAGES_MODELS = new Set(); // Models served by /zen/v1/responses; every other model stays on /chat/completions.
const RESPONSES_MODELS = new Set([
"muse-spark-1.2-contributor-free",
"muse-spark-1.3-contributor-free",
]);
function generateRequestId() { function generateRequestId() {
return `msg_${crypto.randomUUID().replace(/-/g, "")}`; return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
@@ -15,19 +21,48 @@ function generateSessionId() {
return `ses_${crypto.randomUUID().replace(/-/g, "")}`; return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
} }
// Normalize any resolved id into opencode's ses_ format (stable per-conversation) // Strip the thinking suffix "model(level)" so registry lookups hit the base id.
function toOpencodeSession(id) { function baseModelId(model) {
const stripped = String(id || "").replace(/^ses_/, "").replace(/-/g, ""); return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
return stripped ? `ses_${stripped}` : null; }
function isResponsesModel(model) {
const base = baseModelId(model);
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
} }
function resolveOpencodeSession(body, credentials) { function resolveOpencodeSession(body, credentials) {
return toOpencodeSession(resolveSessionId({ const headers = credentials?.rawHeaders || {};
headers: credentials?.rawHeaders, return resolveSessionId({
headers,
body, body,
connectionId: credentials?.connectionId, connectionId: credentials?.connectionId,
scope: "opencode", scope: "opencode",
})); generate: generateSessionId,
});
}
function normalizeOpencodeReasoning(model, body) {
const current = body.reasoning;
const currentReasoning = current && typeof current === "object" && !Array.isArray(current)
? current
: null;
const requestedEffort = typeof body.reasoning_effort === "string"
? body.reasoning_effort
: currentReasoning?.effort;
if (typeof requestedEffort !== "string") return;
const cleanModel = baseModelId(model || body.model);
const supportedLevels = getThinkingLevels("opencode", cleanModel);
let effort = requestedEffort.toLowerCase().trim();
if ((effort === "max" || effort === "ultra") && supportedLevels?.length && !supportedLevels.includes(effort)) {
if (effort === "ultra" && supportedLevels.includes("max")) effort = "max";
else if (supportedLevels.includes("xhigh")) effort = "xhigh";
}
body.reasoning = { ...currentReasoning, effort };
if (!body.reasoning.summary) body.reasoning.summary = "auto";
delete body.reasoning_effort;
} }
export class OpenCodeExecutor extends BaseExecutor { export class OpenCodeExecutor extends BaseExecutor {
@@ -38,13 +73,24 @@ export class OpenCodeExecutor extends BaseExecutor {
transformRequest(model, body, stream, credentials) { transformRequest(model, body, stream, credentials) {
this._currentSessionId = resolveOpencodeSession(body, credentials); this._currentSessionId = resolveOpencodeSession(body, credentials);
if (isResponsesModel(model)) {
// Responses API names the output cap max_output_tokens and takes thinking
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
if (body.max_output_tokens === undefined) {
if (body.max_completion_tokens !== undefined) body.max_output_tokens = body.max_completion_tokens;
else if (body.max_tokens !== undefined) body.max_output_tokens = body.max_tokens;
}
delete body.max_tokens;
delete body.max_completion_tokens;
normalizeOpencodeReasoning(model, body);
}
return injectReasoningContent({ provider: this.provider, model, body }); return injectReasoningContent({ provider: this.provider, model, body });
} }
buildUrl(model) { buildUrl(model) {
const base = this.config.baseUrl; const base = this.config.baseUrl;
return MESSAGES_MODELS.has(model) return isResponsesModel(model)
? `${base}/zen/v1/messages` ? `${base}/zen/v1/responses`
: `${base}/zen/v1/chat/completions`; : `${base}/zen/v1/chat/completions`;
} }

View File

@@ -30,6 +30,7 @@ import { PROVIDERS } from "../config/providers.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { SSE_DONE } from "../utils/sseConstants.js"; import { SSE_DONE } from "../utils/sseConstants.js";
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js"; import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
import { import {
QODER_CHAT_URL_ENCODED, QODER_CHAT_URL_ENCODED,
QODER_CHAT_BASE_ALT, QODER_CHAT_BASE_ALT,
@@ -37,10 +38,13 @@ import {
QODER_MODEL_MAP, QODER_MODEL_MAP,
} from "../shared/qoder/constants.js"; } from "../shared/qoder/constants.js";
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js"; import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
import { encodeDataUri } from "../translator/concerns/image.js";
/** /**
* Hoist role:"system" messages out of the messages array (Qoder rejects * Hoist role:"system" messages out of the messages array (Qoder rejects
* system in messages) and flatten any multipart content arrays. * system in messages) and flatten multipart content arrays — EXCEPT image
* blocks, which are preserved (see normalizeContent).
*/ */
function normalizeMessages(messages) { function normalizeMessages(messages) {
if (!Array.isArray(messages) || messages.length === 0) { if (!Array.isArray(messages) || messages.length === 0) {
@@ -50,18 +54,72 @@ function normalizeMessages(messages) {
const out = []; const out = [];
for (const msg of messages) { for (const msg of messages) {
if (!msg || typeof msg !== "object") continue; if (!msg || typeof msg !== "object") continue;
const text = extractText(msg.content);
if (msg.role === "system") { if (msg.role === "system") {
const text = extractText(msg.content);
if (text) systemParts.push(text); if (text) systemParts.push(text);
continue; continue;
} }
const cloned = { ...msg }; const cloned = { ...msg };
cloned.content = text; cloned.content = normalizeContent(msg.content);
out.push(cloned); out.push(cloned);
} }
return { messages: out, systemText: systemParts.join("\n\n") }; return { messages: out, systemText: systemParts.join("\n\n") };
} }
/**
* Normalize one message's content for Qoder.
*
* Text-only content is flattened to a plain string (Qoder's historical
* shape). When images are present the content stays an array and image
* blocks are kept as OpenAI-style `image_url` parts — verified against the
* upstream: it accepts both http(s) URLs and inline base64 data: URIs
* directly, no pre-upload to the /image/upload OSS flow required (that is
* a qodercli client-side choice, not a protocol requirement). The legacy
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
* qodercli leaves them null too.
*
* Claude-style `{type:"image", source:{...}}` blocks are converted to
* `image_url` so claude-format clients also round-trip.
*/
function normalizeContent(content) {
if (typeof content === "string") return content;
if (content == null) return "";
if (!Array.isArray(content)) return String(content);
const blocks = [];
const textParts = [];
let hasImage = false;
for (const item of content) {
if (!item || typeof item !== "object") continue;
if (item.type === OPENAI_BLOCK.IMAGE_URL && typeof item.image_url?.url === "string" && item.image_url.url) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: item.image_url.url } });
hasImage = true;
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
// Claude base64/url image → OpenAI image_url equivalent.
const src = item.source;
const url = src.type === "base64" && src.data
? encodeDataUri(src.media_type || "image/png", src.data)
: typeof src.url === "string" && src.url ? src.url : null;
if (url) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
hasImage = true;
}
} else if (typeof item.text === "string" && item.text) {
if (hasImage || blocks.length) {
// Keep ordering faithful once images are in play.
blocks.push({ type: OPENAI_BLOCK.TEXT, text: item.text });
} else {
textParts.push(item.text);
}
}
}
if (!hasImage) return textParts.join("\n");
// Prepend any text collected before the first image block.
if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") });
return blocks;
}
function extractText(content) { function extractText(content) {
if (typeof content === "string") return content; if (typeof content === "string") return content;
if (content == null) return ""; if (content == null) return "";
@@ -84,9 +142,9 @@ function extractText(content) {
function lastUserText(messages) { function lastUserText(messages) {
for (let i = messages.length - 1; i >= 0; i--) { for (let i = messages.length - 1; i >= 0; i--) {
const m = messages[i]; const m = messages[i];
if (m?.role === "user" && typeof m.content === "string") { if (m?.role !== "user") continue;
return m.content; if (typeof m.content === "string") return m.content;
} if (Array.isArray(m.content)) return extractText(m.content);
} }
return ""; return "";
} }
@@ -110,6 +168,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) {
if (m.role) { h.update("\0"); h.update(m.role); } if (m.role) { h.update("\0"); h.update(m.role); }
if (typeof m.content === "string" && m.content) { if (typeof m.content === "string" && m.content) {
h.update("\0"); h.update(m.content); h.update("\0"); h.update(m.content);
} else if (Array.isArray(m.content)) {
// Include image refs so the same prompt with a different image gets
// a distinct chat_record_id.
h.update("\0");
try { h.update(JSON.stringify(m.content)); } catch {}
} }
} }
if (tools) { if (tools) {
@@ -536,7 +599,7 @@ export class QoderExecutor extends BaseExecutor {
}; };
// Abort if upstream doesn't return response headers within connect timeout. // Abort if upstream doesn't return response headers within connect timeout.
const timeoutMs = this.config?.timeoutMs || FETCH_CONNECT_TIMEOUT_MS; const timeoutMs = await resolveProviderTimeoutMs(this.provider, this.config?.timeoutMs, FETCH_CONNECT_TIMEOUT_MS);
const connectCtrl = new AbortController(); const connectCtrl = new AbortController();
const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs); const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs);
const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal; const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal;

View File

@@ -11,7 +11,7 @@ import { PROVIDERS } from "../config/providers.js";
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js"; import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js"; import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js";
import { handleBypassRequest } from "../utils/bypassHandler.js"; import { handleBypassRequest } from "../utils/bypassHandler.js";
import { trackPendingRequest, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js";
import { getExecutor } from "../executors/index.js"; import { getExecutor } from "../executors/index.js";
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js"; import { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
import { buildRequestDetail, extractRequestConfig } from "./chatCore/requestDetail.js"; import { buildRequestDetail, extractRequestConfig } from "./chatCore/requestDetail.js";
@@ -28,7 +28,9 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
import { getCapabilitiesForModel } from "../providers/capabilities.js"; import { getCapabilitiesForModel } from "../providers/capabilities.js";
import { stripUnsupportedModalities } from "../translator/concerns/modality.js"; import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js"; import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
import { resolveSessionId } from "../utils/sessionManager.js"; import { resolveSessionId } from "../utils/sessionManager.js";
import { maybeRejectEarlyStreamError } from "../utils/streamErrorPeek.js";
/** /**
* Core chat handler - shared between SSE and Worker * Core chat handler - shared between SSE and Worker
@@ -57,7 +59,7 @@ export function stripContinuityFields(body) {
return body; return body;
} }
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking }) { export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, capsOverride = null, streamErrorPatterns = null }) {
const { provider, model } = modelInfo; const { provider, model } = modelInfo;
const requestStartTime = Date.now(); const requestStartTime = Date.now();
// Stable per-session color so all lines of one CLI conversation share a tag // Stable per-session color so all lines of one CLI conversation share a tag
@@ -90,7 +92,12 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// differ — kimi/glm only do /chat/completions). Undeclared models keep the // differ — kimi/glm only do /chat/completions). Undeclared models keep the
// upstream default (use the transport), preserving behavior for glm/deepseek/... // upstream default (use the transport), preserving behavior for glm/deepseek/...
const useTransport = (!modelSupportedFormats || modelSupportedFormats.includes(sourceFormat)) ? runtimeTransport : null; const useTransport = (!modelSupportedFormats || modelSupportedFormats.includes(sourceFormat)) ? runtimeTransport : null;
const targetFormat = modelTargetFormat || useTransport?.format || getTargetFormat(provider, credentials); // A source-format-matched endpoint keeps the request lossless. Prefer it
// over a model-level targetFormat, which is only the fallback for clients
// whose wire format has no supported transport (for example MiniMax-M3:
// OpenAI clients should stay on /chat/completions; other clients can fall
// back to its declared Claude target).
const targetFormat = useTransport?.format || modelTargetFormat || getTargetFormat(provider, credentials);
if (useTransport && credentials) credentials.runtimeTransport = useTransport; if (useTransport && credentials) credentials.runtimeTransport = useTransport;
const stripList = getModelStrip(alias, model); const stripList = getModelStrip(alias, model);
const upstreamModel = getModelUpstreamId(alias, model); const upstreamModel = getModelUpstreamId(alias, model);
@@ -100,7 +107,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
if (providerThinking?.mode && providerThinking.mode !== "auto") { if (providerThinking?.mode && providerThinking.mode !== "auto") {
const mode = providerThinking.mode; const mode = providerThinking.mode;
if (mode === "on" && !body.thinking) { if (mode === "on" && !body.thinking) {
console.log("Injecting provider-level thinking config override: on"); log?.debug?.("THINKING", `provider-level override: on`);
body = { ...body, thinking: { type: "enabled", budget_tokens: 10000 } }; body = { ...body, thinking: { type: "enabled", budget_tokens: 10000 } };
} else if (mode === "off" && !body.thinking) { } else if (mode === "off" && !body.thinking) {
body = { ...body, thinking: { type: "disabled" } }; body = { ...body, thinking: { type: "disabled" } };
@@ -149,8 +156,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
if (credentials) credentials.rawHeaders = clientRawRequest?.headers || {}; if (credentials) credentials.rawHeaders = clientRawRequest?.headers || {};
// Auto-strip media blocks the model can't read (vision/audio/pdf) before translation. // Auto-strip media blocks the model can't read (vision/audio/pdf) before translation.
// capsOverride lets the app layer assert per-model capabilities (e.g. user-registered
// models) on top of the static tables.
if (!passthrough) { if (!passthrough) {
const caps = getCapabilitiesForModel(provider, model); const caps = { ...getCapabilitiesForModel(provider, model), ...(capsOverride || {}) };
if (stripUnsupportedModalities(body, sourceFormat, caps)) { if (stripUnsupportedModalities(body, sourceFormat, caps)) {
log?.debug?.("MODALITY", `stripped unsupported media for ${provider}/${model}`); log?.debug?.("MODALITY", `stripped unsupported media for ${provider}/${model}`);
} }
@@ -235,17 +244,23 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
delete translatedBody.tools; delete translatedBody.tools;
} }
// Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax)
// reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing.
if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) {
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
}
// Per-request opt-out: client can bypass all token savers via header // Per-request opt-out: client can bypass all token savers via header
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off"; const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
// RTK: compress tool_result content // RTK: compress tool_result content
const rtkStats = compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled); const rtkStats = compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled);
const rtkLine = formatRtkLog(rtkStats); const rtkLine = formatRtkLog(rtkStats);
if (rtkLine) console.log(rtkLine); if (rtkLine) log?.info?.("RTK", rtkLine.replace(/^\[RTK\] /, ""));
// Headroom: optional external proxy compression; fail open if proxy is absent. // Headroom: optional external proxy compression; fail open if proxy is absent.
const headroomDiagnostics = {}; const headroomDiagnostics = {};
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, diagnostics: headroomDiagnostics }); const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, timeoutMs: headroomTimeoutMs, diagnostics: headroomDiagnostics });
const headroomLine = formatHeadroomLog(headroomStats); const headroomLine = formatHeadroomLog(headroomStats);
const headroomSizeLine = formatHeadroomSizeLog(headroomDiagnostics); const headroomSizeLine = formatHeadroomSizeLog(headroomDiagnostics);
if (headroomLine) { if (headroomLine) {
@@ -291,7 +306,6 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
const executor = getExecutor(provider); const executor = getExecutor(provider);
trackPendingRequest(model, provider, connectionId, true); trackPendingRequest(model, provider, connectionId, true);
appendRequestLog({ model, provider, connectionId, status: "PENDING" }).catch(() => { });
const msgCount = translatedBody.messages?.length || translatedBody.input?.length || translatedBody.contents?.length || translatedBody.request?.contents?.length || 0; const msgCount = translatedBody.messages?.length || translatedBody.input?.length || translatedBody.contents?.length || translatedBody.request?.contents?.length || 0;
log?.debug?.("REQUEST", `${provider.toUpperCase()} | ${model} | ${msgCount} msgs`); log?.debug?.("REQUEST", `${provider.toUpperCase()} | ${model} | ${msgCount} msgs`);
@@ -344,7 +358,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// exception: it is decoded by the executor into OpenAI-compatible output. // exception: it is decoded by the executor into OpenAI-compatible output.
let providerResponseFormat = targetFormat; let providerResponseFormat = targetFormat;
try { try {
const result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions }); const result = await executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
providerResponse = result.response; providerResponse = result.response;
providerUrl = result.url; providerUrl = result.url;
providerHeaders = result.headers; providerHeaders = result.headers;
@@ -353,7 +377,6 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody);
} catch (error) { } catch (error) {
trackPendingRequest(model, provider, connectionId, false, true); trackPendingRequest(model, provider, connectionId, false, true);
appendRequestLog({ model, provider, connectionId, status: `FAILED ${error.name === "AbortError" ? 499 : HTTP_STATUS.BAD_GATEWAY}` }).catch(() => { });
saveRequestDetail(buildRequestDetail({ saveRequestDetail(buildRequestDetail({
provider, model, connectionId, provider, model, connectionId,
latency: { ttft: 0, total: Date.now() - requestStartTime }, latency: { ttft: 0, total: Date.now() - requestStartTime },
@@ -398,7 +421,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); } try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
} }
try { try {
const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions }); const retryResult = await executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
if (retryResult.response.ok) { if (retryResult.response.ok) {
providerResponse = retryResult.response; providerResponse = retryResult.response;
providerUrl = retryResult.url; providerUrl = retryResult.url;
@@ -413,11 +446,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
} }
} }
// Provider returned error // Provider returned error
if (!providerResponse.ok) { if (!providerResponse.ok) {
trackPendingRequest(model, provider, connectionId, false, true); trackPendingRequest(model, provider, connectionId, false, true);
const { statusCode, message, resetsAtMs } = await parseUpstreamError(providerResponse, executor); const { statusCode, message, resetsAtMs } = await parseUpstreamError(providerResponse, executor);
appendRequestLog({ model, provider, connectionId, status: `FAILED ${statusCode}` }).catch(() => { });
saveRequestDetail(buildRequestDetail({ saveRequestDetail(buildRequestDetail({
provider, model, connectionId, provider, model, connectionId,
latency: { ttft: 0, total: Date.now() - requestStartTime }, latency: { ttft: 0, total: Date.now() - requestStartTime },
@@ -438,8 +471,31 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
return createErrorResult(statusCode, errMsg, resetsAtMs); return createErrorResult(statusCode, errMsg, resetsAtMs);
} }
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log }; const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers
const appendLog = (extra) => appendRequestLog({ model, provider, connectionId, ...extra }).catch(() => { }); const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log, streamErrorPatterns };
// Early-peek streaming responses for configured in-stream error patterns.
// Some upstreams fail INSIDE a 200 SSE stream; without this the failure is
// piped to the client verbatim and account/combo fallback never triggers
// (see AGENTS.md "HTTP 200 in-stream errors"). Fail-open: no patterns → pass-through.
if (providerResponse.ok && stream) {
const peeked = await maybeRejectEarlyStreamError(
providerResponse,
streamErrorPatterns?.[provider],
{ signal: streamController.signal },
);
if (!peeked.ok) {
const { message } = await parseUpstreamError(peeked).catch(() => ({ message: "Stream error pattern matched" }));
trackPendingRequest(model, provider, connectionId, false, true);
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
if (log?.errorLine) {
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms (in-stream)\n ${message}`);
}
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, message);
}
providerResponse = peeked;
}
const trackDone = () => trackPendingRequest(model, provider, connectionId, false); const trackDone = () => trackPendingRequest(model, provider, connectionId, false);
// Provider forced streaming but client wants JSON // Provider forced streaming but client wants JSON
@@ -457,7 +513,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Streaming response // Streaming response
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx }); const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId }); return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials });
} }
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) { export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {

View File

@@ -7,7 +7,8 @@ import { createErrorResult } from "../../utils/error.js";
import { HTTP_STATUS } from "../../config/runtimeConfig.js"; import { HTTP_STATUS } from "../../config/runtimeConfig.js";
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js"; import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { saveRequestDetail } from "@/lib/usageDb.js";
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
import { decloakToolNames } from "../../utils/claudeCloaking.js"; import { decloakToolNames } from "../../utils/claudeCloaking.js";
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js"; import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
@@ -281,7 +282,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
/** /**
* Handle non-streaming response from provider. * Handle non-streaming response from provider.
*/ */
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log }) { export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log, streamErrorPatterns }) {
trackDone(); trackDone();
const contentType = providerResponse.headers.get("content-type") || ""; const contentType = providerResponse.headers.get("content-type") || "";
let responseBody; let responseBody;
@@ -316,6 +317,21 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
// Decloak tool_use names once on raw Claude body, before any translation (INPUT side) // Decloak tool_use names once on raw Claude body, before any translation (INPUT side)
responseBody = decloakToolNames(responseBody, toolNameMap); responseBody = decloakToolNames(responseBody, toolNameMap);
// Config-driven in-stream error detection: the HTTP call succeeded but the
// assembled content signals an upstream failure — treat it as an error so
// account/combo fallback and FAILED logging kick in (AGENTS.md hook #3).
const matchedPattern = matchStreamErrorPatterns(
streamErrorPatterns?.[provider],
responseBody?.choices?.[0]?.message?.content || responseBody?.content || "",
);
if (matchedPattern) {
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
if (log?.errorLine) {
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms (in-stream)\n Stream error pattern matched: ${matchedPattern}`);
}
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Stream error pattern matched: ${matchedPattern}`);
}
const usage = extractUsageFromResponse(responseBody); const usage = extractUsageFromResponse(responseBody);
appendLog({ tokens: usage, status: "200 OK" }); appendLog({ tokens: usage, status: "200 OK" });
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true }); saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
@@ -372,7 +388,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
const totalLatency = Date.now() - requestStartTime; const totalLatency = Date.now() - requestStartTime;
saveRequestDetail(buildRequestDetail({ saveRequestDetail(buildRequestDetail({
provider, model, connectionId, provider, model, connectionId, apiKey,
latency: { ttft: totalLatency, total: totalLatency }, latency: { ttft: totalLatency, total: totalLatency },
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 }, tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
request: extractRequestConfig(body, stream), request: extractRequestConfig(body, stream),

View File

@@ -1,4 +1,4 @@
import { saveRequestUsage, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { saveRequestUsage, saveRequestDetail } from "@/lib/usageDb.js";
import { COLORS } from "../../utils/stream.js"; import { COLORS } from "../../utils/stream.js";
import { canonicalizeUsage } from "../../utils/usageTracking.js"; import { canonicalizeUsage } from "../../utils/usageTracking.js";
@@ -25,10 +25,16 @@ export function extractUsageFromResponse(responseBody) {
if (!responseBody || typeof responseBody !== "object") return null; if (!responseBody || typeof responseBody !== "object") return null;
// Claude format // Claude format
// Note: OpenAI Responses usage ({input_tokens, input_tokens_details:{cached_tokens}})
// also matches this branch. Its prompt is cache-INCLUSIVE and its cache rides in
// input_tokens_details, so emit it as cached_tokens — the convention
// canonicalizeUsage() passes through without folding. Reading it here keeps
// cache accounting correct for /v1/responses and codex traffic.
if (responseBody.usage?.input_tokens !== undefined) { if (responseBody.usage?.input_tokens !== undefined) {
return { return {
prompt_tokens: responseBody.usage.input_tokens || 0, prompt_tokens: responseBody.usage.input_tokens || 0,
completion_tokens: responseBody.usage.output_tokens || 0, completion_tokens: responseBody.usage.output_tokens || 0,
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.input_tokens_details?.cached_tokens,
cache_read_input_tokens: responseBody.usage.cache_read_input_tokens, cache_read_input_tokens: responseBody.usage.cache_read_input_tokens,
cache_creation_input_tokens: responseBody.usage.cache_creation_input_tokens cache_creation_input_tokens: responseBody.usage.cache_creation_input_tokens
}; };
@@ -39,7 +45,7 @@ export function extractUsageFromResponse(responseBody) {
return { return {
prompt_tokens: responseBody.usage.prompt_tokens || 0, prompt_tokens: responseBody.usage.prompt_tokens || 0,
completion_tokens: responseBody.usage.completion_tokens || 0, completion_tokens: responseBody.usage.completion_tokens || 0,
cached_tokens: responseBody.usage.prompt_tokens_details?.cached_tokens, cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.prompt_tokens_details?.cached_tokens,
reasoning_tokens: responseBody.usage.completion_tokens_details?.reasoning_tokens reasoning_tokens: responseBody.usage.completion_tokens_details?.reasoning_tokens
}; };
} }
@@ -58,11 +64,21 @@ export function extractUsageFromResponse(responseBody) {
return null; return null;
} }
// Mask API keys before they reach the requestDetails data blob / DB column.
// Only the prefix is kept — enough to distinguish keys without leaking them.
export function maskApiKey(key) {
if (!key || typeof key !== "string") return undefined;
const trimmed = key.trim();
if (trimmed.length <= 8) return trimmed.charAt(0) + "***";
return trimmed.slice(0, 8) + "***";
}
export function buildRequestDetail(base, overrides = {}) { export function buildRequestDetail(base, overrides = {}) {
return { return {
provider: base.provider || "unknown", provider: base.provider || "unknown",
model: base.model || "unknown", model: base.model || "unknown",
connectionId: base.connectionId || undefined, connectionId: base.connectionId || undefined,
apiKey: maskApiKey(base.apiKey),
timestamp: new Date().toISOString(), timestamp: new Date().toISOString(),
latency: base.latency || { ttft: 0, total: 0 }, latency: base.latency || { ttft: 0, total: 0 },
tokens: base.tokens || { prompt_tokens: 0, completion_tokens: 0 }, tokens: base.tokens || { prompt_tokens: 0, completion_tokens: 0 },

View File

@@ -1,22 +1,24 @@
import { convertResponsesStreamToJson } from "../../transformer/streamToJsonConverter.js"; import { convertResponsesStreamToJson } from "../../transformer/streamToJsonConverter.js";
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
import { createErrorResult } from "../../utils/error.js"; import { createErrorResult } from "../../utils/error.js";
import { HTTP_STATUS } from "../../config/runtimeConfig.js"; import { HTTP_STATUS } from "../../config/runtimeConfig.js";
import { FORMATS } from "../../translator/formats.js"; import { FORMATS } from "../../translator/formats.js";
import { PROVIDERS } from "../../config/providers.js"; import { PROVIDERS } from "../../config/providers.js";
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
import { saveRequestDetail } from "@/lib/usageDb.js";
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js"; import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape // Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES; const isResponsesProvider = (p) =>
import { saveRequestDetail, appendRequestLog } from "@/lib/usageDb.js"; PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
function textFromResponsesMessageItem(item) { function textFromResponsesMessageItem(item) {
if (!item?.content || !Array.isArray(item.content)) return ""; if (!item?.content || !Array.isArray(item.content)) return "";
const byType = item.content.find((c) => c.type === "output_text"); const byType = item.content.find((c) => c.type === "output_text");
if (typeof byType?.text === "string") return byType.text; if (typeof byType?.text === "string") return byType.text;
const anyText = item.content.find((c) => typeof c.text === "string"); const anyText = item.content.find((c) => typeof c.text === "string");
if (typeof anyText?.text === "string") return anyText.text; if (typeof anyText?.text === "string") return anyText.text;
return ""; return "";
} }
/** /**
@@ -24,15 +26,15 @@ function textFromResponsesMessageItem(item) {
* Early message blocks often have empty output_text; the user-visible answer is usually in the last non-empty message. * Early message blocks often have empty output_text; the user-visible answer is usually in the last non-empty message.
*/ */
function pickAssistantMessageForChatCompletion(output) { function pickAssistantMessageForChatCompletion(output) {
if (!Array.isArray(output)) return { msgItem: null, textContent: null }; if (!Array.isArray(output)) return { msgItem: null, textContent: null };
const messages = output.filter((item) => item?.type === "message"); const messages = output.filter((item) => item?.type === "message");
if (messages.length === 0) return { msgItem: null, textContent: null }; if (messages.length === 0) return { msgItem: null, textContent: null };
for (let i = messages.length - 1; i >= 0; i--) { for (let i = messages.length - 1; i >= 0; i--) {
const text = textFromResponsesMessageItem(messages[i]); const text = textFromResponsesMessageItem(messages[i]);
if (text.length > 0) return { msgItem: messages[i], textContent: text }; if (text.length > 0) return { msgItem: messages[i], textContent: text };
} }
const last = messages[messages.length - 1]; const last = messages[messages.length - 1];
return { msgItem: last, textContent: textFromResponsesMessageItem(last) }; return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
} }
/** /**
@@ -110,250 +112,414 @@ function chatCompletionToResponses(responseBody, customToolNames = null) {
* Used when provider forces streaming but client wants non-streaming. * Used when provider forces streaming but client wants non-streaming.
*/ */
export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) { export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
const chunks = []; const chunks = [];
let streamError = null; let streamError = null;
for (const line of String(rawSSE || "").split("\n")) { for (const line of String(rawSSE || "").split("\n")) {
const trimmed = line.trim(); const trimmed = line.trim();
if (!trimmed.startsWith("data:")) continue; if (!trimmed.startsWith("data:")) continue;
const payload = trimmed.slice(5).trim(); const payload = trimmed.slice(5).trim();
if (!payload || payload === "[DONE]") continue; if (!payload || payload === "[DONE]") continue;
try { try {
const chunk = JSON.parse(payload); const chunk = JSON.parse(payload);
if (chunk?.error) streamError = chunk.error; if (chunk?.error) streamError = chunk.error;
else chunks.push(chunk); else chunks.push(chunk);
} catch { /* ignore malformed lines */ } } catch {
} /* ignore malformed lines */
}
}
if (streamError) return { error: streamError }; if (streamError) return { error: streamError };
if (chunks.length === 0) return null; if (chunks.length === 0) return null;
const first = chunks[0]; const first = chunks[0];
const contentParts = []; const contentParts = [];
const reasoningParts = []; const reasoningParts = [];
const toolCallMap = new Map(); // index -> { id, type, function: { name, arguments } } const toolCallMap = new Map(); // index -> { id, type, function: { name, arguments } }
let finishReason = "stop"; let finishReason = "stop";
let usage = null; let usage = null;
for (const chunk of chunks) { for (const chunk of chunks) {
const choice = chunk?.choices?.[0]; const choice = chunk?.choices?.[0];
const delta = choice?.delta || {}; const delta = choice?.delta || {};
if (typeof delta.content === "string" && delta.content.length > 0) contentParts.push(delta.content); if (typeof delta.content === "string" && delta.content.length > 0)
if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0) reasoningParts.push(delta.reasoning_content); contentParts.push(delta.content);
if (choice?.finish_reason) finishReason = choice.finish_reason; if (
if (chunk?.usage && typeof chunk.usage === "object") usage = chunk.usage; typeof delta.reasoning_content === "string" &&
delta.reasoning_content.length > 0
)
reasoningParts.push(delta.reasoning_content);
if (choice?.finish_reason) finishReason = choice.finish_reason;
if (chunk?.usage && typeof chunk.usage === "object") usage = chunk.usage;
// Accumulate tool_calls from streaming deltas // Accumulate tool_calls from streaming deltas
if (Array.isArray(delta.tool_calls)) { if (Array.isArray(delta.tool_calls)) {
for (const tc of delta.tool_calls) { for (const tc of delta.tool_calls) {
const idx = tc.index ?? 0; const idx = tc.index ?? 0;
if (!toolCallMap.has(idx)) { if (!toolCallMap.has(idx)) {
toolCallMap.set(idx, { id: tc.id || "", type: "function", function: { name: "", arguments: "" } }); toolCallMap.set(idx, {
} id: tc.id || "",
const existing = toolCallMap.get(idx); type: "function",
if (tc.id) existing.id = tc.id; function: { name: "", arguments: "" },
if (tc.function?.name) existing.function.name += tc.function.name; });
if (tc.function?.arguments) existing.function.arguments += tc.function.arguments; }
} const existing = toolCallMap.get(idx);
} if (tc.id) existing.id = tc.id;
} if (tc.function?.name) existing.function.name += tc.function.name;
if (tc.function?.arguments)
existing.function.arguments += tc.function.arguments;
}
}
}
const message = { role: "assistant", content: contentParts.join("") || (toolCallMap.size > 0 ? null : "") }; const message = {
if (reasoningParts.length > 0) message.reasoning_content = reasoningParts.join(""); role: "assistant",
if (toolCallMap.size > 0) { content: contentParts.join("") || (toolCallMap.size > 0 ? null : ""),
message.tool_calls = [...toolCallMap.entries()].sort((a, b) => a[0] - b[0]).map(([, tc]) => tc); };
} if (reasoningParts.length > 0)
message.reasoning_content = reasoningParts.join("");
if (toolCallMap.size > 0) {
message.tool_calls = [...toolCallMap.entries()]
.sort((a, b) => a[0] - b[0])
.map(([, tc]) => tc);
}
const result = { const result = {
id: first.id || `chatcmpl-${Date.now()}`, id: first.id || `chatcmpl-${Date.now()}`,
object: "chat.completion", object: "chat.completion",
created: first.created || Math.floor(Date.now() / 1000), created: first.created || Math.floor(Date.now() / 1000),
model: first.model || fallbackModel || "unknown", model: first.model || fallbackModel || "unknown",
choices: [{ index: 0, message, finish_reason: finishReason }] choices: [{ index: 0, message, finish_reason: finishReason }],
}; };
if (usage) result.usage = usage; if (usage) result.usage = usage;
return result; return result;
} }
/** /**
* Handle case: provider forced streaming but client wants JSON. * Handle case: provider forced streaming but client wants JSON.
* Supports both Codex/Responses API SSE and standard Chat Completions SSE. * Supports both Codex/Responses API SSE and standard Chat Completions SSE.
*/ */
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, targetFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, customToolNames, trackDone, appendLog, reqTag, log }) { export async function handleForcedSSEToJson({
const contentType = providerResponse.headers.get("content-type") || ""; providerResponse,
const isSSE = contentType.includes("text/event-stream") || (contentType === "" && isResponsesProvider(provider)); sourceFormat,
if (!isSSE) return null; // not handled here targetFormat,
provider,
model,
body,
stream,
translatedBody,
finalBody,
requestStartTime,
connectionId,
apiKey,
clientRawRequest,
onRequestSuccess,
customToolNames,
trackDone,
appendLog,
reqTag,
log,
streamErrorPatterns,
}) {
const contentType = providerResponse.headers.get("content-type") || "";
const isSSE =
contentType.includes("text/event-stream") ||
(contentType === "" && isResponsesProvider(provider));
if (!isSSE) return null; // not handled here
trackDone(); trackDone();
const ctx = { const ctx = {
provider, model, connectionId, provider,
request: extractRequestConfig(body, stream), model,
providerRequest: finalBody || translatedBody || null connectionId,
}; request: extractRequestConfig(body, stream),
providerRequest: finalBody || translatedBody || null,
};
// Codex/Responses API SSE path // Codex/Responses API SSE path
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in), // Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
// not the client format: a Responses-API client behind a chat-native forced-streaming // not the client format: a Responses-API client behind a chat-native forced-streaming
// provider still receives chat SSE chunks, which must go through the standard path. // provider still receives chat SSE chunks, which must go through the standard path.
const isCodexResponsesApi = isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES; const isCodexResponsesApi =
if (isCodexResponsesApi) { isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
try { if (isCodexResponsesApi) {
const jsonResponse = await convertResponsesStreamToJson(providerResponse.body); try {
if (onRequestSuccess) await onRequestSuccess(); const jsonResponse = await convertResponsesStreamToJson(
providerResponse.body,
);
if (onRequestSuccess) await onRequestSuccess();
const usage = jsonResponse.usage || {}; const usage = jsonResponse.usage || {};
appendLog({ tokens: usage, status: "200 OK" }); appendLog({ tokens: usage, status: "200 OK" });
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true }); saveUsageStats({
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } })); provider,
model,
tokens: usage,
connectionId,
apiKey,
endpoint: clientRawRequest?.endpoint,
silent: true,
});
if (log?.line)
log.line(
reqTag,
"📊",
formatDoneLine({
usage,
latency: { total: Date.now() - requestStartTime },
}),
);
// Same cache-inclusive total for the recorded detail, so the DB and the // Same cache-inclusive total for the recorded detail, so the DB and the
// client-facing usage can never disagree. // client-facing usage can never disagree.
const inTokensForLog = (usage.input_tokens || 0) const inTokensForLog = (usage.input_tokens || 0)
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0) + (usage.cache_read_input_tokens || usage.cached_tokens || 0)
+ (usage.cache_creation_input_tokens || 0); + (usage.cache_creation_input_tokens || 0);
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(jsonResponse.output); const { msgItem, textContent } = pickAssistantMessageForChatCompletion(
const totalLatency = Date.now() - requestStartTime; jsonResponse.output,
);
const totalLatency = Date.now() - requestStartTime;
saveRequestDetail(buildRequestDetail({ saveRequestDetail(
...ctx, buildRequestDetail(
latency: { ttft: totalLatency, total: totalLatency }, {
tokens: { prompt_tokens: inTokensForLog, completion_tokens: usage.output_tokens || 0 }, ...ctx,
response: { content: textContent, thinking: null, finish_reason: jsonResponse.status || "unknown" }, apiKey,
status: "success" latency: { ttft: totalLatency, total: totalLatency },
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {}); tokens: {
prompt_tokens: inTokensForLog,
completion_tokens: usage.output_tokens || 0,
},
response: {
content: textContent,
thinking: null,
finish_reason: jsonResponse.status || "unknown",
},
status: "success",
},
{ endpoint: clientRawRequest?.endpoint || null },
),
).catch(() => {});
// Client is Responses API → return as-is // Client is Responses API → return as-is
if (sourceFormat === FORMATS.OPENAI_RESPONSES) { if (sourceFormat === FORMATS.OPENAI_RESPONSES) {
return { success: true, response: new Response(JSON.stringify(jsonResponse), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) }; return {
} success: true,
response: new Response(JSON.stringify(jsonResponse), {
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
},
}),
};
}
// Build client-format response. // Build client-format response.
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing // input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
// only input+output under-reports prompt_tokens — measured: 2012 reported // only input+output under-reports prompt_tokens — measured: 2012 reported
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache // where the real prompt was ~5344 with 5332 served from cache. Fold the cache
// counters in, and keep them visible in prompt_tokens_details so a client can // counters in, and keep them visible in prompt_tokens_details so a client can
// tell a cache hit from a small prompt. // tell a cache hit from a small prompt.
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0; const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
const cacheCreate = usage.cache_creation_input_tokens || 0; const cacheCreate = usage.cache_creation_input_tokens || 0;
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate; const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
const outTokens = usage.output_tokens || 0; const outTokens = usage.output_tokens || 0;
const cacheDetails = (cacheRead > 0 || cacheCreate > 0) const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
? { prompt_tokens_details: { ? {
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}), prompt_tokens_details: {
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } } ...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
: {}; ...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
let finalResp; : {};
let finalResp;
// Extract tool calls from Responses API output (function_call items) // Extract tool calls from Responses API output (function_call items)
const funcCallItems = (jsonResponse.output || []).filter(item => item.type === "function_call"); const funcCallItems = (jsonResponse.output || []).filter(
const toolCalls = funcCallItems.map((item, idx) => ({ (item) => item.type === "function_call",
id: item.call_id || `call_${item.name}_${Date.now()}_${idx}`, );
type: "function", const toolCalls = funcCallItems.map((item, idx) => ({
function: { id: item.call_id || `call_${item.name}_${Date.now()}_${idx}`,
name: item.name, type: "function",
arguments: typeof item.arguments === "string" ? item.arguments : JSON.stringify(item.arguments || {}) function: {
} name: item.name,
})); arguments:
const hasToolCalls = toolCalls.length > 0; typeof item.arguments === "string"
? item.arguments
: JSON.stringify(item.arguments || {}),
},
}));
const hasToolCalls = toolCalls.length > 0;
if (sourceFormat === FORMATS.ANTIGRAVITY || sourceFormat === FORMATS.GEMINI || sourceFormat === FORMATS.GEMINI_CLI) { if (
finalResp = { sourceFormat === FORMATS.ANTIGRAVITY ||
response: { sourceFormat === FORMATS.GEMINI ||
candidates: [{ content: { role: "model", parts: [{ text: textContent || "" }] }, finishReason: "STOP", index: 0 }], sourceFormat === FORMATS.GEMINI_CLI
usageMetadata: { promptTokenCount: inTokens, candidatesTokenCount: outTokens, totalTokenCount: inTokens + outTokens }, ) {
modelVersion: model, finalResp = {
responseId: jsonResponse.id || `resp_${Date.now()}` response: {
} candidates: [
}; {
} else { content: {
const message = { role: "assistant", content: textContent || (hasToolCalls ? null : "") }; role: "model",
if (hasToolCalls) message.tool_calls = toolCalls; parts: [{ text: textContent || "" }],
const responseDone = jsonResponse.status === "completed" || jsonResponse.status === "done"; },
const finishReason = hasToolCalls ? "tool_calls" : (responseDone ? "stop" : (jsonResponse.status || "stop")); finishReason: "STOP",
finalResp = { index: 0,
id: jsonResponse.id || `chatcmpl-${Date.now()}`, },
object: "chat.completion", ],
created: jsonResponse.created_at || Math.floor(Date.now() / 1000), usageMetadata: {
model: jsonResponse.model || model, promptTokenCount: inTokens,
choices: [{ index: 0, message, finish_reason: finishReason }], candidatesTokenCount: outTokens,
usage: { prompt_tokens: inTokens, completion_tokens: outTokens, total_tokens: inTokens + outTokens, ...cacheDetails } totalTokenCount: inTokens + outTokens,
}; },
} modelVersion: model,
responseId: jsonResponse.id || `resp_${Date.now()}`,
},
};
} else {
const message = {
role: "assistant",
content: textContent || (hasToolCalls ? null : ""),
};
if (hasToolCalls) message.tool_calls = toolCalls;
const responseDone =
jsonResponse.status === "completed" || jsonResponse.status === "done";
const finishReason = hasToolCalls
? "tool_calls"
: responseDone
? "stop"
: jsonResponse.status || "stop";
finalResp = {
id: jsonResponse.id || `chatcmpl-${Date.now()}`,
object: "chat.completion",
created: jsonResponse.created_at || Math.floor(Date.now() / 1000),
model: jsonResponse.model || model,
choices: [{ index: 0, message, finish_reason: finishReason }],
usage: {
prompt_tokens: inTokens,
completion_tokens: outTokens,
total_tokens: inTokens + outTokens,
...cacheDetails,
},
};
}
return { success: true, response: new Response(JSON.stringify(finalResp), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) }; return {
} catch (err) { success: true,
console.error("[ChatCore] Responses API SSE→JSON failed:", err); response: new Response(JSON.stringify(finalResp), {
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON"); headers: {
} "Content-Type": "application/json",
} "Access-Control-Allow-Origin": "*",
},
}),
};
} catch (err) {
console.error("[ChatCore] Responses API SSE→JSON failed:", err);
return createErrorResult(
HTTP_STATUS.BAD_GATEWAY,
"Failed to convert streaming response to JSON",
);
}
}
// Standard Chat Completions SSE path // Standard Chat Completions SSE path
try { try {
const sseText = await providerResponse.text(); const sseText = await providerResponse.text();
const parsed = parseSSEToOpenAIResponse(sseText, model); const parsed = parseSSEToOpenAIResponse(sseText, model);
if (!parsed) return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Invalid SSE response for non-streaming request"); if (!parsed)
if (parsed.error) { return createErrorResult(
return createErrorResult( HTTP_STATUS.BAD_GATEWAY,
HTTP_STATUS.BAD_GATEWAY, "Invalid SSE response for non-streaming request",
parsed.error.message || "Upstream SSE stream failed" );
); if (parsed.error) {
} return createErrorResult(
HTTP_STATUS.BAD_GATEWAY,
parsed.error.message || "Upstream SSE stream failed",
);
}
if (onRequestSuccess) await onRequestSuccess(); // Config-driven in-stream error detection: the request "succeeded" at the
// HTTP level, but the content signals an upstream failure — treat it as an
// error so account/combo fallback and FAILED logging kick in.
const matchedPattern = matchStreamErrorPatterns(
streamErrorPatterns?.[provider],
parsed.choices?.[0]?.message?.content || "",
);
if (matchedPattern) {
return createErrorResult(
HTTP_STATUS.BAD_GATEWAY,
`Stream error pattern matched: ${matchedPattern}`,
);
}
const usage = parsed.usage || {}; if (onRequestSuccess) await onRequestSuccess();
appendLog({ tokens: usage, status: "200 OK" });
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
const totalLatency = Date.now() - requestStartTime; const usage = parsed.usage || {};
saveRequestDetail(buildRequestDetail({ appendLog({ tokens: usage, status: "200 OK" });
...ctx, saveUsageStats({
latency: { ttft: totalLatency, total: totalLatency }, provider,
tokens: usage, model,
response: { tokens: usage,
content: parsed.choices?.[0]?.message?.content || null, connectionId,
thinking: parsed.choices?.[0]?.message?.reasoning_content || null, apiKey,
finish_reason: parsed.choices?.[0]?.finish_reason || "unknown" endpoint: clientRawRequest?.endpoint,
}, silent: true,
status: "success" });
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {}); if (log?.line)
log.line(
reqTag,
"📊",
formatDoneLine({
usage,
latency: { total: Date.now() - requestStartTime },
}),
);
// Re-attach usage explicitly. This handler already HAS the correct usage — it is // Re-attach usage explicitly. This handler already HAS the correct usage — it is
// the same object written to the usage DB, and for a cached Claude request that DB // the same object written to the usage DB, and for a cached Claude request that DB
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving // row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched // no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
// on both sides). Whatever drops it between assembly and serialisation, the client // on both sides). Whatever drops it between assembly and serialisation, the client
// must not be left unable to account for its own token spend: a caller cannot tell // must not be left unable to account for its own token spend: a caller cannot tell
// a 90%-cached request from a cheap one without this. // a 90%-cached request from a cheap one without this.
if (usage && Object.keys(usage).length > 0) parsed.usage = usage; if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
// Strip reasoning_content only when content is non-empty. // Strip reasoning_content only when content is non-empty.
// When content is empty (e.g. thinking models that used all tokens for reasoning), // When content is empty (e.g. thinking models that used all tokens for reasoning),
// reasoning_content is the only useful output and must be preserved. // reasoning_content is the only useful output and must be preserved.
// Previously this was unconditional, which broke Qwen3.5, Claude extended thinking, etc. // Previously this was unconditional, which broke Qwen3.5, Claude extended thinking, etc.
if (parsed?.choices) { if (parsed?.choices) {
for (const choice of parsed.choices) { for (const choice of parsed.choices) {
if (choice?.message?.reasoning_content && choice.message.content) { if (choice?.message?.reasoning_content && choice.message.content) {
delete choice.message.reasoning_content; delete choice.message.reasoning_content;
} }
} }
} }
// A Responses-format client (e.g. Codex) forced this provider to stream, // A Responses-format client (e.g. Codex) forced this provider to stream,
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions // but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
// body; convert it to the Responses `output` shape so tool_calls are not // body; convert it to the Responses `output` shape so tool_calls are not
// lost on the non-streaming return path. Inlined (not imported from // lost on the non-streaming return path. Inlined (not imported from
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler // nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
// already imports parseSSEToOpenAIResponse from this module. // already imports parseSSEToOpenAIResponse from this module.
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
? chatCompletionToResponses(parsed, customToolNames) ? chatCompletionToResponses(parsed, customToolNames)
: parsed; : parsed;
return { success: true, response: new Response(JSON.stringify(finalBody), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) }; return {
} catch (err) { success: true,
console.error("[ChatCore] Chat Completions SSE→JSON failed:", err); response: new Response(JSON.stringify(finalBody), {
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON"); headers: {
} "Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
},
}),
};
} catch (err) {
console.error("[ChatCore] Chat Completions SSE→JSON failed:", err);
return createErrorResult(
HTTP_STATUS.BAD_GATEWAY,
"Failed to convert streaming response to JSON",
);
}
} }

View File

@@ -1,28 +1,37 @@
import { FORMATS } from "../../translator/formats.js"; import { FORMATS } from "../../translator/formats.js";
import { needsTranslation } from "../../translator/index.js"; import { needsTranslation } from "../../translator/index.js";
import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js"; import {
createSSETransformStreamWithLogger,
createPassthroughStreamWithLogger,
} from "../../utils/stream.js";
import { pipeWithDisconnect } from "../../utils/streamHandler.js"; import { pipeWithDisconnect } from "../../utils/streamHandler.js";
import { PROVIDERS } from "../../config/providers.js"; import { PROVIDERS } from "../../config/providers.js";
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js"; import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js"; import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import {
buildRequestDetail,
extractRequestConfig,
saveUsageStats,
formatDoneLine,
} from "./requestDetail.js";
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
import { saveRequestDetail } from "@/lib/usageDb.js"; import { saveRequestDetail } from "@/lib/usageDb.js";
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js"; import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
// Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat. // Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat.
// Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI. // Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI.
const CODEX_SOURCE_TO_TARGET = { const CODEX_SOURCE_TO_TARGET = {
[FORMATS.OPENAI_RESPONSES]: FORMATS.OPENAI_RESPONSES, [FORMATS.OPENAI_RESPONSES]: FORMATS.OPENAI_RESPONSES,
[FORMATS.CLAUDE]: FORMATS.CLAUDE, [FORMATS.CLAUDE]: FORMATS.CLAUDE,
[FORMATS.ANTIGRAVITY]: FORMATS.ANTIGRAVITY, [FORMATS.ANTIGRAVITY]: FORMATS.ANTIGRAVITY,
[FORMATS.GEMINI]: FORMATS.ANTIGRAVITY, [FORMATS.GEMINI]: FORMATS.ANTIGRAVITY,
[FORMATS.GEMINI_CLI]: FORMATS.ANTIGRAVITY, [FORMATS.GEMINI_CLI]: FORMATS.ANTIGRAVITY,
}; };
/** /**
* Determine which SSE transform stream to use based on provider/format. * Determine which SSE transform stream to use based on provider/format.
*/ */
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) { function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) {
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli"); const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format // Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES; const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
@@ -30,20 +39,28 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
if (needsCodexTranslation) { if (needsCodexTranslation) {
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI; const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames); return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
} }
if (needsTranslation(targetFormat, sourceFormat)) { if (needsTranslation(targetFormat, sourceFormat)) {
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames); return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
} }
return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey); return createPassthroughStreamWithLogger(
provider,
reqLogger,
model,
connectionId,
body,
onStreamComplete,
apiKey,
);
} }
/** /**
* Handle streaming response — pipe provider SSE through transform stream to client. * Handle streaming response — pipe provider SSE through transform stream to client.
*/ */
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) { export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) {
if (onRequestSuccess) { if (onRequestSuccess) {
Promise.resolve() Promise.resolve()
.then(onRequestSuccess) .then(onRequestSuccess)
@@ -52,93 +69,192 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
}); });
} }
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error // When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
// page), piping it through the SSE transform stream causes Next.js // page), piping it through the SSE transform stream causes Next.js
// "failed to pipe response" and crashes the chat router. Read the body, // "failed to pipe response" and crashes the chat router. Read the body,
// pull a short human-readable message from the <title>, sanitize it, and // pull a short human-readable message from the <title>, sanitize it, and
// return a clean JSON error instead. The message is stripped of HTML tags // return a clean JSON error instead. The message is stripped of HTML tags
// and clamped so untrusted upstream text never reaches the client verbatim // and clamped so untrusted upstream text never reaches the client verbatim
// (the UI may render error.message as HTML). // (the UI may render error.message as HTML).
const upstreamContentType = (providerResponse.headers.get('content-type') || '').toLowerCase(); const upstreamContentType = (
if (upstreamContentType && !upstreamContentType.includes('text/event-stream') && !upstreamContentType.includes('application/json')) { providerResponse.headers.get("content-type") || ""
const bodyText = await providerResponse.text().catch(() => ''); ).toLowerCase();
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i); if (
const sanitizedTitle = (titleMatch?.[1] || '').replace(/<[^>]*>/g, '').replace(/[\r\n]+/g, ' ').trim().slice(0, 160); upstreamContentType &&
const shortMsg = sanitizedTitle !upstreamContentType.includes("text/event-stream") &&
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`); !upstreamContentType.includes("application/json")
const status = providerResponse.status || 502; ) {
if (log?.errorLine) log.errorLine(reqTag, "✗", `BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`); const bodyText = await providerResponse.text().catch(() => "");
else console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`); const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`)); const sanitizedTitle = (titleMatch?.[1] || "")
return { .replace(/<[^>]*>/g, "")
success: false, .replace(/[\r\n]+/g, " ")
response: new Response(JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }), { .trim()
status, .slice(0, 160);
headers: { 'Content-Type': 'application/json', 'Access-Control-Allow-Origin': '*' }, const shortMsg =
}), sanitizedTitle ||
}; (bodyText.length < 200
} ? bodyText
.replace(/<[^>]*>/g, "")
.trim()
.slice(0, 160)
: `Upstream returned non-SSE response (${upstreamContentType})`);
const status = providerResponse.status || 502;
if (log?.errorLine)
log.errorLine(
reqTag,
"✗",
`BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`,
);
else
console.warn(
`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`,
);
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
return {
success: false,
response: new Response(
JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }),
{
status,
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*",
},
},
),
};
}
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }); const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event // Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES; const isResponsesPassthrough =
const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null; sourceFormat === FORMATS.OPENAI_RESPONSES &&
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS; targetFormat === FORMATS.OPENAI_RESPONSES;
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs); const onAbortTerminal = isResponsesPassthrough
? buildAbortedResponsesTerminalBytes
: null;
const stallTimeoutMs =
PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
const transformedBody = pipeWithDisconnect(
providerResponse,
transformStream,
streamController,
onAbortTerminal,
stallTimeoutMs,
);
saveRequestDetail(buildRequestDetail({ saveRequestDetail(
provider, model, connectionId, buildRequestDetail(
latency: { ttft: 0, total: Date.now() - requestStartTime }, {
tokens: { prompt_tokens: 0, completion_tokens: 0 }, provider,
request: extractRequestConfig(body, stream), model,
providerRequest: finalBody || translatedBody || null, connectionId,
providerResponse: "[Streaming - raw response not captured]", apiKey,
response: { content: "[Streaming in progress...]", thinking: null, type: "streaming" }, latency: { ttft: 0, total: Date.now() - requestStartTime },
pxpipe, tokens: { prompt_tokens: 0, completion_tokens: 0 },
status: "success" request: extractRequestConfig(body, stream),
}, { id: streamDetailId })).catch(err => { providerRequest: finalBody || translatedBody || null,
console.error("[RequestDetail] Failed to save streaming request:", err.message); providerResponse: "[Streaming - raw response not captured]",
}); response: {
content: "[Streaming in progress...]",
thinking: null,
type: "streaming",
},
pxpipe,
status: "success",
},
{ id: streamDetailId },
),
).catch((err) => {
console.error(
"[RequestDetail] Failed to save streaming request:",
err.message,
);
});
return { return {
success: true, success: true,
response: new Response(transformedBody, { headers: SSE_HEADERS }) response: new Response(transformedBody, { headers: SSE_HEADERS }),
}; };
} }
/** /**
* Build onStreamComplete callback for streaming usage tracking. * Build onStreamComplete callback for streaming usage tracking.
*/ */
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest, pxpipe, reqTag, log }) { export function buildOnStreamComplete({
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`; provider,
model,
connectionId,
apiKey,
requestStartTime,
body,
stream,
finalBody,
translatedBody,
clientRawRequest,
pxpipe,
reqTag,
log,
streamErrorPatterns,
}) {
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
const onStreamComplete = (contentObj, usage, ttftAt) => { const onStreamComplete = (contentObj, usage, ttftAt) => {
const latency = { const latency = {
ttft: ttftAt ? ttftAt - requestStartTime : Date.now() - requestStartTime, ttft: ttftAt ? ttftAt - requestStartTime : Date.now() - requestStartTime,
total: Date.now() - requestStartTime total: Date.now() - requestStartTime,
}; };
const safeContent = contentObj?.content || "[Empty streaming response]"; const safeContent = contentObj?.content || "[Empty streaming response]";
const safeThinking = contentObj?.thinking || null; const safeThinking = contentObj?.thinking || null;
const rawProviderText = typeof contentObj?.rawProviderText === "string" ? contentObj.rawProviderText : "";
saveRequestDetail(buildRequestDetail({ saveRequestDetail(
provider, model, connectionId, buildRequestDetail(
latency, {
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 }, provider,
request: extractRequestConfig(body, stream), model,
providerRequest: finalBody || translatedBody || null, connectionId,
providerResponse: safeContent, apiKey,
response: { content: safeContent, thinking: safeThinking, type: "streaming" }, latency,
pxpipe, tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
status: "success" request: extractRequestConfig(body, stream),
}, { id: streamDetailId })).catch(err => { providerRequest: finalBody || translatedBody || null,
console.error("[RequestDetail] Failed to update streaming content:", err.message); providerResponse: rawProviderText || safeContent,
}); response: {
content: safeContent,
thinking: safeThinking,
type: "streaming",
},
pxpipe,
status: streamStatusForContent(
streamErrorPatterns?.[provider],
safeContent,
),
},
{ id: streamDetailId },
),
).catch((err) => {
console.error(
"[RequestDetail] Failed to update streaming content:",
err.message,
);
});
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative) // Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, label: "STREAM USAGE", silent: true }); saveUsageStats({
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency })); provider,
}; model,
tokens: usage,
connectionId,
apiKey,
endpoint: clientRawRequest?.endpoint,
label: "STREAM USAGE",
silent: true,
});
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency }));
};
return { onStreamComplete, streamDetailId }; return { onStreamComplete, streamDetailId };
} }

View File

@@ -1,4 +1,4 @@
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa // Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama
// Returns normalized shape across all providers // Returns normalized shape across all providers
const DEFAULT_TIMEOUT_MS = 15000; const DEFAULT_TIMEOUT_MS = 15000;
@@ -56,8 +56,8 @@ function parseJinaTitle(text) {
return m ? m[1].trim() : null; return m ? m[1].trim() : null;
} }
function buildData({ provider, url, title, format, text, costUsd, responseMs, upstreamMs }) { function buildData({ provider, url, title, format, text, links, costUsd, responseMs, upstreamMs }) {
return { const data = {
provider, provider,
url, url,
title: title || null, title: title || null,
@@ -66,6 +66,8 @@ function buildData({ provider, url, title, format, text, costUsd, responseMs, up
usage: { fetch_cost_usd: costUsd ?? null }, usage: { fetch_cost_usd: costUsd ?? null },
metrics: { response_time_ms: responseMs, upstream_latency_ms: upstreamMs } metrics: { response_time_ms: responseMs, upstream_latency_ms: upstreamMs }
}; };
if (Array.isArray(links)) data.links = links;
return data;
} }
async function readJsonOrText(res) { async function readJsonOrText(res) {
@@ -115,6 +117,18 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
if (provider === "exa") { if (provider === "exa") {
return await runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt }); return await runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt });
} }
if (provider === "ollama") {
return await runOllama({
url,
fmt,
timeoutMs,
apiKey,
maxCharacters,
costPerQuery,
startedAt,
baseUrl: providerConfig?.baseUrl,
});
}
return { success: false, status: 400, error: `Unsupported provider: ${provider}` }; return { success: false, status: 400, error: `Unsupported provider: ${provider}` };
} catch (err) { } catch (err) {
log?.("fetch handler error:", err?.message || err); log?.("fetch handler error:", err?.message || err);
@@ -241,3 +255,56 @@ async function runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery
}) })
}; };
} }
async function runOllama({
url,
fmt,
timeoutMs,
apiKey,
maxCharacters,
costPerQuery,
startedAt,
baseUrl,
}) {
const upstreamStart = Date.now();
const r = await tryFetch(baseUrl, {
method: "POST",
headers: {
"content-type": "application/json",
...(apiKey ? { authorization: `Bearer ${apiKey}` } : {})
},
body: JSON.stringify({ url })
}, timeoutMs);
if (!r.ok) {
return { success: false, status: r.timeout ? 504 : 502, error: r.error };
}
const upstreamMs = Date.now() - upstreamStart;
const { json, text: responseText } = await readJsonOrText(r.res);
if (!r.res.ok) {
const error = json?.error
|| json?.message
|| responseText?.slice(0, 500)
|| `Ollama error: ${r.res.status}`;
return { success: false, status: r.res.status, error };
}
if (!json || typeof json.content !== "string") {
return { success: false, status: 502, error: "Ollama returned an empty or invalid web fetch response" };
}
const text = truncate(json.content, maxCharacters);
return {
success: true,
data: buildData({
provider: "ollama",
url,
title: json.title || null,
format: fmt,
text,
links: json.links,
costUsd: costPerQuery,
responseMs: Date.now() - startedAt,
upstreamMs
})
};
}

View File

@@ -96,7 +96,7 @@ export async function handleImageGenerationCore({
let requestBody; let requestBody;
try { try {
url = adapter.buildUrl(model, credentials); url = adapter.buildUrl(model, credentials, body);
requestBody = await adapter.buildBody(model, body); requestBody = await adapter.buildBody(model, body);
headers = adapter.buildHeaders(credentials, requestBody, model, body); headers = adapter.buildHeaders(credentials, requestBody, model, body);
} catch (error) { } catch (error) {
@@ -140,7 +140,7 @@ export async function handleImageGenerationCore({
try { try {
const retryBody = await adapter.buildBody(model, body); const retryBody = await adapter.buildBody(model, body);
const retryHeaders = adapter.buildHeaders(credentials, retryBody, model, body); const retryHeaders = adapter.buildHeaders(credentials, retryBody, model, body);
const retryUrl = adapter.buildUrl(model, credentials); const retryUrl = adapter.buildUrl(model, credentials, body);
providerResponse = await fetch(retryUrl, { providerResponse = await fetch(retryUrl, {
method: "POST", method: "POST",
headers: retryHeaders, headers: retryHeaders,

View File

@@ -1,6 +1,6 @@
// Antigravity image adapter - delegates to the executor for correct request // Antigravity image adapter - delegates to the executor for correct request
// envelope (project, model, requestType, sessionId) and auth headers. // envelope (project, model, requestType, sessionId) and auth headers.
import { nowSec } from "./_base.js"; import { nowSec, sizeToAspectRatio } from "./_base.js";
import { getExecutor } from "../../executors/index.js"; import { getExecutor } from "../../executors/index.js";
// Convert image input (data URI or raw base64) to Gemini inlineData part // Convert image input (data URI or raw base64) to Gemini inlineData part
@@ -31,6 +31,19 @@ export default {
const executor = getExecutor("antigravity"); const executor = getExecutor("antigravity");
if (!executor) throw new Error("Antigravity executor not found"); if (!executor) throw new Error("Antigravity executor not found");
// Ensure we use an image model for image generation
const isImageModel = (m) => /image|imagen|image-generation/i.test(m || "");
let targetModel = isImageModel(model) ? model : "gemini-3.1-flash-image";
// If body.size is provided, resolve aspect ratio and append to model
if (body.size && typeof body.size === "string") {
const ratio = sizeToAspectRatio(body.size);
const suffix = ratio.replace(":", "x");
if (!targetModel.includes(suffix)) {
targetModel = `${targetModel}-${suffix}`;
}
}
// Build parts: text prompt + optional input image for editing // Build parts: text prompt + optional input image for editing
const parts = [{ text: body.prompt }]; const parts = [{ text: body.prompt }];
const imageInput = body.image || (Array.isArray(body.images) && body.images[0]); const imageInput = body.image || (Array.isArray(body.images) && body.images[0]);
@@ -44,7 +57,7 @@ export default {
}; };
const result = await executor.execute({ const result = await executor.execute({
model, model: targetModel,
body: chatBody, body: chatBody,
stream: false, stream: false,
credentials, credentials,

View File

@@ -12,6 +12,7 @@ import blackForestLabs from "./blackForestLabs.js";
import runwayml from "./runwayml.js"; import runwayml from "./runwayml.js";
import cloudflareAi from "./cloudflareAi.js"; import cloudflareAi from "./cloudflareAi.js";
import antigravity from "./antigravity.js"; import antigravity from "./antigravity.js";
import xai from "./xai.js";
const ADAPTERS = { const ADAPTERS = {
openai: createOpenAIAdapter("openai"), openai: createOpenAIAdapter("openai"),
@@ -19,7 +20,7 @@ const ADAPTERS = {
openrouter: createOpenAIAdapter("openrouter"), openrouter: createOpenAIAdapter("openrouter"),
recraft: createOpenAIAdapter("recraft"), recraft: createOpenAIAdapter("recraft"),
"vercel-ai-gateway": createOpenAIAdapter("vercel-ai-gateway"), "vercel-ai-gateway": createOpenAIAdapter("vercel-ai-gateway"),
xai: createOpenAIAdapter("xai"), xai,
gemini, gemini,
codex, codex,
sdwebui, sdwebui,

View File

@@ -0,0 +1,137 @@
// xAI Grok Imagine — text-to-image + single/multi image editing
// Docs:
// https://docs.x.ai/developers/model-capabilities/images/generation
// https://docs.x.ai/developers/model-capabilities/images/editing
// https://docs.x.ai/developers/model-capabilities/images/multi-image-editing
import { sizeToAspectRatio } from "./_base.js";
import { PROVIDER_MEDIA } from "../../providers/index.js";
const IMG_CFG = PROVIDER_MEDIA["xai"]?.imageConfig || {};
const GENERATIONS_URL = IMG_CFG.baseUrl || "https://api.x.ai/v1/images/generations";
const EDITS_URL = IMG_CFG.editsUrl || "https://api.x.ai/v1/images/edits";
const ASPECT_RATIOS = new Set([
"auto",
"1:1",
"16:9",
"9:16",
"4:3",
"3:2",
"2:3",
"9:19.5",
"20:9",
]);
function hasEditInput(body) {
if (!body || typeof body !== "object") return false;
if (body.image) return true;
return Array.isArray(body.images) && body.images.some(Boolean);
}
/** Normalize client image input → xAI image ref object */
function toXaiImageRef(input) {
if (!input) return null;
if (typeof input === "object") {
// Already xAI-shaped or partial
if (input.file_id) {
return {
type: input.type || "image_url",
file_id: input.file_id,
...(input.url ? { url: input.url } : {}),
};
}
if (input.url) {
return { type: input.type || "image_url", url: input.url };
}
return null;
}
if (typeof input !== "string") return null;
const trimmed = input.trim();
if (!trimmed) return null;
// Public URL or data URI
if (/^https?:\/\//i.test(trimmed) || /^data:image\//i.test(trimmed)) {
return { type: "image_url", url: trimmed };
}
// Raw base64 → data URI
return { type: "image_url", url: `data:image/png;base64,${trimmed}` };
}
function collectImageRefs(body) {
const refs = [];
if (Array.isArray(body.images)) {
for (const item of body.images) {
const ref = toXaiImageRef(item);
if (ref) refs.push(ref);
}
}
if (body.image) {
const ref = toXaiImageRef(body.image);
if (ref) refs.push(ref);
}
// xAI multi-edit supports up to 3 source images
return refs.slice(0, 3);
}
function resolveAspectRatio(body) {
if (typeof body.aspect_ratio === "string" && body.aspect_ratio.trim()) {
const ratio = body.aspect_ratio.trim();
if (ASPECT_RATIOS.has(ratio)) return ratio;
// Pass through unknown ratio strings (upstream will validate)
return ratio;
}
// OpenAI-style size → aspect ratio (skip auto)
if (body.size && body.size !== "auto") {
return sizeToAspectRatio(body.size);
}
return undefined;
}
function resolveResolution(body) {
if (typeof body.resolution !== "string") return undefined;
const value = body.resolution.trim().toLowerCase();
if (!value || value === "auto") return undefined;
return value; // "1k" | "2k"
}
export default {
buildUrl: (_model, _credentials, body) => (hasEditInput(body) ? EDITS_URL : GENERATIONS_URL),
buildHeaders: (creds) => {
const headers = { "Content-Type": "application/json", ...(IMG_CFG.headers || {}) };
const key = creds?.apiKey || creds?.accessToken;
if (key) headers["Authorization"] = `Bearer ${key}`;
return headers;
},
buildBody: (model, body) => {
const req = {
model,
prompt: body.prompt,
};
if (body.n != null) req.n = body.n;
if (body.response_format) req.response_format = body.response_format;
const aspectRatio = resolveAspectRatio(body);
if (aspectRatio) req.aspect_ratio = aspectRatio;
const resolution = resolveResolution(body);
if (resolution) req.resolution = resolution;
const refs = collectImageRefs(body);
if (refs.length === 1) {
req.image = refs[0];
} else if (refs.length > 1) {
req.images = refs;
}
return req;
},
// xAI already returns OpenAI-compatible { created, data: [{ url | b64_json }] }
normalize: (responseBody) => responseBody,
};

View File

@@ -347,6 +347,81 @@ function buildSearxngRequest(config, params) {
}; };
} }
function buildXquikRequest(config, params) {
const apiKey = params.token;
if (!apiKey) throw new Error("Xquik requires an API key");
const queryType = getProviderSetting(params, "queryType");
if (queryType && !["Latest", "Top"].includes(queryType)) {
throw new Error("Xquik queryType must be Latest or Top");
}
const qp = new URLSearchParams({
q: params.query,
limit: String(params.maxResults),
});
const cursor = getProviderSetting(params, "cursor");
if (cursor) qp.set("cursor", cursor);
if (queryType) qp.set("queryType", queryType);
if (params.language) qp.set("language", params.language);
return {
url: `${resolveBaseUrl(config, params)}?${qp}`,
init: {
method: "GET",
headers: { Accept: "application/json", "x-api-key": apiKey },
},
};
}
// ── Ollama Cloud web_search ──────────────────────────────────────────────
// POST https://ollama.com/api/web_search { query, max_results }
// Response: { results: [{ title, url, content, published_at? }] }
function buildOllamaSearchRequest(config, params) {
const body = { query: params.query, max_results: params.maxResults };
if (params.country) body.country = params.country;
if (params.language) body.language = params.language;
return {
url: resolveBaseUrl(config, params),
init: {
method: "POST",
headers: {
"Content-Type": "application/json",
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
},
body: JSON.stringify(body),
},
};
}
// ── GLM Coding plan MCP web_search_prime ──────────────────────────────────
// POST https://api.z.ai/api/mcp/web_search_prime/mcp
// JSON-RPC envelope: { jsonrpc, id, method: "tools/call",
// params: { name: "web_search_prime", arguments: { search_query, count } } }
// Response: { result: { content: [{ type: "text", text: "<json>" }] } }
function buildGlmSearchRequest(config, params) {
const body = {
jsonrpc: "2.0",
id: `9r-${Date.now()}`,
method: "tools/call",
params: {
name: "web_search_prime",
arguments: { search_query: params.query, count: params.maxResults },
},
};
return {
url: resolveBaseUrl(config, params),
init: {
method: "POST",
headers: {
"Content-Type": "application/json",
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
},
body: JSON.stringify(body),
},
};
}
// ── Dispatcher ────────────────────────────────────────────────────────── // ── Dispatcher ──────────────────────────────────────────────────────────
const BUILDERS = { const BUILDERS = {
@@ -360,6 +435,9 @@ const BUILDERS = {
"searchapi": buildSearchApiRequest, "searchapi": buildSearchApiRequest,
"youcom": buildYouComRequest, "youcom": buildYouComRequest,
"searxng": buildSearxngRequest, "searxng": buildSearxngRequest,
"xquik": buildXquikRequest,
"ollama-search": buildOllamaSearchRequest,
"glm": buildGlmSearchRequest,
}; };
/** /**

View File

@@ -1,8 +1,10 @@
/** /**
* Wrap chat-completions endpoints (with built-in web search) into the unified * Wrap chat-completions endpoints (with built-in web search) into the unified
* /v1/search response format. Supports gemini, openai, xai, kimi, minimax, perplexity. * /v1/search response format. Supports gemini, antigravity, openai, xai, kimi,
* minimax, perplexity.
*/ */
import { PROVIDER_MEDIA } from "../../providers/index.js"; import { PROVIDER_MEDIA } from "../../providers/index.js";
import { ANTIGRAVITY_IDE_USER_AGENT } from "../../providers/shared.js";
// Default search model + endpoint derive from registry searchViaChat (single source) // Default search model + endpoint derive from registry searchViaChat (single source)
const searchModel = (id) => PROVIDER_MEDIA[id]?.searchViaChat?.defaultModel; const searchModel = (id) => PROVIDER_MEDIA[id]?.searchViaChat?.defaultModel;
@@ -28,13 +30,37 @@ function toResult(c, index, provider, retrievedAt) {
score: null, score: null,
published_at: null, published_at: null,
favicon_url: null, favicon_url: null,
content: null, content: c.content || null,
metadata: {}, metadata: {},
citation: { provider, retrieved_at: retrievedAt, rank: index + 1 }, citation: { provider, retrieved_at: retrievedAt, rank: index + 1 },
provider_raw: null provider_raw: null
}; };
} }
// Antigravity search request envelope (mirrors the IDE client)
const AG_CLIENT_NAME = "antigravity";
const AG_SEARCH_GENERATION_CONFIG = { temperature: 1.0, maxOutputTokens: 8192 };
const AG_CONTEXT_BEFORE = 150;
const AG_CONTEXT_AFTER = 250;
/** Widen a grounded segment to its surrounding sentence(s) in the answer text. */
function expandSegment(text, segment) {
const { startIndex, endIndex } = segment || {};
if (!text || !Number.isInteger(startIndex) || !Number.isInteger(endIndex)) return "";
const start = Math.max(0, startIndex - AG_CONTEXT_BEFORE);
const end = Math.min(text.length, endIndex + AG_CONTEXT_AFTER);
let out = text.slice(start, end).trim();
// Drop the partial words the window cut off at either edge
if (start > 0) out = `...${out.replace(/^\S+/, "")}`;
if (end < text.length) out = `${out.replace(/\S+$/, "")}...`;
return out.trim();
}
/** Join deduped grounding pieces, skipping empties. */
function joinPieces(set, sep) {
return [...(set || [])].filter(Boolean).join(sep).trim();
}
/** Coerce a citation that might be a raw URL string or an object. */ /** Coerce a citation that might be a raw URL string or an object. */
function normalizeCitation(c) { function normalizeCitation(c) {
if (!c) return null; if (!c) return null;
@@ -46,6 +72,8 @@ function normalizeCitation(c) {
/** /**
* Provider-specific configuration map. All providers must implement: * Provider-specific configuration map. All providers must implement:
* { endpoint, defaultModel, buildBody, buildHeaders, extractAnswer } * { endpoint, defaultModel, buildBody, buildHeaders, extractAnswer }
* Optional: requireCredentials(credentials) → error string when a provider needs
* more than a token (returns null when satisfied).
*/ */
const CHAT_SEARCH_CONFIG = { const CHAT_SEARCH_CONFIG = {
gemini: { gemini: {
@@ -73,6 +101,71 @@ const CHAT_SEARCH_CONFIG = {
} }
}, },
antigravity: {
endpoint: () => searchEndpoint("antigravity"),
// Upstream 403s on a missing or fabricated project — surface the real cause
requireCredentials: (credentials) =>
credentials?.projectId ? null : "Antigravity account has no projectId — reconnect the account",
buildBody: (query, model, credentials) => ({
project: credentials.projectId,
model,
userAgent: AG_CLIENT_NAME,
requestType: "search",
request: {
contents: [{ role: "user", parts: [{ text: query }] }],
tools: [{ googleSearch: {} }],
generationConfig: AG_SEARCH_GENERATION_CONFIG
}
}),
buildHeaders: (token) => ({
"Content-Type": "application/json",
Authorization: `Bearer ${token}`,
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT
}),
extractAnswer: (data) => {
// Antigravity wraps the Gemini payload in { response: {...} }
const response = data?.response || data;
const candidate = response?.candidates?.[0];
const parts = candidate?.content?.parts || [];
const text = parts.map((p) => p?.text || "").filter(Boolean).join("");
const grounding = candidate?.groundingMetadata || {};
const chunks = grounding.groundingChunks || [];
const supports = grounding.groundingSupports || [];
// Upstream repeats the same source across chunks — key by URL so it stays one citation.
// Map, not a plain object: both the index and the URL come from upstream.
const sources = new Map();
const byIndex = chunks.map((ch) => {
const web = ch?.web;
const url = web?.uri || web?.url || "";
if (!url) return null;
if (!sources.has(url)) sources.set(url, { title: web.title || "", snippets: new Set(), contexts: new Set() });
return sources.get(url);
});
// Each support ties a sentence of the answer back to the chunks that grounded it
for (const s of supports) {
const segment = s?.segment;
const grounded = segment?.text || "";
const expanded = expandSegment(text, segment) || grounded;
for (const idx of s?.groundingChunkIndices || []) {
const source = Number.isInteger(idx) ? byIndex[idx] : null;
if (!source) continue;
if (grounded) source.snippets.add(grounded);
if (expanded) source.contexts.add(expanded);
}
}
const citations = [...sources].map(([url, src]) => {
const snippet = joinPieces(src.snippets, " | ") || src.title;
return { url, title: src.title, snippet, content: joinPieces(src.contexts, "\n\n") || snippet };
});
const tokens = response?.usageMetadata?.totalTokenCount || 0;
return { text, citations, tokens };
}
},
openai: { openai: {
endpoint: () => searchEndpoint("openai"), endpoint: () => searchEndpoint("openai"),
buildBody: (query, model) => { buildBody: (query, model) => {
@@ -366,13 +459,18 @@ export async function handleChatSearch({
}; };
} }
const credentialError = cfg.requireCredentials?.(credentials);
if (credentialError) {
return { success: false, status: 401, error: credentialError };
}
const limit = const limit =
Number.isFinite(maxResults) && maxResults > 0 Number.isFinite(maxResults) && maxResults > 0
? Math.floor(maxResults) ? Math.floor(maxResults)
: DEFAULT_MAX_RESULTS; : DEFAULT_MAX_RESULTS;
const useModel = model || searchModel(provider); const useModel = model || searchModel(provider);
const url = cfg.endpoint(useModel); const url = cfg.endpoint(useModel);
const body = cfg.buildBody(query, useModel); const body = cfg.buildBody(query, useModel, credentials);
const headers = cfg.buildHeaders(token); const headers = cfg.buildHeaders(token);
const controller = new AbortController(); const controller = new AbortController();

View File

@@ -10,6 +10,7 @@
import { buildSearchRequest } from "./callers.js"; import { buildSearchRequest } from "./callers.js";
import { normalizeSearchResponse } from "./normalizers.js"; import { normalizeSearchResponse } from "./normalizers.js";
import { handleChatSearch } from "./chatSearch.js"; import { handleChatSearch } from "./chatSearch.js";
import { fetchPublic } from "../../../src/shared/utils/ssrfGuard.js";
const GLOBAL_TIMEOUT_MS = 15000; const GLOBAL_TIMEOUT_MS = 15000;
const NON_RETRIABLE = new Set([400, 401, 403, 404]); const NON_RETRIABLE = new Set([400, 401, 403, 404]);
@@ -100,7 +101,7 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
log?.info?.("SEARCH", `${provider.id} | "${params.query.slice(0, 80)}" | type=${params.searchType}`); log?.info?.("SEARCH", `${provider.id} | "${params.query.slice(0, 80)}" | type=${params.searchType}`);
try { try {
const resp = await fetch(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal }); const resp = await fetchPublic(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
clearTimeout(timer); clearTimeout(timer);
if (!resp.ok) { if (!resp.ok) {
const errText = await resp.text().catch(() => ""); const errText = await resp.text().catch(() => "");
@@ -111,6 +112,13 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType); const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType);
const results = normalized.results.slice(0, params.maxResults); const results = normalized.results.slice(0, params.maxResults);
const duration = Date.now() - startTime; const duration = Date.now() - startTime;
const usage = {
queries_used: 1,
search_cost_usd: providerConfig.costPerQuery ?? null,
};
if (Number.isFinite(providerConfig.creditsPerResult)) {
usage.provider_credits_used = results.length * providerConfig.creditsPerResult;
}
return { return {
success: true, success: true,
@@ -119,7 +127,8 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
query: params.query, query: params.query,
results, results,
answer: null, answer: null,
usage: { queries_used: 1, search_cost_usd: providerConfig.costPerQuery || 0 }, usage,
...(normalized.pagination ? { pagination: normalized.pagination } : {}),
metrics: { response_time_ms: duration, upstream_latency_ms: duration, total_results_available: normalized.totalResults }, metrics: { response_time_ms: duration, upstream_latency_ms: duration, total_results_available: normalized.totalResults },
errors: [] errors: []
} }

View File

@@ -199,6 +199,89 @@ function normalizeSearxng(data, _query, _searchType) {
return { results, totalResults: results.length }; return { results, totalResults: results.length };
} }
function normalizeXquik(data, _query, _searchType) {
const now = new Date().toISOString();
const items = Array.isArray(data.tweets) ? data.tweets : [];
const results = items.map((item, idx) => {
const username = typeof item?.author?.username === "string" ? item.author.username : "";
const authorName = typeof item?.author?.name === "string" ? item.author.name : "";
const tweetId = typeof item?.id === "string" ? item.id : String(item?.id || "");
const url = username && tweetId
? `https://x.com/${encodeURIComponent(username)}/status/${encodeURIComponent(tweetId)}`
: tweetId
? `https://x.com/i/web/status/${encodeURIComponent(tweetId)}`
: "";
const author = username ? `@${username}` : authorName || null;
const title = author ? `${author} on X` : "X post";
const imageUrl = Array.isArray(item?.media)
? item.media.find((media) => typeof media?.mediaUrl === "string")?.mediaUrl
: null;
return makeResult("xquik", {
title,
url,
snippet: typeof item?.text === "string" ? item.text : "",
published_at: typeof item?.createdAt === "string" ? item.createdAt : null,
author,
image_url: imageUrl || null,
source_type: "x_post",
full_text: typeof item?.text === "string" ? item.text : undefined,
text_format: "text",
}, idx, now);
});
const nextCursor = typeof data.next_cursor === "string" && data.next_cursor ? data.next_cursor : null;
return {
results,
totalResults: null,
pagination: {
has_more: data.has_next_page === true,
next_cursor: nextCursor,
},
};
}
function normalizeOllamaSearch(data, _query, _searchType) {
const now = new Date().toISOString();
const items = Array.isArray(data?.results) ? data.results : (Array.isArray(data) ? data : []);
const results = items.map((item, idx) =>
makeResult("ollama-search", {
title: item.title,
url: item.url,
snippet: item.content || item.snippet || "",
full_text: item.content,
text_format: "text",
published_at: item.published_at || null,
source_type: item.source || null,
}, idx, now)
);
return { results, totalResults: results.length };
}
function normalizeGlmSearch(data, _query, _searchType) {
const now = new Date().toISOString();
// MCP envelope: { result: { content: [{ type: "text", text: "<json>" }] } }
let payload = data;
const textContent = data?.result?.content?.[0]?.text;
if (typeof textContent === "string") {
try { payload = JSON.parse(textContent); } catch { payload = {}; }
}
const items = Array.isArray(payload?.results) ? payload.results
: Array.isArray(payload?.news) ? payload.news
: Array.isArray(payload) ? payload
: [];
const results = items.map((item, idx) =>
makeResult("glm", {
title: item.title,
url: item.link || item.url,
snippet: item.content || "",
published_at: item.publish_date || item.published_at || null,
favicon_url: item.icon || null,
source_type: item.media || null,
}, idx, now)
);
return { results, totalResults: results.length };
}
const NORMALIZERS = { const NORMALIZERS = {
"serper": normalizeSerper, "serper": normalizeSerper,
"brave-search": normalizeBrave, "brave-search": normalizeBrave,
@@ -210,11 +293,14 @@ const NORMALIZERS = {
"searchapi": normalizeSearchApi, "searchapi": normalizeSearchApi,
"youcom": normalizeYouCom, "youcom": normalizeYouCom,
"searxng": normalizeSearxng, "searxng": normalizeSearxng,
"xquik": normalizeXquik,
"ollama-search": normalizeOllamaSearch,
"glm": normalizeGlmSearch,
}; };
/** /**
* Dispatch to the appropriate normalizer based on providerId. * Dispatch to the appropriate normalizer based on providerId.
* @returns {{results: Array, totalResults: number|null}} * @returns {{results: Array, totalResults: number|null, pagination?: object}}
*/ */
export function normalizeSearchResponse(providerId, data, query, searchType) { export function normalizeSearchResponse(providerId, data, query, searchType) {
const fn = NORMALIZERS[providerId]; const fn = NORMALIZERS[providerId];

View File

@@ -6,6 +6,16 @@
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic // 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
// 4. DEFAULT_CAPABILITIES — safe floor (always returned) // 4. DEFAULT_CAPABILITIES — safe floor (always returned)
// //
// Two extra layers then refine the result, and neither can override the hand
// written tables above (steps 1-2 short-circuit before they are consulted):
// • the synced catalog — modalities keyed by model, limits keyed by provider
// + model, refreshed from models.dev in the background. It reads a file, so
// the server installs it via setCatalogSource(); this module stays free of
// node:fs because the dashboard bundles it into the browser too.
// • visionPatterns.js — name-based vision detection, last resort so a model
// nobody has catalogued yet still accepts images.
// Both only ever turn a capability ON.
//
// ── HOW TO ADD / UPDATE A MODEL ────────────────────────────────────── // ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+ // Authoritative data source: https://models.dev/api.json (145 providers, 4000+
// models, MIT). Each model exposes the exact fields we map below: // models, MIT). Each model exposes the exact fields we map below:
@@ -23,6 +33,7 @@
// 2.0+, Grok, Perplexity). Verify with: curl -s https://models.dev/api.json // 2.0+, Grok, Perplexity). Verify with: curl -s https://models.dev/api.json
import { matchPattern } from "./pricing.js"; import { matchPattern } from "./pricing.js";
import { looksLikeVisionModel } from "./visionPatterns.js";
/** /**
* Safe floor — every resolved result is merged over this so consumers * Safe floor — every resolved result is merged over this so consumers
@@ -46,6 +57,7 @@ export const DEFAULT_CAPABILITIES = {
thinkingFormat: null, thinkingFormat: null,
thinkingCanDisable: true, // false → model cannot turn thinking off (clamp to min instead of disable) thinkingCanDisable: true, // false → model cannot turn thinking off (clamp to min instead of disable)
thinkingRange: null, // { min, max } for budget formats; null = no clamp thinkingRange: null, // { min, max } for budget formats; null = no clamp
thinkingEffortSupported: false, // zai format only: model accepts a reasoning_effort level (GLM-5.2+; older GLM ignores it)
// limits (tokens) // limits (tokens)
contextWindow: 200000, contextWindow: 200000,
maxOutput: 64000, maxOutput: 64000,
@@ -71,7 +83,8 @@ export function capabilitiesFromServiceKind(kind) {
* otherwise mis-match. Only declare deltas vs DEFAULT. * otherwise mis-match. Only declare deltas vs DEFAULT.
*/ */
export const MODEL_CAPABILITIES = { export const MODEL_CAPABILITIES = {
// Claude Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern) // Claude Fable 5.1, Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
"claude-fable-5-1": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
@@ -94,8 +107,14 @@ export const MODEL_CAPABILITIES = {
// Gemini image-gen / OpenAI image / xai image variants // Gemini image-gen / OpenAI image / xai image variants
"gpt-image-1": { imageOutput: true, tools: false }, "gpt-image-1": { imageOutput: true, tools: false },
// GLM vision variant (text GLM has no vision) // GLM vision variants (text GLM has no vision) — 5.3-Flash and 5V-Turbo are
"glm-4.6v": { vision: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000 }, // natively multimodal per z.ai, and 5.3-Flash carries the full 1M window.
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "zai", contextWindow: 1000000, maxOutput: 131072 },
"glm-4.6v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000, maxOutput: 32768 },
"glm-4.5v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 64000, maxOutput: 16384 },
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases // Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 }, "vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
@@ -108,6 +127,10 @@ export const MODEL_CAPABILITIES = {
"kimi-for-coding-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 }, "kimi-for-coding-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
"kimi-k2.7-code": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 }, "kimi-k2.7-code": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
"kimi-k2.7-code-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 }, "kimi-k2.7-code-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
// OpenCode Free Muse Spark — multimodal (text+image per models.dev meta/muse-spark)
// via OpenAI Responses input_image; reasoning supports up to xhigh.
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
}; };
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
@@ -131,6 +154,7 @@ export const PROVIDER_CAPABILITIES = {
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 }, "deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
}, },
"codex": { "codex": {
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS, "gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
@@ -155,24 +179,69 @@ export const PROVIDER_CAPABILITIES = {
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model // CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision= // config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort // supportsImages). Every model reasons via OpenAI-style reasoning_effort
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking // (see registry thinkingFormat). For thinkingCanDisable use the server's
// off → thinkingCanDisable:false (clamped to minimal instead of disabled). // reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
// below; it is NOT the inverse of onlyReasoning.
"codebuddy-cn": { "codebuddy-cn": {
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 }, "glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, "glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 }, "glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, // maxOutput 64000 per both the plugin-baked fallback and the live server
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 }, // table (the old 38000 had no source and truncated output).
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 }, "glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 }, "minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, "kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, "kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 }, // Per-model values mirror the server's product-config payload (the plugin
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, // fetches it from copilot.tencent.com; the `models[]` entries carry
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, // maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, // maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 }, // plugin-baked fallback disagree, the server table wins.
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
// their thinking is switchable; the hy* models are forced always-on.
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
// (200K) without this map. contextWindow follows the real model family's
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
// max_output_tokens arrives as 0 for every model, so outputs are
// best-guess from the real model family. Vision tags below follow the
// upstream is_vl flag per explicit request, even though the executor
// currently sends image_urls:null (image pass-through over the agent_chat
// SSE protocol is unverified). reasoning:true on all of them — every model can
// reason; the upstream is_reasoning flag only drives model_config selection.
// thinkingFormat keeps the true-model family for documentation/UI, but
// thinkingCanDisable:false everywhere: the executor only forwards
// messages/tools/max_tokens, and thinking is fixed upstream via
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
// must never be offered as an option.
"qoder": {
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
}, },
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output). // Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
"poolside": { "poolside": {
@@ -205,6 +274,7 @@ export const PATTERN_CAPABILITIES = [
// ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─ // ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─
{ pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } }, { pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } },
{ pattern: "*gemini-3.8*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
{ pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } }, { pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
{ pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } }, { pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } },
{ pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } }, { pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
@@ -214,6 +284,9 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } }, { pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } }, { pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
// ── OpenAI GPT-5.x (vision + thinking + web search) ────────────── // ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } }, { pattern: "*gpt-5*image*", caps: { imageOutput: true } },
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, { pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
@@ -234,6 +307,8 @@ export const PATTERN_CAPABILITIES = [
// ── Grok (vision + Live Search) ────────────────────────────────── // ── Grok (vision + Live Search) ──────────────────────────────────
{ pattern: "*grok*image*", caps: { imageOutput: true } }, { pattern: "*grok*image*", caps: { imageOutput: true } },
{ pattern: "*grok-code*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 256000 } }, { pattern: "*grok-code*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 256000 } },
// Grok 4.6: 500k context, no text output limit (docs.x.ai/developers/grok-4-6)
{ pattern: "*grok-4.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 500000 } },
// Grok 4.5 (Grok CLI / Grok Build): 500k context per cli-chat-proxy /v1/models // Grok 4.5 (Grok CLI / Grok Build): 500k context per cli-chat-proxy /v1/models
{ pattern: "*grok-4.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 64000 } }, { pattern: "*grok-4.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 64000 } },
{ pattern: "*grok-4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } }, { pattern: "*grok-4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
@@ -261,6 +336,10 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*kimi*", caps: { reasoning: true, thinkingFormat: "kimi", contextWindow: 262144 } }, { pattern: "*kimi*", caps: { reasoning: true, thinkingFormat: "kimi", contextWindow: 262144 } },
// ── GLM / Z.ai (thinking.enabled; disable via enable_thinking:false) ─ // ── GLM / Z.ai (thinking.enabled; disable via enable_thinking:false) ─
// reasoning_effort is only read by z.ai from GLM-5.2 onward (docs.z.ai/guides/capabilities/thinking) —
// older GLM (4.x, 5.0, 5.1, 5-turbo, 5v-turbo) ignore it, so gate it per exact version, not the "*glm-5*" catch-all.
{ pattern: "*glm-5.3*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-5.2*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-5*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } }, { pattern: "*glm-5*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-4.7*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } }, { pattern: "*glm-4.7*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
{ pattern: "*glm-4*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } }, { pattern: "*glm-4*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
@@ -309,6 +388,9 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*laguna-s-2.1*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 } }, { pattern: "*laguna-s-2.1*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 } },
{ pattern: "*laguna*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 } }, { pattern: "*laguna*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 } },
// ── OpenCode Free Muse Spark (multimodal text+image; OpenAI Responses reasoning supports up to xhigh) ─
{ pattern: "*muse*spark*", caps: { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 } },
// ── Others ─────────────────────────────────────────────────────── // ── Others ───────────────────────────────────────────────────────
{ pattern: "*hunyuan*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } }, { pattern: "*hunyuan*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
{ pattern: "hy3*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } }, { pattern: "hy3*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
@@ -317,6 +399,11 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*ling-*", caps: { reasoning: true, contextWindow: 128000 } }, { pattern: "*ling-*", caps: { reasoning: true, contextWindow: 128000 } },
]; ];
// OpenRouter-style gateways validate modalities upstream — a text-only model
// sent an image gets a clear upstream error instead of silent corruption. So for
// unknown models on these providers, trust vision instead of stripping images.
const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
/** /**
* Resolve capabilities for a model using the 4-step fallback chain, * Resolve capabilities for a model using the 4-step fallback chain,
* merged over DEFAULT_CAPABILITIES so the result is always complete. * merged over DEFAULT_CAPABILITIES so the result is always complete.
@@ -325,6 +412,46 @@ export const PATTERN_CAPABILITIES = [
* @param {string} model * @param {string} model
* @returns {object} full capabilities object * @returns {object} full capabilities object
*/ */
const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
// Catalog lookups, installed by the server at startup. Left as no-ops in the
// browser bundle, where there is no file to read.
let catalogSource = null;
/**
* Install the synced catalog reader (server only).
* @param {{ getModalities: Function, getLimits: Function } | null} source
*/
export function setCatalogSource(source) {
catalogSource = source;
}
// Apply the synced catalog + name heuristic on top of a table-resolved result.
// Strictly additive: a capability already true stays true, and a false one only
// flips when an outside source positively declares support.
function refine(base, provider, model) {
const result = { ...DEFAULT_CAPABILITIES, ...base };
if (catalogSource) {
const modalities = catalogSource.getModalities(model);
if (modalities) {
for (const key of MODALITY_KEYS) {
if (modalities[key] === true) result[key] = true;
}
}
const limits = catalogSource.getLimits(provider, model);
if (limits) {
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
}
}
if (!result.vision && looksLikeVisionModel(model)) result.vision = true;
return result;
}
export function getCapabilitiesForModel(provider, model) { export function getCapabilitiesForModel(provider, model) {
if (!model) return { ...DEFAULT_CAPABILITIES }; if (!model) return { ...DEFAULT_CAPABILITIES };
@@ -342,13 +469,13 @@ export function getCapabilitiesForModel(provider, model) {
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] }; if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] }; if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
// 3. Pattern match (first match wins) // 3. Pattern match (first match wins), refined by catalog + name heuristic
for (const { pattern, caps } of PATTERN_CAPABILITIES) { for (const { pattern, caps } of PATTERN_CAPABILITIES) {
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) { if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
return { ...DEFAULT_CAPABILITIES, ...caps }; return refine(caps, provider, model);
} }
} }
// 4. Floor // 4. Floor
return { ...DEFAULT_CAPABILITIES }; return refine(null, provider, model);
} }

View File

@@ -0,0 +1,72 @@
// Read side of the model catalog synced from models.dev.
//
// The file is the source of truth; the only thing held in memory is a parsed
// copy dropped as soon as the file's mtime changes. getCapabilitiesForModel is
// synchronous and runs per request, so the hot path is one stat (~1us) and the
// parse (~0.1ms on a ~18KB file) only reruns after a sync.
import fs from "node:fs";
import path from "node:path";
import { DATA_DIR } from "@/lib/dataDir.js";
export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
// Trimmed upstream catalog, read by the add-models skill (not by the router).
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
const EMPTY = { models: {}, providers: {} };
let cache = EMPTY;
let cachedMtime = -1;
// "zai-org/GLM-4.6V:free" -> "glm-4.6v"
function baseId(model) {
if (!model) return "";
const withoutVendor = model.includes("/") ? model.split("/").pop() : model;
return withoutVendor.toLowerCase().split(":")[0];
}
function load() {
let mtime;
try {
mtime = fs.statSync(CATALOG_FILE).mtimeMs;
} catch {
cache = EMPTY;
cachedMtime = -1;
return cache;
}
if (mtime === cachedMtime) return cache;
cachedMtime = mtime;
try {
const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8"));
cache = { models: parsed?.models || {}, providers: parsed?.providers || {} };
} catch {
cache = EMPTY;
}
return cache;
}
// Modality is a property of the model itself — any gateway serving it inherits
// the same image/video/pdf support, so this is keyed by model id alone.
export function getCatalogModalities(model) {
return load().models[baseId(model)] || null;
}
// Context and output limits are a property of the gateway, not the model: each
// one truncates differently, so these stay keyed by provider + model.
export function getCatalogLimits(provider, model) {
const byProvider = provider && load().providers[provider];
if (!byProvider) return null;
return byProvider[model] || byProvider[baseId(model)] || null;
}
// Force a re-read on the next lookup (called right after a sync writes the file).
export function invalidateCatalog() {
cachedMtime = -1;
}
// Hand the reader to capabilities.js. That module is bundled into the browser
// too, so it cannot import this file directly — the server pushes it in.
export async function installCatalogSource() {
const { setCatalogSource } = await import("./capabilities.js");
setCatalogSource({ getModalities: getCatalogModalities, getLimits: getCatalogLimits });
}

View File

@@ -18,3 +18,10 @@ export function withCodexReviewModels(models) {
]; ];
}); });
} }
export function isMuseSparkModel(modelId) {
if (!modelId || typeof modelId !== "string") return false;
const clean = modelId.replace(/\([^()]+\)\s*$/, "").trim();
const base = clean.includes("/") ? clean.split("/").pop() : clean;
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
}

View File

@@ -53,10 +53,15 @@ export const MODEL_PRICING = {
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 }, "gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 }, "gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, "gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 }, "o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 }, "o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
// === Gemini === // === Gemini ===
"gemini-3.8-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.8-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.8-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.8-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 }, "gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 }, "gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
"gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 }, "gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
@@ -260,6 +265,7 @@ export const PROVIDER_PRICING = {
"z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 }, "z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 },
"z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 }, "z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 },
"z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 }, "z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 },
"z-ai/glm-5.3-free": { input: 0, output: 0, cached: 0, reasoning: 0 },
}, },
}; };

View File

@@ -17,7 +17,7 @@ export default {
deprecationNotice: "RISK_NOTICE", deprecationNotice: "RISK_NOTICE",
}, },
category: "oauth", category: "oauth",
serviceKinds: ["llm", "image"], serviceKinds: ["llm", "image", "webSearch"],
transport: { transport: {
baseUrls: [ANTIGRAVITY_IDE_BASE_URL], baseUrls: [ANTIGRAVITY_IDE_BASE_URL],
format: "antigravity", format: "antigravity",
@@ -36,8 +36,7 @@ export default {
}, },
}, },
usage: { usage: {
// Discovery (quota/project) on PROD; daily host rejects these. quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
quotaApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels",
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist", loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
tokenUrl: "https://oauth2.googleapis.com/token", tokenUrl: "https://oauth2.googleapis.com/token",
}, },
@@ -45,6 +44,10 @@ export default {
clientSecret: "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", clientSecret: "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",
}, },
models: [ models: [
{ id: "gemini-3.8-flash-high", name: "Gemini 3.8 Flash (High)", upstreamModelId: "gemini-3.8-flash-high(high)" },
{ id: "gemini-3.8-flash-medium", name: "Gemini 3.8 Flash (Medium)", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
{ id: "gemini-3.8-flash-low", name: "Gemini 3.8 Flash (Low)", upstreamModelId: "gemini-3.8-flash-low(low)" },
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
{ id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" }, { id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" },
{ id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" }, { id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" },
{ id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" }, { id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" },
@@ -82,6 +85,11 @@ export default {
loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT, loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT,
refreshLeadMs: 300000, refreshLeadMs: 300000,
}, },
searchViaChat: {
defaultModel: "gemini-2.5-flash",
endpoint: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:generateContent`,
freeTier: "Free — Google Search grounding through an Antigravity OAuth account.",
},
features: { features: {
usage: true, usage: true,
}, },

View File

@@ -1,4 +1,4 @@
import { CLAUDE_CLI_SPOOF_HEADERS } from "../shared.js"; import { CLAUDE_CLI_VERSION } from "../shared.js";
export default { export default {
id: "claude", id: "claude",
@@ -25,7 +25,7 @@ export default {
"Anthropic-Version": "2023-06-01", "Anthropic-Version": "2023-06-01",
"Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28", "Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28",
"Anthropic-Dangerous-Direct-Browser-Access": "true", "Anthropic-Dangerous-Direct-Browser-Access": "true",
"User-Agent": "claude-cli/2.1.92 (external, sdk-cli)", "User-Agent": `claude-cli/${CLAUDE_CLI_VERSION} (external, sdk-cli)`,
"X-App": "cli", "X-App": "cli",
"X-Stainless-Helper-Method": "stream", "X-Stainless-Helper-Method": "stream",
"X-Stainless-Retry-Count": "0", "X-Stainless-Retry-Count": "0",
@@ -58,6 +58,7 @@ export default {
}, },
models: [ models: [
{ id: "claude-opus-5", name: "Claude Opus 5" }, { id: "claude-opus-5", name: "Claude Opus 5" },
{ id: "claude-fable-5-1", name: "Claude Fable 5.1" },
{ id: "claude-fable-5", name: "Claude Fable 5" }, { id: "claude-fable-5", name: "Claude Fable 5" },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" }, { id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" }, { id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" },

View File

@@ -47,19 +47,26 @@ export default {
models: [ models: [
{ id: "glm-5.2", name: "GLM-5.2" }, { id: "glm-5.2", name: "GLM-5.2" },
{ id: "glm-5.1", name: "GLM-5.1" }, { id: "glm-5.1", name: "GLM-5.1" },
{ id: "glm-5.0", name: "GLM-5.0" },
{ id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" },
{ id: "glm-5v-turbo", name: "GLM-5v-Turbo" }, { id: "glm-5v-turbo", name: "GLM-5v-Turbo" },
{ id: "glm-4.7", name: "GLM-4.7" },
{ id: "minimax-m3", name: "MiniMax-M3" }, { id: "minimax-m3", name: "MiniMax-M3" },
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
{ id: "kimi-k2.7", name: "Kimi-K2.7-Code" }, { id: "kimi-k2.7", name: "Kimi-K2.7-Code" },
{ id: "kimi-k2.6", name: "Kimi-K2.6" }, { id: "kimi-k2.6", name: "Kimi-K2.6" },
{ id: "kimi-k2.5", name: "Kimi-K2.5" }, // Catalog mirrors the server's product-config payload (the plugin fetches
{ id: "hy3-preview", name: "Hy3 Preview" }, // it from copilot.tencent.com). Models the server no longer publishes are
// removed even when the chat endpoint still answers them — the published
// list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x
// (endpoint returns 11102 "model service info not found"), plus
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
// deepseek-v3-2-volc (absent from the server list, though still answering
// 200) and hy3-x (paid tier, not used here).
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
{ id: "hy3", name: "Hy3" },
{ id: "hy4-preview", name: "Hy4-Preview" },
{ id: "glm-5.3", name: "GLM-5.3" },
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
{ id: "kimi-k3-1", name: "Kimi-K3" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" }, { id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" }, { id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
], ],
oauth: { oauth: {
baseUrl: "https://copilot.tencent.com", baseUrl: "https://copilot.tencent.com",

View File

@@ -45,6 +45,7 @@ export default {
}, },
}, },
models: [ models: [
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" }, { id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
@@ -59,6 +60,9 @@ export default {
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" }, { id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" }, { id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" }, { id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },

View File

@@ -23,21 +23,55 @@ export default {
format: "commandcode", format: "commandcode",
forceStream: true, forceStream: true,
headers: { headers: {
"x-command-code-version": "0.25.7", "x-command-code-version": "0.45.0",
"x-cli-environment": "cli", "x-cli-environment": "cli",
"User-Agent": "cli",
}, },
// Quota/billing endpoints (same alpha API the official CLI /usage calls).
// whoami resolves orgId; credits+subscription+usage/summary then report the
// 5-hour/weekly windows, plan, and period credits. See services/usage/commandcode.js.
usage: {
baseUrl: "https://api.commandcode.ai",
whoamiUrl: "/alpha/whoami",
creditsUrl: "/alpha/billing/credits",
subscriptionsUrl: "/alpha/billing/subscriptions",
summaryUrl: "/alpha/usage/summary",
},
},
features: {
usage: true,
usageApikey: true,
}, },
models: [ models: [
{ id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" }, { id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" }, { id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "moonshotai/Kimi-K2.7-Code", name: "Kimi K2.7 Code" },
{ id: "moonshotai/Kimi-K2.7-Code-Highspeed", name: "Kimi K2.7 Code Highspeed" },
{ id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6" }, { id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6" },
{ id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5" }, { id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5" },
{ id: "zai-org/GLM-5.2", name: "GLM 5.2" },
{ id: "zai-org/GLM-5.2-Fast", name: "GLM 5.2 Fast" },
{ id: "zai-org/GLM-5.1", name: "GLM 5.1" }, { id: "zai-org/GLM-5.1", name: "GLM 5.1" },
{ id: "zai-org/GLM-5", name: "GLM 5" }, { id: "zai-org/GLM-5", name: "GLM 5" },
{ id: "MiniMaxAI/MiniMax-M3", name: "MiniMax M3" },
{ id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7" }, { id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7" },
{ id: "MiniMaxAI/MiniMax-M2.5", name: "MiniMax M2.5" }, { id: "MiniMaxAI/MiniMax-M2.5", name: "MiniMax M2.5" },
{ id: "xiaomi/mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
{ id: "xiaomi/mimo-v2.5", name: "MiMo V2.5" },
{ id: "Qwen/Qwen3.7-Max", name: "Qwen 3.7 Max" },
{ id: "Qwen/Qwen3.7-Plus", name: "Qwen 3.7 Plus" },
{ id: "Qwen/Qwen3.6-Max-Preview", name: "Qwen 3.6 Max Preview" }, { id: "Qwen/Qwen3.6-Max-Preview", name: "Qwen 3.6 Max Preview" },
{ id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" }, { id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" },
{ id: "stepfun/Step-3.7-Flash", name: "Step 3.7 Flash" },
{ id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" }, { id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" },
{ id: "tencent/Hy3", name: "Tencent Hy3" },
{ id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B A55B" },
{ id: "thinkingmachines/inkling", name: "Inkling" },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6" },
{ id: "claude-fable-5", name: "Claude Fable 5" },
{ id: "claude-opus-4-8", name: "Claude Opus 4.8" },
{ id: "claude-opus-4-7", name: "Claude Opus 4.7" },
{ id: "claude-haiku-4-5", name: "Claude Haiku 4.5" },
], ],
}; };

View File

@@ -45,6 +45,7 @@ export default {
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" }, { id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" }, { id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" }, { id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
{ id: "deepseek-reasoner", name: "DeepSeek V3.2 Reasoner" }, { id: "deepseek-reasoner", name: "DeepSeek V3.2 Reasoner" },
], ],

View File

@@ -36,6 +36,7 @@ export default {
}, },
}, },
models: [ models: [
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash" },
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" }, { id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" },
{ id: "gemini-3.6-flash", name: "Gemini 3.6 Flash" }, { id: "gemini-3.6-flash", name: "Gemini 3.6 Flash" },
{ id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" }, { id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },

View File

@@ -22,10 +22,13 @@ export default {
}, },
models: [ models: [
{ id: "glm-5.3", name: "GLM 5.3" }, { id: "glm-5.3", name: "GLM 5.3" },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" }, { id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" }, { id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" }, { id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM-4.7" }, { id: "glm-4.7", name: "GLM-4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
{ id: "glm-4.6", name: "GLM-4.6" }, { id: "glm-4.6", name: "GLM-4.6" },
{ id: "glm-4.5-air", name: "GLM-4.5-Air" }, { id: "glm-4.5-air", name: "GLM-4.5-Air" },
], ],

View File

@@ -46,12 +46,28 @@ export default {
], ],
models: [ models: [
{ id: "glm-5.3", name: "GLM 5.3" }, { id: "glm-5.3", name: "GLM 5.3" },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" }, { id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" }, { id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" }, { id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM 4.7" }, { id: "glm-4.7", name: "GLM 4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" }, { id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
], ],
serviceKinds: ["llm", "webSearch"],
// Coding plan bundles web search on the same API key as chat.
searchConfig: {
baseUrl: "https://api.z.ai/api/mcp/web_search_prime/mcp",
method: "POST",
authType: "apikey",
authHeader: "bearer",
costPerQuery: 0,
searchTypes: ["web"],
defaultMaxResults: 5,
maxMaxResults: 50,
timeoutMs: 10000,
cacheTTLMs: 300000,
},
features: { features: {
usage: true, usage: true,
usageApikey: true, usageApikey: true,

View File

@@ -17,6 +17,12 @@ export default {
transport: { transport: {
baseUrl: "https://api.groq.com/openai/v1/chat/completions", baseUrl: "https://api.groq.com/openai/v1/chat/completions",
validateUrl: "https://api.groq.com/openai/v1/models", validateUrl: "https://api.groq.com/openai/v1/models",
// No dedicated quota endpoint; rate-limit info rides on x-ratelimit-*
// response headers, always included. Reuse the models list (already
// used as validateUrl) so reading usage never costs tokens.
usage: {
url: "https://api.groq.com/openai/v1/models",
},
}, },
models: [ models: [
{ id: "llama-3.3-70b-versatile", name: "Llama 3.3 70B" }, { id: "llama-3.3-70b-versatile", name: "Llama 3.3 70B" },
@@ -34,4 +40,8 @@ export default {
authHeader: "bearer", authHeader: "bearer",
format: "openai", format: "openai",
}, },
features: {
usage: true,
usageApikey: true,
},
}; };

View File

@@ -66,6 +66,7 @@ import p63 from "./nebius.js";
import p64 from "./nvidia.js"; import p64 from "./nvidia.js";
import p65 from "./ollama-local.js"; import p65 from "./ollama-local.js";
import p66 from "./ollama.js"; import p66 from "./ollama.js";
import p123 from "./ollama-search.js";
import p67 from "./openai.js"; import p67 from "./openai.js";
import p68 from "./opencode-go.js"; import p68 from "./opencode-go.js";
import p69 from "./opencode.js"; import p69 from "./opencode.js";
@@ -121,6 +122,7 @@ import p118 from "./selfhosted-tts.js";
import p119 from "./selfhosted-embedding.js"; import p119 from "./selfhosted-embedding.js";
import p120 from "./fish-audio.js"; import p120 from "./fish-audio.js";
import p121 from "./alitp-intl.js"; import p121 from "./alitp-intl.js";
import p122 from "./xquik.js";
export default [ export default [
p0, p0,
@@ -190,6 +192,7 @@ export default [
p64, p64,
p65, p65,
p66, p66,
p123,
p67, p67,
p68, p68,
p69, p69,
@@ -243,4 +246,5 @@ export default [
p119, p119,
p120, p120,
p121, p121,
p122,
]; ];

View File

@@ -0,0 +1,35 @@
export default {
id: "ollama-search",
alias: "ollama-search",
display: {
name: "Ollama Search",
icon: "cloud",
color: "#ffffff",
textIcon: "OL",
website: "https://ollama.com",
notice: {
text: "Web search via Ollama Cloud subscription. Reuses the API key from the Ollama (chat) provider.",
apiKeyUrl: "https://ollama.com/settings/keys",
},
},
category: "apikey",
authType: "apikey",
authModes: ["apikey"],
serviceKinds: ["webSearch"],
// Credential fallback: reuses the API key registered under the `ollama`
// chat provider — one key, chat + search.
credentialFallback: "ollama",
searchConfig: {
baseUrl: "https://ollama.com/api/web_search",
method: "POST",
authType: "apikey",
authHeader: "bearer",
costPerQuery: 0,
freeMonthlyQuota: 1000,
searchTypes: ["web"],
defaultMaxResults: 5,
maxMaxResults: 10,
timeoutMs: 10000,
cacheTTLMs: 300000,
},
};

View File

@@ -31,7 +31,16 @@ export default {
{ id: "qwen3.5", name: "Qwen3.5" }, { id: "qwen3.5", name: "Qwen3.5" },
{ id: "minimax-m3", name: "MiniMax M3" }, { id: "minimax-m3", name: "MiniMax M3" },
], ],
serviceKinds: ["llm"], serviceKinds: ["llm", "webFetch"],
fetchConfig: {
baseUrl: "https://ollama.com/api/web_fetch",
method: "POST",
authType: "apikey",
authHeader: "bearer",
formats: ["markdown"],
maxCharacters: 200000,
timeoutMs: 30000,
},
features: { features: {
usage: true, usage: true,
usageApikey: true, usageApikey: true,

View File

@@ -21,6 +21,9 @@ export default {
transport: { transport: {
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
headers: {}, headers: {},
usage: {
url: "https://opencode.ai/zen/go/v1/usage",
},
}, },
// Multi-endpoint: pick the transport matching the client sourceFormat to skip // Multi-endpoint: pick the transport matching the client sourceFormat to skip
// translation. Guarded per-model by `supportedFormats` (see chatCore) because // translation. Guarded per-model by `supportedFormats` (see chatCore) because
@@ -31,12 +34,14 @@ export default {
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, { format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
], ],
models: [ models: [
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] }, { id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] }, { id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] }, { id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] }, { id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] }, { id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] }, { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] }, { id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
@@ -45,5 +50,13 @@ export default {
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
// Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces
// chatCore past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
], ],
features: {
usage: true,
usageApikey: true,
},
}; };

View File

@@ -19,7 +19,12 @@ export default {
}, },
noAuth: true, noAuth: true,
}, },
models: [], models: [
// Muse Spark models are served by /zen/v1/responses; the rest stay on
// /chat/completions, so the format is declared per-model, not per-provider.
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
],
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" }, modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true, passthroughModels: true,
}; };

View File

@@ -30,12 +30,15 @@ export default {
{ id: "auto", name: "Auto" }, { id: "auto", name: "Auto" },
{ id: "performance", name: "Performance" }, { id: "performance", name: "Performance" },
{ id: "efficient", name: "Efficient" }, { id: "efficient", name: "Efficient" },
{ id: "qmodel_preview", name: "Qwen3.8-Max-Preview" }, { id: "lite", name: "Lite" },
{ id: "qmodel_38max", name: "Qwen3.8-Max" },
{ id: "qmodel_latest", name: "Qwen3.7-Max" }, { id: "qmodel_latest", name: "Qwen3.7-Max" },
{ id: "qmodel", name: "Qwen3.7-Plus" }, { id: "qmodel", name: "Qwen3.7-Plus" },
{ id: "qfmodel", name: "Qwen3.8-Flash" },
{ id: "kmodel_latest", name: "Kimi-K3" }, { id: "kmodel_latest", name: "Kimi-K3" },
{ id: "kmodel", name: "Kimi-K2.7-Code" }, { id: "kmodel", name: "Kimi-K2.7-Code" },
{ id: "gm51model", name: "GLM-5.2" }, { id: "gmodel", name: "GLM-5.3" },
{ id: "gfmodel", name: "GLM-5.3-Flash" },
{ id: "dmodel", name: "DeepSeek-V4-Pro" }, { id: "dmodel", name: "DeepSeek-V4-Pro" },
{ id: "dfmodel", name: "DeepSeek-V4-Flash" }, { id: "dfmodel", name: "DeepSeek-V4-Flash" },
{ id: "mmodel", name: "MiniMax-M3" }, { id: "mmodel", name: "MiniMax-M3" },

View File

@@ -24,129 +24,31 @@ export default {
validateUrl: "https://api.tokenrouter.com/v1/models", validateUrl: "https://api.tokenrouter.com/v1/models",
thinkingFormat: "tokenrouter", thinkingFormat: "tokenrouter",
}, },
// Seed snapshot from live /v1/models (120 entries). Latest catalogue is // Seed snapshot from live /v1/models. Latest catalogue is
// fetched via modelsFetcher; other ids still accepted via passthroughModels. // fetched via modelsFetcher; other ids still accepted via passthroughModels.
models: [ models: [
{ id: "MiniMax-Hailuo-2.3", name: "Minimax Hailuo 2.3", kind: "video" },
{ id: "MiniMax-M3", name: "Minimax M3" },
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5" },
{ id: "anthropic/claude-haiku-4.5", name: "Claude Haiku 4.5" }, { id: "anthropic/claude-haiku-4.5", name: "Claude Haiku 4.5" },
{ id: "anthropic/claude-opus-4.5", name: "Claude Opus 4.5" }, { id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
{ id: "anthropic/claude-opus-4.6", name: "Claude Opus 4.6" },
{ id: "anthropic/claude-opus-4.7", name: "Claude Opus 4.7" },
{ id: "anthropic/claude-opus-4.7-fast", name: "Claude Opus 4.7 Fast" },
{ id: "anthropic/claude-opus-4.8", name: "Claude Opus 4.8" }, { id: "anthropic/claude-opus-4.8", name: "Claude Opus 4.8" },
{ id: "anthropic/claude-opus-4.8-fast", name: "Claude Opus 4.8 Fast" }, { id: "anthropic/claude-opus-4.8-fast", name: "Claude Opus 4.8 Fast" },
{ id: "anthropic/claude-opus-5", name: "Claude Opus 5" },
{ id: "anthropic/claude-opus-5-fast", name: "Claude Opus 5 Fast" },
{ id: "anthropic/claude-sonnet-4", name: "Claude Sonnet 4" },
{ id: "anthropic/claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
{ id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "bytedance-seed/seedream-4.5", name: "Seedream 4.5", kind: "image" },
{ id: "bytedance-seed/seedream-5.0-lite", name: "Seedream 5.0 Lite", kind: "image" },
{ id: "bytedance-seed/seedream-5.0-pro", name: "Seedream 5.0 Pro", kind: "image" },
{ id: "claude-haiku-4-5", name: "Claude Haiku 4 5" },
{ id: "claude-opus-4-8-m-aws", name: "Claude Opus 4 8 M Aws" },
{ id: "deepseek/deepseek-v3.2", name: "Deepseek V3.2" },
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
{ id: "deepseek/deepseek-v4-flash-0731", name: "Deepseek V4 Flash 0731" },
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
{ id: "ex/gpt-5.4", name: "Gpt 5.4" },
{ id: "google/gemini-2.5-flash-image", name: "Gemini 2.5 Flash Image" },
{ id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
{ id: "google/gemini-3-pro-image-preview", name: "Gemini 3 Pro Image Preview" },
{ id: "google/gemini-3.1-flash-image-preview", name: "Gemini 3.1 Flash Image Preview" },
{ id: "google/gemini-3.1-flash-lite-image", name: "Gemini 3.1 Flash Lite Image" },
{ id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
{ id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
{ id: "google/gemini-embedding-2", name: "Gemini Embedding 2" },
{ id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B It" },
{ id: "happyhorse-1.0-t2v", name: "Happyhorse 1.0 T2V", kind: "video" },
{ id: "kling-3.0-turbo", name: "Kling 3.0 Turbo", kind: "video" },
{ id: "kling-v2-6", name: "Kling V2 6", kind: "video" },
{ id: "kling-v3", name: "Kling V3", kind: "video" },
{ id: "kling-v3-omni", name: "Kling V3 Omni", kind: "video" },
{ id: "microsoft/mai-image-2.5", name: "Mai Image 2.5" },
{ id: "minimax/minimax-m2-her", name: "Minimax M2 Her" },
{ id: "minimax/minimax-m2.1", name: "Minimax M2.1" },
{ id: "minimax/minimax-m2.1-highspeed", name: "Minimax M2.1 Highspeed" },
{ id: "minimax/minimax-m2.5", name: "Minimax M2.5" },
{ id: "minimax/minimax-m2.7", name: "Minimax M2.7" },
{ id: "minimax/minimax-m2.7-highspeed", name: "Minimax M2.7 Highspeed" },
{ id: "miromind/mirothinker-1-7-deepresearch", name: "Mirothinker 1 7 Deepresearch" },
{ id: "miromind/mirothinker-1-7-deepresearch-mini", name: "Mirothinker 1 7 Deepresearch Mini" },
{ id: "mistralai/devstral-2512", name: "Devstral 2512" },
{ id: "mistralai/mistral-medium-3-5", name: "Mistral Medium 3 5" },
{ id: "mistralai/mistral-small-2603", name: "Mistral Small 2603" },
{ id: "mistralai/voxtral-small-24b-2507", name: "Voxtral Small 24B 2507" },
{ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5" },
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
{ id: "moonshotai/kimi-k3", name: "Kimi K3" },
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
{ id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", name: "Nemotron 3 Nano Omni 30B A3B Reasoning:Free" },
{ id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B" },
{ id: "openai/gpt-4o-mini", name: "Gpt 4O Mini" },
{ id: "openai/gpt-5", name: "Gpt 5" },
{ id: "openai/gpt-5-image", name: "Gpt 5 Image" },
{ id: "openai/gpt-5-image-mini", name: "Gpt 5 Image Mini" },
{ id: "openai/gpt-5-mini", name: "Gpt 5 Mini" },
{ id: "openai/gpt-5.2", name: "Gpt 5.2" },
{ id: "openai/gpt-5.4", name: "Gpt 5.4" }, { id: "openai/gpt-5.4", name: "Gpt 5.4" },
{ id: "openai/gpt-5.4-image-2", name: "Gpt 5.4 Image 2", kind: "image" },
{ id: "openai/gpt-5.4-mini", name: "Gpt 5.4 Mini" }, { id: "openai/gpt-5.4-mini", name: "Gpt 5.4 Mini" },
{ id: "openai/gpt-5.4-nano", name: "Gpt 5.4 Nano" },
{ id: "openai/gpt-5.4-pro", name: "Gpt 5.4 Pro" }, { id: "openai/gpt-5.4-pro", name: "Gpt 5.4 Pro" },
{ id: "openai/gpt-5.5", name: "Gpt 5.5" }, { id: "openai/gpt-5.5", name: "Gpt 5.5" },
{ id: "openai/gpt-5.5-pro", name: "Gpt 5.5 Pro" },
{ id: "openai/gpt-5.6-luna", name: "Gpt 5.6 Luna" },
{ id: "openai/gpt-5.6-sol", name: "Gpt 5.6 Sol" }, { id: "openai/gpt-5.6-sol", name: "Gpt 5.6 Sol" },
{ id: "openai/gpt-5.6-terra", name: "Gpt 5.6 Terra" }, { id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
{ id: "openai/gpt-audio", name: "Gpt Audio", kind: "audio" }, { id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
{ id: "openai/gpt-audio-mini", name: "Gpt Audio Mini", kind: "audio" }, { id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
{ id: "openai/gpt-oss-120b", name: "Gpt Oss 120B" }, { id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
{ id: "qwen/qwen3-coder-next", name: "Qwen3 Coder Next" }, { id: "qwen/qwen3-coder-next", name: "Qwen3 Coder Next" },
{ id: "qwen/qwen3.5-122b-a10b", name: "Qwen3.5 122B A10B" },
{ id: "qwen/qwen3.5-35b-a3b", name: "Qwen3.5 35B A3B" },
{ id: "qwen/qwen3.5-397b-a17b", name: "Qwen3.5 397B A17B" },
{ id: "qwen/qwen3.5-9b", name: "Qwen3.5 9B" },
{ id: "qwen/qwen3.5-flash", name: "Qwen3.5 Flash" },
{ id: "qwen/qwen3.5-plus-02-15", name: "Qwen3.5 Plus 02 15" },
{ id: "qwen/qwen3.6-plus", name: "Qwen3.6 Plus" },
{ id: "qwen/qwen3.7-max", name: "Qwen3.7 Max" }, { id: "qwen/qwen3.7-max", name: "Qwen3.7 Max" },
{ id: "qwen/qwen3.7-plus", name: "Qwen3.7 Plus" },
{ id: "qwen/qwen3.8-max", name: "Qwen3.8 Max" }, { id: "qwen/qwen3.8-max", name: "Qwen3.8 Max" },
{ id: "qwen3.5-omni-plus", name: "Qwen3.5 Omni Plus" }, { id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" }, { id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
{ id: "sakana/fugu-ultra", name: "Fugu Ultra" }, { id: "z-ai/glm-5.3-free", name: "Glm 5.3 Free" },
{ id: "seed-2-0-code-preview-260328", name: "Seed 2 0 Code Preview 260328" },
{ id: "seed-2-0-lite-260428", name: "Seed 2 0 Lite 260428" },
{ id: "seed-2-0-mini-260428", name: "Seed 2 0 Mini 260428" },
{ id: "seed-2-0-pro-260328", name: "Seed 2 0 Pro 260328" },
{ id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash" },
{ id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash" },
{ id: "tencent/hy3-preview", name: "Hy3 Preview" },
{ id: "x-ai/grok-4.1-fast", name: "Grok 4.1 Fast" },
{ id: "x-ai/grok-4.20-beta", name: "Grok 4.20 Beta" },
{ id: "x-ai/grok-4.3", name: "Grok 4.3" },
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
{ id: "x-ai/grok-build-0.1", name: "Grok Build 0.1" },
{ id: "xiaomi/mimo-v2-flash", name: "Mimo V2 Flash" },
{ id: "xiaomi/mimo-v2-omni", name: "Mimo V2 Omni" },
{ id: "xiaomi/mimo-v2-pro", name: "Mimo V2 Pro" },
{ id: "xiaomi/mimo-v2.5", name: "Mimo V2.5" },
{ id: "xiaomi/mimo-v2.5-pro", name: "Mimo V2.5 Pro" },
{ id: "z-ai/glm-4.5-air", name: "Glm 4.5 Air" },
{ id: "z-ai/glm-4.6", name: "Glm 4.6" },
{ id: "z-ai/glm-4.6v", name: "Glm 4.6V" },
{ id: "z-ai/glm-4.7", name: "Glm 4.7" },
{ id: "z-ai/glm-5", name: "Glm 5" },
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
{ id: "z-ai/glm-5.1", name: "Glm 5.1" },
{ id: "z-ai/glm-5.2", name: "Glm 5.2" }, { id: "z-ai/glm-5.2", name: "Glm 5.2" },
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
], ],
serviceKinds: ["llm", "embedding", "image"], serviceKinds: ["llm", "embedding", "image"],
embeddingConfig: { embeddingConfig: {

View File

@@ -25,17 +25,50 @@ export default {
clientId: "b1a00492-073a-47ea-816f-4c329264a828", clientId: "b1a00492-073a-47ea-816f-4c329264a828",
tokenUrl: "https://auth.x.ai/oauth2/token", tokenUrl: "https://auth.x.ai/oauth2/token",
refreshUrl: "https://auth.x.ai/oauth2/token", refreshUrl: "https://auth.x.ai/oauth2/token",
// OAuth-only SuperGrok quota surfaces:
// - url: monthly API usage allotment (JSON)
// - creditsUrl: weekly SuperGrok limit (grpc-web)
// - settingsUrl: plan label (subscription_tier_display)
usage: {
url: "https://cli-chat-proxy.grok.com/v1/billing",
creditsUrl: "https://grok.com/grok_api_v2.GrokBuildBilling/GetGrokCreditsConfig",
settingsUrl: "https://cli-chat-proxy.grok.com/v1/settings",
},
}, },
models: [ models: [
{ id: "grok-4.6", name: "Grok 4.6" },
{ id: "grok-4.5", name: "Grok 4.5" },
{ id: "grok-4", name: "Grok 4" }, { id: "grok-4", name: "Grok 4" },
{ id: "grok-4-fast-reasoning", name: "Grok 4 Fast Reasoning" }, { id: "grok-4-fast-reasoning", name: "Grok 4 Fast Reasoning" },
{ id: "grok-code-fast-1", name: "Grok Code Fast" }, { id: "grok-code-fast-1", name: "Grok Code Fast" },
{ id: "grok-3", name: "Grok 3" }, { id: "grok-3", name: "Grok 3" },
{ id: "grok-2-image-1212", name: "Grok 2 Image", params: ["n","response_format"], kind: "image" }, {
{ id: "grok-imagine-video", name: "Grok Imagine Video", params: ["duration","aspect_ratio","resolution"], kind: "video" }, id: "grok-imagine-image-quality",
name: "Grok Imagine Image Quality",
capabilities: ["text2img", "edit"],
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
kind: "image",
},
{
id: "grok-2-image-1212",
name: "Grok 2 Image",
capabilities: ["text2img", "edit"],
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
kind: "image",
},
{
id: "grok-imagine-video",
name: "Grok Imagine Video",
params: ["duration", "aspect_ratio", "resolution"],
kind: "video",
},
], ],
serviceKinds: ["llm","imageToText","webSearch","image","video"], serviceKinds: ["llm", "imageToText", "webSearch", "image", "video"],
imageConfig: { baseUrl: "https://api.x.ai/v1/images/generations", bodyFields: ["model","prompt","n","response_format"] }, imageConfig: {
baseUrl: "https://api.x.ai/v1/images/generations",
editsUrl: "https://api.x.ai/v1/images/edits",
bodyFields: ["model", "prompt", "n", "response_format", "aspect_ratio", "resolution", "image", "images"],
},
// Async video jobs (POST returns { request_id }, GET polls until done/failed). // Async video jobs (POST returns { request_id }, GET polls until done/failed).
// Docs: https://docs.x.ai/developers/rest-api-reference/inference/videos // Docs: https://docs.x.ai/developers/rest-api-reference/inference/videos
videoConfig: { baseUrl: "https://api.x.ai/v1/videos" }, videoConfig: { baseUrl: "https://api.x.ai/v1/videos" },
@@ -44,4 +77,7 @@ export default {
endpoint: "https://api.x.ai/v1/responses", endpoint: "https://api.x.ai/v1/responses",
pricingUrl: "https://x.ai/api#pricing", pricingUrl: "https://x.ai/api#pricing",
}, },
features: {
usage: true,
},
}; };

View File

@@ -0,0 +1,35 @@
export default {
id: "xquik",
alias: "xquik",
display: {
name: "Xquik",
icon: "tag",
color: "#5C3327",
textIcon: "XQ",
website: "https://docs.xquik.com/api-reference/x/search-tweets",
notice: {
apiKeyUrl: "https://xquik.com",
text: "Searches public X posts. Billing uses 1 Xquik credit per returned post."
}
},
category: "apikey",
authType: "apikey",
serviceKinds: [
"webSearch"
],
searchConfig: {
baseUrl: "https://xquik.com/api/v1/x/tweets/search",
validateUrl: "https://xquik.com/api/v1/credits",
method: "GET",
authType: "apikey",
authHeader: "x-api-key",
searchTypes: [
"x"
],
defaultMaxResults: 5,
maxMaxResults: 100,
timeoutMs: 10000,
cacheTTLMs: 60000,
creditsPerResult: 1
}
};

View File

@@ -22,6 +22,7 @@ export function mapStainlessArch() {
// Anthropic API version (single source — reused across claude-format providers/executors) // Anthropic API version (single source — reused across claude-format providers/executors)
export const ANTHROPIC_API_VERSION = "2023-06-01"; export const ANTHROPIC_API_VERSION = "2023-06-01";
export const CLAUDE_CLI_VERSION = "2.1.258";
// Shared Claude-compatible API headers (reused across claude-format providers) // Shared Claude-compatible API headers (reused across claude-format providers)
export const CLAUDE_API_HEADERS = { export const CLAUDE_API_HEADERS = {
@@ -34,7 +35,7 @@ export const CLAUDE_CLI_SPOOF_HEADERS = {
"Anthropic-Version": ANTHROPIC_API_VERSION, "Anthropic-Version": ANTHROPIC_API_VERSION,
"Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28", "Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28",
"Anthropic-Dangerous-Direct-Browser-Access": "true", "Anthropic-Dangerous-Direct-Browser-Access": "true",
"User-Agent": "claude-cli/2.1.92 (external, sdk-cli)", "User-Agent": `claude-cli/${CLAUDE_CLI_VERSION} (external, sdk-cli)`,
"X-App": "cli", "X-App": "cli",
"X-Stainless-Helper-Method": "stream", "X-Stainless-Helper-Method": "stream",
"X-Stainless-Retry-Count": "0", "X-Stainless-Retry-Count": "0",
@@ -74,10 +75,10 @@ export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
export const OPENAI_COMPAT_BASE = "https://api.openai.com/v1"; export const OPENAI_COMPAT_BASE = "https://api.openai.com/v1";
export const ANTHROPIC_COMPAT_BASE = "https://api.anthropic.com/v1"; export const ANTHROPIC_COMPAT_BASE = "https://api.anthropic.com/v1";
// Official Antigravity IDE Desktop 2.1.1 fingerprint captured from macOS arm64. // Official Antigravity IDE Desktop 2.11.0 fingerprint captured from macOS arm64.
// Keep this static even when 9router runs on Linux: the provider profile is // Keep this static even when 9router runs on Linux: the provider profile is
// intentionally matching the IDE client, not the server host. // intentionally matching the IDE client, not the server host.
export const ANTIGRAVITY_IDE_VERSION = "2.1.1"; export const ANTIGRAVITY_IDE_VERSION = "2.11.0";
export const ANTIGRAVITY_IDE_BASE_URL = "https://daily-cloudcode-pa.googleapis.com"; export const ANTIGRAVITY_IDE_BASE_URL = "https://daily-cloudcode-pa.googleapis.com";
export const ANTIGRAVITY_IDE_USER_AGENT = `antigravity/ide/${ANTIGRAVITY_IDE_VERSION} darwin/arm64`; export const ANTIGRAVITY_IDE_USER_AGENT = `antigravity/ide/${ANTIGRAVITY_IDE_VERSION} darwin/arm64`;

View File

@@ -35,10 +35,23 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh
// Model-name pattern overrides (glob, first match wins) — more precise than format default. // Model-name pattern overrides (glob, first match wins) — more precise than format default.
const PATTERN_THINKING = [ const PATTERN_THINKING = [
{ provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS },
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS }, { provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking { pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
// codebuddy-cn per-model effort sets — the server's product-config payload
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
// → all 200), but values outside a model's supportedEfforts are silently
// clamped, so the declared set stays authoritative for the picker. Models
// that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x /
// kimi-k3-1 / minimax-m3) fall through to the openai format default.
{ provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] },
{ provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
]; ];
// Returns valid thinking levels for a model, or null when the model has no reasoning. // Returns valid thinking levels for a model, or null when the model has no reasoning.

View File

@@ -0,0 +1,42 @@
// Name-based vision detection — last resort when neither the catalog file nor
// the capability tables know a model. Vendors put the modality in the id
// ("qwen3-vl-plus", "glm-4.6v", "deepseek-v4-flash-vision-exp"), so a custom or
// freshly released model still gets image input instead of silently dropping it.
//
// Only ever turns vision ON. Never used to turn a declared capability off.
const SEP = "[-_/:.]";
// Image GENERATION, video generation, and non-chat models also carry these
// words but take no image input — checked first so they can never match.
const NOT_VISION = new RegExp(
[
`(^|${SEP})(image|img)(${SEP}|$)`,
"stable-image", "gen[0-9]_image", "nanobanana", "imagine",
"t2v", "i2v", "flux", "dall", "sdxl", "diffusion",
"embed", "rerank", "guard", "moderation",
"tts", "stt", "whisper", "voice", "speech", "audio",
].join("|"),
"i"
);
// Explicit modality words, plus the "<digit>v" suffix vendors use for vision
// variants (glm-4.6v, glm-5v-turbo). The digit-v branch requires a dotted
// version so the never-shipped `gpt-4v` cannot match.
const VISION_NAME = new RegExp(
[
`(^|${SEP})(vision|vl|vlm|multimodal|omni|visual)(${SEP}|$)`,
`[0-9]\\.[0-9]+v(${SEP}|$)`,
`(^|${SEP})glm-[0-9]+v(${SEP}|$)`,
"(^|[-_/:.])(llava|pixtral|internvl|cogvlm|minicpm-v|moondream|idefics|fuyu)",
].join("|"),
"i"
);
// Does this model id look like a vision model? Name signal only.
export function looksLikeVisionModel(modelId) {
if (!modelId) return false;
const id = String(modelId).toLowerCase();
if (NOT_VISION.test(id)) return false;
return VISION_NAME.test(id);
}

View File

@@ -7,6 +7,12 @@ import {
const DEFAULT_TIMEOUT_MS = 3000; const DEFAULT_TIMEOUT_MS = 3000;
function normalizeTimeout(value) {
return typeof value === "number" && Number.isFinite(value) && value > 0
? value
: DEFAULT_TIMEOUT_MS;
}
function jsonBytes(value) { function jsonBytes(value) {
try { try {
return new TextEncoder().encode(JSON.stringify(value) || "").length; return new TextEncoder().encode(JSON.stringify(value) || "").length;
@@ -240,6 +246,7 @@ async function callCompress(url, messages, model, timeoutMs, compressUserMessage
// /v1/compress only understands OpenAI shape, so Claude bodies are translated // /v1/compress only understands OpenAI shape, so Claude bodies are translated
// to OpenAI, compressed, then translated back using 9Router's own translators. // to OpenAI, compressed, then translated back using 9Router's own translators.
export async function compressWithHeadroom(body, { enabled, url, model, format, compressUserMessages, timeoutMs = DEFAULT_TIMEOUT_MS, diagnostics = null } = {}) { export async function compressWithHeadroom(body, { enabled, url, model, format, compressUserMessages, timeoutMs = DEFAULT_TIMEOUT_MS, diagnostics = null } = {}) {
timeoutMs = normalizeTimeout(timeoutMs);
if (!enabled) { if (!enabled) {
setDiagnostic(diagnostics, "disabled"); setDiagnostic(diagnostics, "disabled");
return null; return null;
@@ -281,7 +288,10 @@ export async function compressWithHeadroom(body, { enabled, url, model, format,
return null; return null;
} }
const oai = openaiResponsesToOpenAIRequest(model, body, false); const oai = openaiResponsesToOpenAIRequest(model, body, false);
if (!Array.isArray(oai?.messages)) return null; if (!Array.isArray(oai?.messages)) {
setDiagnostic(diagnostics, "openai-responses request did not translate to messages[]");
return null;
}
const data = await callCompress(url, oai.messages, model, timeoutMs, compressUserMessages, diagnostics || {}); const data = await callCompress(url, oai.messages, model, timeoutMs, compressUserMessages, diagnostics || {});
if (!data) return null; if (!data) return null;
// input: undefined so the translator rebuilds input from the compressed // input: undefined so the translator rebuilds input from the compressed

View File

@@ -3,96 +3,335 @@
// native-passthrough flows. Used by caveman.js and ponytail.js. // native-passthrough flows. Used by caveman.js and ponytail.js.
import { FORMATS } from "../translator/formats.js"; import { FORMATS } from "../translator/formats.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK, RESPONSES_ITEM } from "../translator/schema/blocks.js";
import { ROLE } from "../translator/schema/roles.js";
const SEP = "\n\n"; const SEP = "\n\n";
export function injectSystemPrompt(body, format, prompt) { export function injectSystemPrompt(body, format, prompt) {
if (!body || !prompt) return; try {
if (!body || !prompt) return;
if (typeof body !== "object") return;
switch (format) { // Kiro wire shape is unique (conversationState/systemPrompt) — handle directly.
case FORMATS.CLAUDE: if (isKiroBody(body) || format === FORMATS.KIRO) {
injectKiroSystem(body, prompt);
return;
}
// Claude/Gemini own a dedicated system field, yet their bodies also carry
// messages[]/contents[] — decide by format label before the shape sniff below.
// Anthropic rejects a "system" role inside messages[] (no such input role).
if (format === FORMATS.CLAUDE) {
injectClaudeSystem(body, prompt); injectClaudeSystem(body, prompt);
return; return;
case FORMATS.GEMINI: }
case FORMATS.GEMINI_CLI: if (format === FORMATS.GEMINI || format === FORMATS.GEMINI_CLI
case FORMATS.VERTEX: || format === FORMATS.VERTEX || format === FORMATS.ANTIGRAVITY) {
case FORMATS.ANTIGRAVITY:
// Antigravity wraps Gemini shape in body.request → injectGeminiSystem handles it // Antigravity wraps Gemini shape in body.request → injectGeminiSystem handles it
injectGeminiSystem(body, prompt); injectGeminiSystem(body, prompt);
return; return;
default:
// OpenAI and OpenAI-shaped formats (responses/codex/cursor/kiro/ollama)
injectMessagesSystem(body, prompt);
}
}
// OpenAI-shaped: messages[] (chat) or input[] (responses) or instructions (responses string)
function injectMessagesSystem(body, prompt) {
// OpenAI Responses API: top-level string field
if (typeof body.instructions === "string") {
body.instructions = body.instructions
? `${body.instructions}${SEP}${prompt}`
: prompt;
return;
}
const arr = Array.isArray(body.messages) ? body.messages
: Array.isArray(body.input) ? body.input
: null;
if (!arr) return;
const idx = arr.findIndex(m => m && (m.role === "system" || m.role === "developer"));
if (idx >= 0) {
appendToOpenAIMessage(arr[idx], prompt);
} else {
arr.unshift({ role: "system", content: prompt });
}
}
function appendToOpenAIMessage(msg, prompt) {
if (typeof msg.content === "string") {
msg.content = `${msg.content}${SEP}${prompt}`;
} else if (Array.isArray(msg.content)) {
// Responses-style array of parts {type:"input_text"|"text", text}
msg.content.push({ type: "input_text", text: prompt });
} else {
msg.content = prompt;
}
}
// Claude shape: body.system as string | array of {type:"text", text}
// Insert before the last cache_control block to keep injection inside the cached prefix.
function injectClaudeSystem(body, prompt) {
if (typeof body.system === "string" && body.system.length > 0) {
body.system = `${body.system}${SEP}${prompt}`;
return;
}
if (Array.isArray(body.system)) {
const block = { type: "text", text: prompt };
let lastCacheIdx = -1;
for (let i = body.system.length - 1; i >= 0; i--) {
if (body.system[i]?.cache_control) { lastCacheIdx = i; break; }
} }
if (lastCacheIdx >= 0) {
body.system.splice(lastCacheIdx, 0, block); // Dispatch by actual wire shape for OpenAI-shaped formats.
// instructions string takes precedence; messages[] means Chat; input[] means Responses.
if (typeof body.instructions === "string") {
injectInstructionsSystem(body, prompt);
return;
}
if (Array.isArray(body.messages)) {
injectChatSystem(body, prompt);
return;
}
if (Array.isArray(body.input)) {
// Responses input[]: empty array already normalized elsewhere; string stays untouched here
injectResponsesInputSystem(body, prompt);
return;
}
if (typeof body.input === "string") {
// string input must stay untouched
return;
}
// OpenAI-shaped but no array (e.g. empty body) — no-op
} catch (_) {
// fail-open
}
}
function isKiroBody(body) {
if (!body || typeof body !== "object") return false;
if (typeof body.systemPrompt !== "string") return false;
const cs = body.conversationState;
if (!cs || typeof cs !== "object") return false;
return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object");
}
// Exact idempotency: prompt present as its own SEP-delimited segment (or the
// whole string), not as a substring of unrelated text.
function hasPrompt(haystack, prompt) {
if (!haystack || typeof haystack !== "string") return false;
if (haystack === prompt) return true;
return haystack.split(SEP).includes(prompt);
}
function dedupStringAppend(curr, prompt) {
if (!curr) return prompt;
if (hasPrompt(curr, prompt)) return curr;
return `${curr}${SEP}${prompt}`;
}
// ---- OpenAI instructions string ----
function injectInstructionsSystem(body, prompt) {
try {
const curr = body.instructions;
if (typeof curr !== "string") return;
if (hasPrompt(curr, prompt)) return;
const next = curr ? `${curr}${SEP}${prompt}` : prompt;
try { body.instructions = next; } catch (_) { /* frozen/proxy fail-open */ }
} catch (_) {}
}
// ---- Chat messages[] ----
function injectChatSystem(body, prompt) {
try {
const arr = body.messages;
if (!Array.isArray(arr)) return;
// Exact idempotency: scan existing system/developer content for full prompt
if (containsPromptInMessages(arr, prompt)) return;
let idx = -1;
try { idx = arr.findIndex(m => m && (m.role === ROLE.SYSTEM || m.role === ROLE.DEVELOPER)); } catch (_) { return; }
if (idx >= 0) {
appendToChatMessage(arr[idx], prompt);
} else { } else {
body.system.push(block); // create typed system message at index 0; fail-open on frozen/proxy
try { arr.unshift({ role: ROLE.SYSTEM, content: prompt }); } catch (_) {}
} }
return; } catch (_) {}
}
body.system = prompt;
} }
// Gemini shape: body.system_instruction | body.systemInstruction | body.request.systemInstruction function containsPromptInMessages(arr, prompt) {
// Each shape: { parts: [{ text }] } try {
function injectGeminiSystem(body, prompt) { for (const m of arr) {
const target = body.request && typeof body.request === "object" ? body.request : body; if (!m || (m.role !== ROLE.SYSTEM && m.role !== ROLE.DEVELOPER)) continue;
const useSnake = Object.prototype.hasOwnProperty.call(target, "system_instruction"); const c = m.content;
const key = useSnake ? "system_instruction" : "systemInstruction"; if (typeof c === "string" && hasPrompt(c, prompt)) return true;
const sys = target[key]; if (Array.isArray(c)) {
if (sys && Array.isArray(sys.parts)) { for (const part of c) {
sys.parts.push({ text: prompt }); if (part && typeof part.text === "string" && hasPrompt(part.text, prompt)) return true;
return; }
} }
target[key] = { parts: [{ text: prompt }] }; }
} catch (_) {}
return false;
}
function appendToChatMessage(msg, prompt) {
try {
if (!msg || typeof msg !== "object") return;
const c = msg.content;
if (typeof c === "string") {
const next = dedupStringAppend(c, prompt);
if (next === c) return;
// avoid partial mutation: try assignment, bail if setter throws
try { msg.content = next; } catch (_) {}
return;
}
if (Array.isArray(c)) {
// already deduped at message level; but guard block-level too
try {
if (c.some(b => b && b.text === prompt)) return;
} catch (_) {}
try { c.push({ type: OPENAI_BLOCK.TEXT, text: prompt }); } catch (_) {}
return;
}
try { msg.content = prompt; } catch (_) {}
} catch (_) {}
}
// ---- Responses input[] ----
function injectResponsesInputSystem(body, prompt) {
try {
const arr = body.input;
if (!Array.isArray(arr)) return;
// instructions already handled above
if (containsPromptInResponsesInput(arr, prompt)) return;
// find system/developer message items only (type === message)
let idx = -1;
try {
idx = arr.findIndex(m => m && m.type === RESPONSES_ITEM.MESSAGE && (m.role === ROLE.SYSTEM || m.role === ROLE.DEVELOPER));
} catch (_) { return; }
if (idx >= 0) {
appendToResponsesMessage(arr[idx], prompt);
} else {
const msg = { type: RESPONSES_ITEM.MESSAGE, role: ROLE.SYSTEM, content: [{ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }] };
try { arr.unshift(msg); } catch (_) {}
}
} catch (_) {}
}
function containsPromptInResponsesInput(arr, prompt) {
try {
for (const item of arr) {
if (!item || item.type !== RESPONSES_ITEM.MESSAGE) continue;
if (item.role !== ROLE.SYSTEM && item.role !== ROLE.DEVELOPER) continue;
const c = item.content;
if (typeof c === "string" && hasPrompt(c, prompt)) return true;
if (Array.isArray(c)) {
for (const part of c) {
if (part && typeof part.text === "string" && hasPrompt(part.text, prompt)) return true;
}
}
}
} catch (_) {}
return false;
}
function appendToResponsesMessage(msg, prompt) {
try {
if (!msg || typeof msg !== "object") return;
const c = msg.content;
if (typeof c === "string") {
const next = dedupStringAppend(c, prompt);
if (next === c) return;
try { msg.content = next; } catch (_) {}
return;
}
if (Array.isArray(c)) {
try { if (c.some(b => b && b.text === prompt)) return; } catch (_) {}
try { c.push({ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }); } catch (_) {}
return;
}
try { msg.content = [{ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }]; } catch (_) {}
} catch (_) {}
}
// ---- Claude ----
function injectClaudeSystem(body, prompt) {
try {
const sys = body.system;
if (typeof sys === "string") {
if (hasPrompt(sys, prompt)) return;
const next = sys.length > 0 ? `${sys}${SEP}${prompt}` : prompt;
try { body.system = next; } catch (_) {}
return;
}
if (Array.isArray(sys)) {
try { if (sys.some(b => b && b.text === prompt)) return; } catch (_) {}
const block = { type: CLAUDE_BLOCK.TEXT, text: prompt };
let lastCacheIdx = -1;
try {
for (let i = sys.length - 1; i >= 0; i--) {
if (sys[i]?.cache_control) { lastCacheIdx = i; break; }
}
} catch (_) {}
try {
if (lastCacheIdx >= 0) sys.splice(lastCacheIdx, 0, block);
else sys.push(block);
} catch (_) {}
return;
}
// absent/null
try { body.system = prompt; } catch (_) {}
} catch (_) {}
}
// ---- Gemini ----
function injectGeminiSystem(body, prompt) {
try {
let target = body;
try {
if (body.request && typeof body.request === "object") target = body.request;
} catch (_) {}
let useSnake = false;
try { useSnake = Object.prototype.hasOwnProperty.call(target, "system_instruction"); } catch (_) {}
const key = useSnake ? "system_instruction" : "systemInstruction";
let sys;
try { sys = target[key]; } catch (_) { sys = undefined; }
if (sys && Array.isArray(sys.parts)) {
try { if (sys.parts.some(p => p && p.text === prompt)) return; } catch (_) {}
try { sys.parts.push({ text: prompt }); } catch (_) {}
return;
}
try { target[key] = { parts: [{ text: prompt }] }; } catch (_) {}
} catch (_) {}
}
// ---- Kiro ----
// Updates top-level systemPrompt and only the mirrored leading prefix of the
// first user history turn, else current user. next = old + SEP + prompt.
// Replace old leading prefix only; preserve time context and user tail.
function injectKiroSystem(body, prompt) {
try {
let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : "";
// Repair path: a previous partial write left systemPrompt updated but user
// content still mirroring the pre-write prefix. Re-derive the effective old
// prefix from content so this pass converges instead of early-returning.
const cs0 = body.conversationState;
let firstUser0 = cs0 && Array.isArray(cs0.history)
? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null)
: null;
if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage;
if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) {
const c0 = firstUser0.content;
if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) {
// systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored
oldPrompt = "";
}
}
if (oldPrompt && hasPrompt(oldPrompt, prompt)) return;
const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt;
// Atomicity: write user content first, then systemPrompt only if content
// write succeeded (or was a no-op). If systemPrompt write then fails, the
// repair heuristic above re-derives from content on retry — no permanent
// half-applied state.
const cs = body.conversationState;
let targetMsg = null;
try {
const hist = Array.isArray(cs?.history) ? cs.history : null;
if (hist) {
for (const item of hist) {
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
}
}
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
targetMsg = cs.currentMessage.userInputMessage;
}
} catch (_) { targetMsg = null; }
let sysWritten = false;
try { body.systemPrompt = next; sysWritten = true; } catch (_) {}
const applyContent = () => {
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
if (oldPrompt === "") {
// Empty old prompt: prepend unless already at head (exact, not substring)
if (content.startsWith(prompt) || content.startsWith(next)) return;
const newContent = content ? `${next}${SEP}${content}` : next;
try { targetMsg.content = newContent; } catch (_) {}
return;
}
if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone
if (content.startsWith(next)) return; // already applied → idempotent
const tail = content.slice(oldPrompt.length);
try { targetMsg.content = `${next}${tail}`; } catch (_) {}
};
try {
if (targetMsg) applyContent();
} catch (_) {}
if (sysWritten && targetMsg) {
// verify convergence: content should now start with next (or be un-mirrored)
let ok = false;
try {
const c = targetMsg.content;
ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt);
} catch (_) {}
if (!ok) {
try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback
}
}
} catch (_) {}
} }

View File

@@ -26,8 +26,14 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) {
: ""; : "";
for (const rule of ERROR_RULES) { for (const rule of ERROR_RULES) {
// Request-scoped rule: the request body itself is at fault — no cooldown,
// no account lock. Caller must stop rotating and surface the error.
if (rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
return { shouldFallback: false, requestScoped: true, cooldownMs: 0 };
}
// Text-based rule: match substring in error message // Text-based rule: match substring in error message
if (rule.text && lowerError && lowerError.includes(rule.text)) { if (rule.text && !rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
if (rule.backoff) { if (rule.backoff) {
const newLevel = Math.min(backoffLevel + 1, BACKOFF_CONFIG.maxLevel); const newLevel = Math.min(backoffLevel + 1, BACKOFF_CONFIG.maxLevel);
return { shouldFallback: true, cooldownMs: getQuotaCooldown(newLevel), newBackoffLevel: newLevel }; return { shouldFallback: true, cooldownMs: getQuotaCooldown(newLevel), newBackoffLevel: newLevel };

View File

@@ -1,71 +0,0 @@
/**
* Shared combo (model combo) handling with fallback support
*/
/**
* Get combo models from combos data
* @param {string} modelStr - Model string to check
* @param {Array|Object} combosData - Array of combos or object with combos
* @returns {string[]|null} Array of models or null if not a combo
*/
export function getComboModelsFromData(modelStr, combosData) {
// Don't check if it's in provider/model format
if (modelStr.includes("/")) return null;
// Handle both array and object formats
const combos = Array.isArray(combosData) ? combosData : (combosData?.combos || []);
const combo = combos.find(c => c.name === modelStr);
if (combo && combo.models && combo.models.length > 0) {
return combo.models;
}
return null;
}
/**
* Handle combo chat with fallback
* @param {Object} options
* @param {Object} options.body - Request body
* @param {string[]} options.models - Array of model strings to try
* @param {Function} options.handleSingleModel - Function to handle single model: (body, modelStr) => Promise<Response>
* @param {Object} options.log - Logger object
* @returns {Promise<Response>}
*/
export async function handleComboChat({ body, models, handleSingleModel, log }) {
let lastError = null;
for (let i = 0; i < models.length; i++) {
const modelStr = models[i];
log.info("COMBO", `Trying model ${i + 1}/${models.length}: ${modelStr}`);
let result;
try {
result = await handleSingleModel(body, modelStr);
} catch (e) {
lastError = `${modelStr}: ${e.message}`;
log.warn("COMBO", `Model threw exception, trying next`, { model: modelStr, error: e.message });
continue;
}
// Success or client error - return response
if (result.ok || result.status < 500) {
return result;
}
// 5xx error - try next model
lastError = `${modelStr}: ${result.statusText || result.status}`;
log.warn("COMBO", `Model failed, trying next`, { model: modelStr, status: result.status });
}
log.warn("COMBO", "All models failed");
// Return 503 with last error
return new Response(
JSON.stringify({ error: lastError || "All combo models unavailable" }),
{
status: 503,
headers: { "Content-Type": "application/json" }
}
);
}

View File

@@ -203,7 +203,8 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA }; const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA };
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS; const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
const MAX_ATTEMPTS = 5; const MAX_ATTEMPTS = Number(process.env.ONBOARD_MAX_ATTEMPTS) || 2;
const BASE_RETRY_DELAY_MS = Number(process.env.ONBOARD_RETRY_DELAY_MS) || 12_000;
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) { for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
// Bail out immediately if the connection was removed // Bail out immediately if the connection was removed
@@ -241,9 +242,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
throw new Error("onboardUser done but no project_id in response"); throw new Error("onboardUser done but no project_id in response");
} }
// Server not done yet – wait and retry // Server not done yet – wait and retry with jitter
const jitter = Math.floor(Math.random() * 5000);
console.log(`[ProjectId] Onboard attempt ${attempt}/${MAX_ATTEMPTS}: not done yet, waiting...`); console.log(`[ProjectId] Onboard attempt ${attempt}/${MAX_ATTEMPTS}: not done yet, waiting...`);
await new Promise(resolve => setTimeout(resolve, 2000)); await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter));
} catch (error) { } catch (error) {
clearTimeout(timeoutId); clearTimeout(timeoutId);
@@ -256,9 +258,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
console.warn(`[ProjectId] onboardUser failed after ${MAX_ATTEMPTS} attempts: ${error.message}`); console.warn(`[ProjectId] onboardUser failed after ${MAX_ATTEMPTS} attempts: ${error.message}`);
return null; return null;
} }
// Continue to next attempt instead of throwing (which would skip remaining retries) // Wait with jitter before retrying
const jitter = Math.floor(Math.random() * 5000);
console.warn(`[ProjectId] onboardUser attempt ${attempt} failed: ${error.message}, retrying...`); console.warn(`[ProjectId] onboardUser attempt ${attempt} failed: ${error.message}, retrying...`);
await new Promise(resolve => setTimeout(resolve, 2000)); await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter));
} finally { } finally {
clearTimeout(timeoutId); clearTimeout(timeoutId);
externalSignal?.removeEventListener("abort", forwardAbort); externalSignal?.removeEventListener("abort", forwardAbort);

View File

@@ -0,0 +1,56 @@
/**
* Per-provider connect timeout overrides from user settings.
* Settings are read from the DB lazily and cached with a short TTL
* so UI changes take effect without requiring a restart.
*/
let cached = {};
let cacheTs = 0;
const CACHE_TTL_MS = 10_000; // 10s — responsive enough for dashboard changes
async function refreshCache() {
const now = Date.now();
if (now - cacheTs < CACHE_TTL_MS && Object.keys(cached).length > 0) return cached;
try {
const { getSettings } = await import("@/lib/localDb");
// Return full settings so we can read providerTimeouts + globalTimeoutMs
cached = await getSettings();
cacheTs = now;
} catch {
// If DB is unavailable, keep stale cache — don't throw on hot path
}
return cached;
}
/**
* Resolve the effective connect timeout for a provider.
* Priority: per-provider override > global default timeout (settings) > registry config > env default.
* @param {string} providerId
* @param {number} configTimeoutMs - timeoutMs from the static provider registry config
* @param {number} envDefaultMs - global default from env (FETCH_CONNECT_TIMEOUT_MS)
* @returns {number} timeout in milliseconds
*/
export async function resolveProviderTimeoutMs(providerId, configTimeoutMs, envDefaultMs) {
const overrides = await refreshCache();
// 1. Per-provider override (set in provider detail page)
const providerOverride = overrides.providerTimeouts?.[providerId];
if (providerOverride?.timeoutMs && Number.isFinite(providerOverride.timeoutMs) && providerOverride.timeoutMs > 0) {
return providerOverride.timeoutMs;
}
// 2. Global default timeout (set in Profile / Settings page)
const globalDefault = overrides.defaultTimeoutMs;
if (globalDefault && Number.isFinite(globalDefault) && globalDefault > 0) {
return globalDefault;
}
// 3. Registry per-provider config
if (configTimeoutMs && Number.isFinite(configTimeoutMs) && configTimeoutMs > 0) {
return configTimeoutMs;
}
// 4. Env default
return envDefaultMs;
}

View File

@@ -0,0 +1,170 @@
import { makeKv } from "../../src/lib/db/helpers/kvStore.js";
const MAX_SIGNATURES = 2000;
const MAX_PERSISTED_SIGNATURES = 10_000;
const MEMORY_TTL_MS = 1000 * 60 * 60; // 1 hour
const PERSISTED_TTL_MS = 1000 * 60 * 60 * 24 * 7; // 7 days
const SCOPE = "gemini_thought_signatures";
const signatureKv = makeKv(SCOPE);
const memorySignatures = new Map();
let pruneCounter = 0;
function pruneMemoryExpired() {
const now = Date.now();
for (const [key, value] of memorySignatures.entries()) {
if (value.expiresAt <= now) {
memorySignatures.delete(key);
}
}
while (memorySignatures.size > MAX_SIGNATURES) {
const oldestKey = memorySignatures.keys().next().value;
if (!oldestKey) break;
memorySignatures.delete(oldestKey);
}
}
async function maybePrunePersisted() {
pruneCounter++;
if (pruneCounter % 100 !== 0) return;
try {
const all = await signatureKv.getAll();
const keys = Object.keys(all);
const now = Date.now();
const expiredKeys = [];
const valid = [];
for (const k of keys) {
const entry = all[k];
if (!entry || typeof entry.signature !== "string" || (entry.expiresAt && entry.expiresAt <= now)) {
expiredKeys.push(k);
} else {
valid.push({ key: k, createdAt: entry.createdAt || 0 });
}
}
for (const k of expiredKeys) {
await signatureKv.remove(k).catch(() => {});
}
if (valid.length > MAX_PERSISTED_SIGNATURES) {
valid.sort((a, b) => b.createdAt - a.createdAt);
const toRemove = valid.slice(MAX_PERSISTED_SIGNATURES);
for (const item of toRemove) {
await signatureKv.remove(item.key).catch(() => {});
}
}
} catch {
// Fail-open
}
}
/**
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async)
*/
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) {
if (typeof toolCallId !== "string" || !toolCallId) return;
if (typeof signature !== "string" || !signature) return;
const now = Date.now();
pruneMemoryExpired();
const keys = [];
if (sessionId && typeof sessionId === "string") {
keys.push(`${sessionId}:${toolCallId}`);
}
keys.push(toolCallId);
for (const k of keys) {
memorySignatures.set(k, {
signature,
expiresAt: now + MEMORY_TTL_MS,
});
// Async persist to SQLite kv table without blocking
signatureKv.set(k, {
signature,
createdAt: now,
expiresAt: now + PERSISTED_TTL_MS,
}).catch(() => {});
}
maybePrunePersisted().catch(() => {});
}
/**
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback)
*/
export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
if (typeof toolCallId !== "string" || !toolCallId) return null;
pruneMemoryExpired();
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionEntry = memorySignatures.get(sessionKey);
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
return sessionEntry.signature;
}
}
const entry = memorySignatures.get(toolCallId);
if (entry && entry.expiresAt > Date.now()) {
return entry.signature;
}
try {
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionRow = await signatureKv.get(sessionKey);
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) {
memorySignatures.set(sessionKey, {
signature: sessionRow.signature,
expiresAt: Date.now() + MEMORY_TTL_MS,
});
return sessionRow.signature;
}
}
const row = await signatureKv.get(toolCallId);
if (row && typeof row.signature === "string") {
if (row.expiresAt && row.expiresAt <= Date.now()) {
signatureKv.remove(toolCallId).catch(() => {});
return null;
}
memorySignatures.set(toolCallId, {
signature: row.signature,
expiresAt: Date.now() + MEMORY_TTL_MS,
});
return row.signature;
}
} catch {
// Fail-open
}
return null;
}
/**
* Synchronous get from RAM cache only (for sync translators)
*/
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) {
if (typeof toolCallId !== "string" || !toolCallId) return null;
pruneMemoryExpired();
if (sessionId && typeof sessionId === "string") {
const sessionKey = `${sessionId}:${toolCallId}`;
const sessionEntry = memorySignatures.get(sessionKey);
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
return sessionEntry.signature;
}
}
const entry = memorySignatures.get(toolCallId);
if (entry && entry.expiresAt > Date.now()) {
return entry.signature;
}
return null;
}

View File

@@ -4,6 +4,7 @@ import {
refreshXaiToken, refreshXaiToken,
refreshAccessToken, refreshAccessToken,
refreshKimiToken, refreshKimiToken,
refreshClineToken,
refreshClaudeOAuthToken, refreshClaudeOAuthToken,
refreshGoogleToken, refreshGoogleToken,
refreshCodexToken, refreshCodexToken,
@@ -23,6 +24,7 @@ import {
export { export {
refreshAccessToken, refreshAccessToken,
refreshKimiToken, refreshKimiToken,
refreshClineToken,
refreshClaudeOAuthToken, refreshClaudeOAuthToken,
refreshGoogleToken, refreshGoogleToken,
refreshCodexToken, refreshCodexToken,
@@ -145,6 +147,7 @@ const REFRESH_HANDLERS = {
"codebuddy-cn": (c, log) => refreshCodebuddyToken(c.refreshToken, log), "codebuddy-cn": (c, log) => refreshCodebuddyToken(c.refreshToken, log),
"codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log), "codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log),
trae: (c, log) => refreshTraeToken(c.refreshToken, c, log), trae: (c, log) => refreshTraeToken(c.refreshToken, c, log),
cline: (c, log) => refreshClineToken(c.refreshToken, log),
zed: () => refreshZedToken(), zed: () => refreshZedToken(),
windsurf: (c, log) => refreshWindsurfToken(c, log), windsurf: (c, log) => refreshWindsurfToken(c, log),
// Kimi Code OAuth (merged into id `kimi`); legacy id still routes here // Kimi Code OAuth (merged into id `kimi`); legacy id still routes here

View File

@@ -147,6 +147,53 @@ export async function refreshKimiToken(refreshToken, credentials, log) {
return refreshAccessToken("kimi", refreshToken, credentials, log); return refreshAccessToken("kimi", refreshToken, credentials, log);
} }
export async function refreshClineToken(refreshToken, log) {
if (!refreshToken) return null;
return dedupRefresh("cline", refreshToken, async () => {
try {
const response = await fetch(PROVIDERS.cline?.refreshUrl, {
method: "POST",
headers: {
"Content-Type": "application/json",
Accept: "application/json",
},
body: JSON.stringify({
refreshToken,
grantType: "refresh_token",
clientType: "extension",
}),
});
if (!response.ok) {
const errorText = await response.text();
log?.error?.("TOKEN_REFRESH", "Failed to refresh Cline token", {
status: response.status,
error: errorText,
});
return null;
}
const body = await response.json();
const tokens = body?.data || body;
if (!tokens?.accessToken) return null;
const expiresIn = tokens.expiresAt
? Math.max(1, Math.floor((new Date(tokens.expiresAt).getTime() - Date.now()) / 1000))
: (tokens.expiresIn || tokens.expires_in || 3600);
return {
accessToken: tokens.accessToken,
refreshToken: tokens.refreshToken || refreshToken,
expiresIn,
};
} catch (error) {
log?.error?.("TOKEN_REFRESH", `Error refreshing Cline token: ${error.message}`);
return null;
}
}, log);
}
// Claude OAuth: JSON body, client_id only. Delegate to refreshAccessToken("claude", ...). // Claude OAuth: JSON body, client_id only. Delegate to refreshAccessToken("claude", ...).
export async function refreshClaudeOAuthToken(refreshToken, log) { export async function refreshClaudeOAuthToken(refreshToken, log) {
return refreshAccessToken("claude", refreshToken, {}, log); return refreshAccessToken("claude", refreshToken, {}, log);

View File

@@ -11,14 +11,18 @@ export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
import { getKiroUsage } from "./usage/kiro.js"; import { getKiroUsage } from "./usage/kiro.js";
import { getMiniMaxUsage } from "./usage/minimax.js"; import { getMiniMaxUsage } from "./usage/minimax.js";
import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js"; import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js";
import { getXaiUsage } from "./usage/xai.js";
import { getGrokCliUsage } from "./usage/grok-cli.js"; import { getGrokCliUsage } from "./usage/grok-cli.js";
import { getKimiUsage } from "./usage/kimi.js"; import { getKimiUsage } from "./usage/kimi.js";
import { getDeepseekUsage } from "./usage/deepseek.js"; import { getDeepseekUsage } from "./usage/deepseek.js";
import { getOpenCodeGoUsage } from "./usage/opencode-go.js";
import { getGroqUsage } from "./usage/groq.js";
import { getZedUsage } from "./usage/zed.js";
import { resolveQoderCredentials } from "./qoderModels.js"; import { resolveQoderCredentials } from "./qoderModels.js";
import { getGlmUsage } from "./usage/glm.js";
import { import {
getIflowUsage, getIflowUsage,
getOllamaUsage, getOllamaUsage,
getGlmUsage,
getVercelAiGatewayUsage, getVercelAiGatewayUsage,
getQoderUsage, getQoderUsage,
} from "./usage/misc.js"; } from "./usage/misc.js";
@@ -50,10 +54,14 @@ const USAGE_HANDLERS = {
"minimax-cn": (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions), "minimax-cn": (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
"vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions), "vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions),
"codebuddy-cn": (c) => getCodeBuddyCnUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions), "codebuddy-cn": (c) => getCodeBuddyCnUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
xai: (c) => getXaiUsage(c.accessToken, c.proxyOptions),
"codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions), "codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), "grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData), kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData),
"opencode-go": (c) => getOpenCodeGoUsage(c.apiKey, c.proxyOptions),
deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions), deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions),
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
}; };
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) { export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {

View File

@@ -102,14 +102,34 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
quotas["weekly (7d)"] = createQuotaObject(data.seven_day); quotas["weekly (7d)"] = createQuotaObject(data.seven_day);
} }
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus) // Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus, seven_day_fable)
const MODEL_DISPLAY_NAMES = {
fable_5_1: "fable",
fable_5: "fable",
};
for (const [key, value] of Object.entries(data)) { for (const [key, value] of Object.entries(data)) {
if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) { if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) {
const modelName = key.replace("seven_day_", ""); const rawName = key.replace("seven_day_", "");
const modelName = MODEL_DISPLAY_NAMES[rawName] || rawName;
quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value); quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value);
} else if ((key === "fable" || key === "fable_5" || key === "fable_5_1") && hasUtilization(value)) {
quotas["weekly fable (7d)"] = createQuotaObject(value);
} }
} }
// Fallback: surface Fable quota row if weekly window exists but Fable was not returned yet
if (!quotas["weekly fable (7d)"] && hasUtilization(data.seven_day)) {
quotas["weekly fable (7d)"] = {
used: 0,
total: 100,
remaining: 100,
remainingPercentage: 100,
resetAt: parseResetTime(data.seven_day.resets_at),
unlimited: false,
};
}
return { return {
plan: "Claude Code", plan: "Claude Code",
extraUsage: data.extra_usage ?? null, extraUsage: data.extra_usage ?? null,

View File

@@ -21,6 +21,13 @@ function toIsoDate(value) {
return Number.isFinite(time) ? date.toISOString() : null; return Number.isFinite(time) ? date.toISOString() : null;
} }
function errorMessage(value, fallback) {
if (!value) return fallback;
if (typeof value === "string") return value;
if (typeof value.message === "string") return value.message;
return JSON.stringify(value);
}
function getCodexAccountId(providerSpecificData) { function getCodexAccountId(providerSpecificData) {
return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null; return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null;
} }
@@ -80,6 +87,23 @@ function getCodexReviewRateLimit(data) {
}) || null; }) || null;
} }
function getCodexSparkRateLimit(data) {
if (data.spark_rate_limit || data.gpt_5_3_codex_spark_rate_limit) {
return data.spark_rate_limit || data.gpt_5_3_codex_spark_rate_limit;
}
const byLimitId = data.rate_limits_by_limit_id;
if (byLimitId && typeof byLimitId === "object" && !Array.isArray(byLimitId)) {
return byLimitId["gpt-5.3-codex-spark"] || byLimitId.gpt_5_3_codex_spark || byLimitId.spark || null;
}
const additional = Array.isArray(data.additional_rate_limits) ? data.additional_rate_limits : [];
return additional.find((entry) => {
const id = String(entry?.limit_name || entry?.metered_feature || entry?.id || "").toLowerCase();
return id.includes("spark") || id.includes("5.3-codex-spark");
}) || null;
}
export async function getCodexUsage(accessToken, proxyOptions = null) { export async function getCodexUsage(accessToken, proxyOptions = null) {
try { try {
const response = await proxyAwareFetch(CODEX_CONFIG.usageUrl, { const response = await proxyAwareFetch(CODEX_CONFIG.usageUrl, {
@@ -97,16 +121,19 @@ export async function getCodexUsage(accessToken, proxyOptions = null) {
const data = await response.json(); const data = await response.json();
const normalRateLimit = data.rate_limit || data.rate_limits || data.rate_limits_by_limit_id?.codex || {}; const normalRateLimit = data.rate_limit || data.rate_limits || data.rate_limits_by_limit_id?.codex || {};
const reviewRateLimit = getCodexReviewRateLimit(data); const reviewRateLimit = getCodexReviewRateLimit(data);
const sparkRateLimit = getCodexSparkRateLimit(data);
const availableResetCredits = Math.max(0, toFiniteNumber(data.rate_limit_reset_credits?.available_count, 0)); const availableResetCredits = Math.max(0, toFiniteNumber(data.rate_limit_reset_credits?.available_count, 0));
const quotas = {}; const quotas = {};
appendCodexQuotaWindows(quotas, "", normalRateLimit); appendCodexQuotaWindows(quotas, "", normalRateLimit);
appendCodexQuotaWindows(quotas, "review", reviewRateLimit); appendCodexQuotaWindows(quotas, "review", reviewRateLimit);
appendCodexQuotaWindows(quotas, "spark", sparkRateLimit);
return { return {
plan: data.plan_type || data.summary?.plan || "unknown", plan: data.plan_type || data.summary?.plan || "unknown",
limitReached: getCodexRateLimitBody(normalRateLimit)?.limit_reached || false, limitReached: getCodexRateLimitBody(normalRateLimit)?.limit_reached || false,
reviewLimitReached: getCodexRateLimitBody(reviewRateLimit)?.limit_reached || false, reviewLimitReached: getCodexRateLimitBody(reviewRateLimit)?.limit_reached || false,
sparkLimitReached: getCodexRateLimitBody(sparkRateLimit)?.limit_reached || false,
resetCredits: { availableCount: availableResetCredits }, resetCredits: { availableCount: availableResetCredits },
quotas, quotas,
}; };
@@ -142,7 +169,7 @@ export async function getCodexRateLimitResetCredits(accessToken, proxyOptions =
} }
if (!response.ok) { if (!response.ok) {
const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`; const message = errorMessage(data?.message || data?.error || data?.detail, `Codex reset credits API unavailable (${response.status}).`);
throw new Error(message); throw new Error(message);
} }

View File

@@ -0,0 +1,207 @@
/**
* CommandCode usage handler
*
* Mirrors the official command-code CLI /usage command: it calls the alpha API
* to surface the 5-hour + weekly usage windows, the subscription plan, and the
* credits consumed in the current billing period.
*
* GET /alpha/whoami → org.id (org-scoped billing; null for personal)
* GET /alpha/billing/credits → { credits: { monthlyCredits, purchasedCredits,
* freeCredits }, windowLimits: { fiveHour, weekly } }
* GET /alpha/billing/subscriptions → { data: { planId, currentPeriodStart, ... } }
* GET /alpha/usage/summary?since= → period token/cost totals
*
* The CLI fetches whoami first (for orgId), then credits + subscription in
* parallel, then the summary with since = currentPeriodStart. We keep the same
* order/dependencies: window limits live on credits, and the plan period start
* determines the summary window.
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U, parseResetTime } from "./shared.js";
const USAGE = U("commandcode");
const BASE = USAGE.baseUrl || "https://api.commandcode.ai";
const WHOAMI_URL = BASE + (USAGE.whoamiUrl || "/alpha/whoami");
const CREDITS_URL = BASE + (USAGE.creditsUrl || "/alpha/billing/credits");
const SUBSCRIPTIONS_URL =
BASE + (USAGE.subscriptionsUrl || "/alpha/billing/subscriptions");
const SUMMARY_URL = BASE + (USAGE.summaryUrl || "/alpha/usage/summary");
function buildHeaders(token) {
return {
Authorization: `Bearer ${token}`,
Accept: "application/json",
};
}
/** Build a normalized quota row. `unit` is "$" — the API reports currency credits. */
function makeQuota({ used, total, resetAt, unlimited = false, unit = "$" }) {
const safeTotal = Math.max(0, Number(total) || 0);
const safeUsed = Math.max(0, Number(used) || 0);
if (unlimited || safeTotal === 0) {
return {
used: safeUsed,
total: 0,
remainingPercentage: unlimited ? 100 : 0,
resetAt: resetAt || null,
unit,
unlimited: true,
};
}
const remaining = Math.max(0, safeTotal - safeUsed);
const remainingPercentage = (remaining / safeTotal) * 100;
return {
used: safeUsed,
total: safeTotal,
remainingPercentage,
resetAt: resetAt || null,
unit,
unlimited: false,
};
}
/**
* @param {string} apiKey - commandcode API key (user_...)
* @param {object|null} proxyOptions
*/
export async function getCommandCodeUsage(apiKey, proxyOptions = null) {
if (!apiKey) {
return { message: "CommandCode credential not available." };
}
const headers = buildHeaders(apiKey);
try {
// whoami resolves the org id (billing is org-scoped; null for personal).
const whoamiRes = await proxyAwareFetch(
WHOAMI_URL,
{ method: "GET", headers },
proxyOptions,
);
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
return { message: "CommandCode credential invalid or expired." };
}
if (!whoamiRes.ok) {
return { message: `CommandCode whoami API error (${whoamiRes.status}).` };
}
const whoami = await whoamiRes.json().catch(() => null);
const orgId = whoami?.org?.id ?? null;
const orgQuery = orgId ? `?orgId=${encodeURIComponent(orgId)}` : "";
const [creditsRes, subsRes] = await Promise.all([
proxyAwareFetch(
CREDITS_URL + orgQuery,
{ method: "GET", headers },
proxyOptions,
),
proxyAwareFetch(
SUBSCRIPTIONS_URL + orgQuery,
{ method: "GET", headers },
proxyOptions,
),
]);
if (
creditsRes.status === 401 ||
creditsRes.status === 403 ||
subsRes.status === 401 ||
subsRes.status === 403
) {
return { message: "CommandCode credential invalid or expired." };
}
if (!creditsRes.ok) {
return {
message: `CommandCode credits API error (${creditsRes.status}).`,
};
}
const credits = await creditsRes.json().catch(() => null);
const subs = await subsRes.json().catch(() => null);
const subData = subs?.data;
const planId = subData?.planId ?? null;
const periodStart = subData?.currentPeriodStart ?? null;
// Summary needs `since`; the CLI falls back to first-of-month when the
// subscription period start is unavailable.
const since = periodStart || firstOfMonth();
const summaryRes = await proxyAwareFetch(
`${SUMMARY_URL}?since=${encodeURIComponent(since)}`,
{ method: "GET", headers },
proxyOptions,
);
const summary = summaryRes.ok
? await summaryRes.json().catch(() => null)
: null;
const quotas = {};
const windowLimits = credits?.windowLimits || {};
const fiveHour = windowLimits.fiveHour;
if (fiveHour && Number(fiveHour.cap) > 0) {
quotas["5-hour window"] = makeQuota({
used: fiveHour.used,
total: fiveHour.cap,
resetAt: parseResetTime(fiveHour.resetAt),
});
}
const weekly = windowLimits.weekly;
if (weekly && Number(weekly.cap) > 0) {
quotas["Weekly window"] = makeQuota({
used: weekly.used,
total: weekly.cap,
resetAt: parseResetTime(weekly.resetAt),
});
}
// The credits API reports remaining balances (monthly/purchased/free),
// not a total. The official CLI renders the monthly line as
// `used = summary.totalCost`, `total = totalCost + remaining` — i.e.
// the plan ceiling is the sum of what was consumed and what is left.
const monthlyUsed =
typeof summary?.totalCredits === "number"
? summary.totalCredits
: typeof summary?.totalCost === "number"
? summary.totalCost
: 0;
const creditsObj = credits?.credits || {};
const remaining =
Math.max(0, Number(creditsObj.monthlyCredits) || 0) +
Math.max(0, Number(creditsObj.purchasedCredits) || 0) +
Math.max(0, Number(creditsObj.freeCredits) || 0);
const monthlyTotal = monthlyUsed + remaining;
if (monthlyTotal > 0 || monthlyUsed > 0) {
quotas["Monthly credits"] = makeQuota({
used: monthlyUsed,
total: monthlyTotal,
resetAt: periodStart ? undefined : null,
});
}
if (Object.keys(quotas).length === 0) {
return {
plan: planId || "CommandCode",
message: "CommandCode connected, but no quota was reported.",
quotas: {},
};
}
return {
plan: planId || "CommandCode",
quotas,
periodBasis: summary?.periodBasis || "billing-period",
};
} catch (error) {
return { message: `CommandCode usage error: ${error.message}` };
}
}
function firstOfMonth() {
const now = new Date();
return new Date(now.getFullYear(), now.getMonth(), 1).toISOString();
}

View File

@@ -0,0 +1,88 @@
/**
* GLM Coding Plan usage (international + China regions)
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U } from "./shared.js";
// GLM quota endpoints (region-aware) — url from registry transport.usage
const GLM_QUOTA_URLS = {
international: U("glm").url,
china: U("glm-cn").url,
};
/**
* GLM Coding Plan usage (international + China regions)
* Supports both TOKENS_LIMIT and CREDIT_LIMIT and dynamic intervals (e.g. session 5h, weekly 7d).
*/
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
if (!apiKey) {
return { message: "GLM API key not available." };
}
const region = provider === "glm-cn" ? "china" : "international";
const quotaUrl = GLM_QUOTA_URLS[region];
try {
const response = await proxyAwareFetch(
quotaUrl,
{
headers: {
Authorization: `Bearer ${apiKey}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (!response.ok) {
if (response.status === 401) {
return { message: "GLM API key invalid or expired." };
}
return { message: `GLM quota API error (${response.status}).` };
}
const json = await response.json();
const data = json?.data && typeof json.data === "object" ? json.data : {};
const limits = Array.isArray(data.limits) ? data.limits : [];
const quotas = {};
for (const limit of limits) {
// 1. Accept both TOKENS_LIMIT and CREDIT_LIMIT from GLM API
if (!limit || (limit.type !== "TOKENS_LIMIT" && limit.type !== "CREDIT_LIMIT")) continue;
const usedPercent = Number(limit.percentage) || 0;
const resetMs = Number(limit.nextResetTime) || 0;
const remaining = Math.max(0, 100 - usedPercent);
// 2. Map key dynamically based on type and period (unit) to avoid overwriting
let key = "session";
if (limit.unit === 3) {
key = `Session (${limit.number}h)`;
} else if (limit.unit === 6) {
key = "Weekly (7d)";
} else if (limit.type === "TOKENS_LIMIT") {
key = "Tokens";
} else {
key = `Limit (${limit.number})`;
}
quotas[key] = {
used: usedPercent,
total: 100,
remaining,
remainingPercentage: remaining,
resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null,
unlimited: false,
};
}
const levelRaw = typeof data.level === "string" ? data.level : "";
const plan = levelRaw
? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase()
: "Unknown";
return { plan, quotas };
} catch (error) {
return { message: `GLM error: ${error.message}` };
}
}

View File

@@ -161,6 +161,9 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
if (data.models) { if (data.models) {
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids) // Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
const importantModels = [ const importantModels = [
'gemini-3.8-flash-high',
'gemini-3.8-flash-medium',
'gemini-3.8-flash-low',
'gemini-3.7-flash-high', 'gemini-3.7-flash-high',
'gemini-3.7-flash-medium', 'gemini-3.7-flash-medium',
'gemini-3.7-flash-low', 'gemini-3.7-flash-low',

View File

@@ -0,0 +1,133 @@
/**
* Groq usage — no dedicated quota endpoint. Rate-limit info instead rides on
* every API response as x-ratelimit-* headers (requests + tokens, always
* included). We piggyback on the models list (already used as
* transport.validateUrl) so reading usage never costs tokens.
*
* Headers:
* x-ratelimit-limit-requests / x-ratelimit-remaining-requests
* x-ratelimit-limit-tokens / x-ratelimit-remaining-tokens
* x-ratelimit-reset-requests / x-ratelimit-reset-tokens (duration strings, e.g. "2m59.56s")
*
* Docs: https://console.groq.com/docs/rate-limits
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U } from "./shared.js";
const MODELS_URL = U("groq").url;
// Groq reset headers are Go-style duration strings ("2m59.56s", "7.66s"), not
// timestamps — parse the h/m/s/ms components and add them to now().
function parseGroqDurationMs(value) {
if (typeof value !== "string" || !value.trim()) return null;
const re = /(\d+(?:\.\d+)?)(ms|s|m|h)/g;
let match;
let totalMs = 0;
let matched = false;
while ((match = re.exec(value))) {
matched = true;
const amount = Number(match[1]);
const unit = match[2];
const unitMs = unit === "h" ? 3600000 : unit === "m" ? 60000 : unit === "ms" ? 1 : 1000;
totalMs += amount * unitMs;
}
return matched ? totalMs : null;
}
function resetAtFromDuration(value) {
const ms = parseGroqDurationMs(value);
return ms === null ? null : new Date(Date.now() + ms).toISOString();
}
function buildRateLimitQuota(headers, limitKey, remainingKey, resetKey) {
// headers.get() returns null when absent, and Number(null) is 0 (a finite
// number) — check presence explicitly so a missing header can't masquerade
// as a real "0 remaining" quota.
const limitRaw = headers.get(limitKey);
const remainingRaw = headers.get(remainingKey);
if (limitRaw === null || remainingRaw === null) return null;
const limit = Number(limitRaw);
const remaining = Number(remainingRaw);
if (!Number.isFinite(limit) || !Number.isFinite(remaining)) return null;
return {
used: Math.max(0, limit - remaining),
total: limit,
resetAt: resetAtFromDuration(headers.get(resetKey)),
unlimited: false,
};
}
/**
* @param {string|null|undefined} apiKey
* @param {object|null} proxyOptions
*/
export async function getGroqUsage(apiKey, proxyOptions = null) {
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
return { message: "Groq API key not available. Add a key to view usage." };
}
try {
const response = await proxyAwareFetch(
MODELS_URL,
{
method: "GET",
headers: {
Authorization: `Bearer ${apiKey.trim()}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (response.status === 401 || response.status === 403) {
return { plan: "Groq", message: "Groq authentication failed. Check the API key." };
}
if (!response.ok) {
const errText = await response.text().catch(() => "");
return {
plan: "Groq",
message: `Groq usage API error (${response.status})${errText ? `: ${errText.slice(0, 120)}` : ""}`,
};
}
// The quota data lives in headers, not the body — drain it so the
// connection can be released without needing the payload.
await response.text().catch(() => {});
const requests = buildRateLimitQuota(
response.headers,
"x-ratelimit-limit-requests",
"x-ratelimit-remaining-requests",
"x-ratelimit-reset-requests",
);
const tokens = buildRateLimitQuota(
response.headers,
"x-ratelimit-limit-tokens",
"x-ratelimit-remaining-tokens",
"x-ratelimit-reset-tokens",
);
if (!requests && !tokens) {
// Key is valid (request succeeded) but no rate-limit bucket reported —
// distinguish "not tracked yet" from an auth/error state.
return {
plan: "Groq",
message: "Groq connected. No rate-limit data reported for this key yet.",
quotas: {},
};
}
const quotas = {};
if (requests) quotas["Requests"] = requests;
if (tokens) quotas["Tokens"] = tokens;
return { plan: "Groq", quotas };
} catch (error) {
return { message: `Groq error: ${error.message}` };
}
}

View File

@@ -5,11 +5,8 @@
import { proxyAwareFetch } from "../../utils/proxyFetch.js"; import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U } from "./shared.js"; import { U } from "./shared.js";
// GLM quota endpoints (region-aware) — url from registry transport.usage export { getGlmUsage } from "./glm.js";
const GLM_QUOTA_URLS = {
international: U("glm").url,
china: U("glm-cn").url,
};
// Vercel AI Gateway credits endpoint // Vercel AI Gateway credits endpoint
// Returns { balance: "95.50", total_used: "4.50" } (USD as decimal strings). // Returns { balance: "95.50", total_used: "4.50" } (USD as decimal strings).
@@ -112,63 +109,7 @@ export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions
} }
} }
/**
* GLM Coding Plan usage (international + China regions)
*/
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
if (!apiKey) {
return { message: "GLM API key not available." };
}
const region = provider === "glm-cn" ? "china" : "international";
const quotaUrl = GLM_QUOTA_URLS[region];
try {
const response = await proxyAwareFetch(quotaUrl, {
headers: {
Authorization: `Bearer ${apiKey}`,
Accept: "application/json",
},
}, proxyOptions);
if (!response.ok) {
if (response.status === 401) {
return { message: "GLM API key invalid or expired." };
}
return { message: `GLM quota API error (${response.status}).` };
}
const json = await response.json();
const data = json?.data && typeof json.data === "object" ? json.data : {};
const limits = Array.isArray(data.limits) ? data.limits : [];
const quotas = {};
for (const limit of limits) {
if (!limit || limit.type !== "TOKENS_LIMIT") continue;
const usedPercent = Number(limit.percentage) || 0;
const resetMs = Number(limit.nextResetTime) || 0;
const remaining = Math.max(0, 100 - usedPercent);
quotas["session"] = {
used: usedPercent,
total: 100,
remaining,
remainingPercentage: remaining,
resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null,
unlimited: false,
};
}
const levelRaw = typeof data.level === "string" ? data.level : "";
const plan = levelRaw
? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase()
: "Unknown";
return { plan, quotas };
} catch (error) {
return { message: `GLM error: ${error.message}` };
}
}
/** /**
* Vercel AI Gateway usage — credit balance for the API key * Vercel AI Gateway usage — credit balance for the API key

View File

@@ -0,0 +1,107 @@
/**
* OpenCode Go usage — GET https://opencode.ai/zen/go/v1/usage
* Auth: Bearer <apiKey>
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { parseResetTime, toFiniteNumber, U } from "./shared.js";
const USAGE_URL = U("opencode-go").url;
const QUOTA_NAMES = {
rolling: "Rolling",
weekly: "Weekly",
monthly: "Monthly",
};
function parsePercent(value) {
if (typeof value === "number" && Number.isFinite(value)) return value;
if (typeof value === "string" && value.trim()) {
const parsed = Number(value);
if (Number.isFinite(parsed)) return parsed;
}
return null;
}
export async function getOpenCodeGoUsage(apiKey = null, proxyOptions = null) {
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
return {
message: "OpenCode Go API key not available. Add a key to view usage.",
};
}
try {
const response = await proxyAwareFetch(
USAGE_URL,
{
method: "GET",
headers: {
Authorization: `Bearer ${apiKey.trim()}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (response.status === 401) {
return {
plan: "OpenCode Go",
message: "OpenCode Go authentication failed. Check the API key.",
};
}
if (response.status === 403) {
const error = await response.json().catch(() => null);
const subscriptionRequired = error?.error?.type === "EntitlementError";
return {
plan: "OpenCode Go",
message: subscriptionRequired
? "OpenCode Go subscription required for this API key."
: "OpenCode Go access forbidden for this API key.",
};
}
if (!response.ok) {
return {
plan: "OpenCode Go",
message: `OpenCode Go usage API error (${response.status}).`,
};
}
const data = await response.json().catch(() => null);
if (!data?.usage || typeof data.usage !== "object") {
return {
plan: "OpenCode Go",
message: "OpenCode Go usage response did not contain quota data.",
};
}
const quotas = {};
for (const [period, name] of Object.entries(QUOTA_NAMES)) {
const quota = data.usage[period];
if (!quota || typeof quota !== "object") continue;
const percent = parsePercent(quota.percent);
if (percent === null) continue;
const used = Math.max(0, Math.min(100, toFiniteNumber(percent, 0)));
quotas[name] = {
used,
total: 100,
remaining: 100 - used,
remainingPercentage: 100 - used,
resetAt: parseResetTime(quota.resetsAt),
unlimited: false,
};
}
if (Object.keys(quotas).length === 0) {
return {
plan: "OpenCode Go",
message: "OpenCode Go usage response did not contain valid quota data.",
};
}
return { plan: "OpenCode Go", quotas };
} catch (error) {
return { message: `OpenCode Go error: ${error.message}` };
}
}

View File

@@ -0,0 +1,326 @@
/**
* xAI (Grok) OAuth usage handler
*
* SuperGrok quota is split across two upstream surfaces (OAuth only):
*
* 1) Monthly / API usage allotment (JSON)
* GET https://cli-chat-proxy.grok.com/v1/billing
* {
* "config": {
* "monthlyLimit": { "val": 15000 },
* "used": { "val": 733 },
* "onDemandCap": { "val": 0 },
* "billingPeriodStart": "...",
* "billingPeriodEnd": "..."
* }
* }
*
* 2) Weekly SuperGrok limit (grpc-web protobuf)
* POST https://grok.com/grok_api_v2.GrokBuildBilling/GetGrokCreditsConfig
* Empty request frame; response message1 contains:
* usedPercent (float32), window start/end timestamps, nested windows.
* This is what the grok.com usage page labels "Weekly limit" / "Resets …".
*
* Plan label comes from cli-chat-proxy settings:
* GET https://cli-chat-proxy.grok.com/v1/settings → subscription_tier_display
*
* Note: grok.com/rest/rate-limits is a short chat-window (e.g. 2h query count)
* behind Cloudflare browser cookies — not usable with pure OAuth bearer.
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { U, parseResetTime, toFiniteNumber } from "./shared.js";
// Empty grpc-web request frame: flag(0) + length(0) + no payload.
const GRPC_WEB_EMPTY_FRAME = Buffer.from([0x00, 0x00, 0x00, 0x00, 0x00]);
function moneyVal(wrapper) {
if (wrapper == null) return null;
if (typeof wrapper === "number") return toFiniteNumber(wrapper, null);
if (typeof wrapper === "object" && wrapper.val != null) {
return toFiniteNumber(wrapper.val, null);
}
return null;
}
function authHeaders(accessToken, extra = {}) {
return {
Authorization: `Bearer ${accessToken}`,
...extra,
};
}
function readVarint(buf, offset) {
let val = 0;
let shift = 0;
let i = offset;
while (i < buf.length) {
const b = buf[i++];
val |= (b & 0x7f) << shift;
if ((b & 0x80) === 0) return { value: val >>> 0, offset: i };
shift += 7;
if (shift > 35) break;
}
return null;
}
/**
* Minimal protobuf decoder for GetGrokCreditsConfig.
* Only understands varint / fixed32 / fixed64 / length-delimited.
*/
function parseProtobufFields(buf) {
const fields = [];
let i = 0;
while (i < buf.length) {
const key = readVarint(buf, i);
if (!key) break;
i = key.offset;
const field = key.value >>> 3;
const wt = key.value & 7;
if (wt === 0) {
const v = readVarint(buf, i);
if (!v) break;
i = v.offset;
fields.push({ field, type: "varint", value: v.value });
} else if (wt === 1) {
if (i + 8 > buf.length) break;
fields.push({ field, type: "fixed64", value: buf.subarray(i, i + 8) });
i += 8;
} else if (wt === 5) {
if (i + 4 > buf.length) break;
fields.push({
field,
type: "fixed32",
value: buf.readFloatLE(i),
});
i += 4;
} else if (wt === 2) {
const ln = readVarint(buf, i);
if (!ln) break;
i = ln.offset;
if (i + ln.value > buf.length) break;
fields.push({
field,
type: "bytes",
value: buf.subarray(i, i + ln.value),
});
i += ln.value;
} else {
break;
}
}
return fields;
}
function parseTimestamp(bytes) {
if (!bytes || !bytes.length) return null;
const fields = parseProtobufFields(bytes);
const seconds = fields.find((f) => f.field === 1 && f.type === "varint")?.value;
if (!Number.isFinite(seconds) || seconds <= 0) return null;
return new Date(seconds * 1000).toISOString();
}
/**
* Parse grpc-web response bytes from GetGrokCreditsConfig.
* Returns { usedPercent, resetAt, periodStart } or null.
*/
export function parseGrokCreditsConfig(raw) {
if (!raw || !raw.length) return null;
const buf = Buffer.isBuffer(raw) ? raw : Buffer.from(raw);
// grpc-web data frame: 1-byte flag + 4-byte big-endian length + message
if (buf.length < 5) return null;
const flag = buf[0];
// Data frames have flag 0; ignore trailer frames (flag 0x80).
if (flag !== 0) return null;
const msgLen = buf.readUInt32BE(1);
if (msgLen <= 0 || 5 + msgLen > buf.length) return null;
const msg = buf.subarray(5, 5 + msgLen);
// Response is typically { 1: CreditsConfig }
const top = parseProtobufFields(msg);
const configBytes = top.find((f) => f.field === 1 && f.type === "bytes")?.value || msg;
const fields = parseProtobufFields(configBytes);
const usedPercentRaw = fields.find((f) => f.field === 1 && f.type === "fixed32")?.value;
const periodStart = parseTimestamp(fields.find((f) => f.field === 4 && f.type === "bytes")?.value);
const periodEnd = parseTimestamp(fields.find((f) => f.field === 5 && f.type === "bytes")?.value);
if (usedPercentRaw == null || !Number.isFinite(usedPercentRaw)) return null;
const usedPercent = Math.max(0, Math.min(100, usedPercentRaw));
return {
usedPercent,
remainingPercent: Math.max(0, 100 - usedPercent),
periodStart,
resetAt: periodEnd,
};
}
async function fetchBilling(accessToken, billingUrl, proxyOptions) {
const response = await proxyAwareFetch(billingUrl, {
method: "GET",
headers: authHeaders(accessToken, { Accept: "application/json" }),
}, proxyOptions);
if (response.status === 401 || response.status === 403) {
return { error: "auth", status: response.status };
}
if (!response.ok) {
return { error: "http", status: response.status };
}
const data = await response.json().catch(() => null);
if (!data || typeof data !== "object") {
return { error: "json" };
}
return { data };
}
async function fetchWeeklyCredits(accessToken, creditsUrl, proxyOptions) {
if (!creditsUrl) return null;
try {
const response = await proxyAwareFetch(creditsUrl, {
method: "POST",
headers: authHeaders(accessToken, {
"Content-Type": "application/grpc-web+proto",
"x-grpc-web": "1",
"x-user-agent": "connect-es/2.1.1",
Accept: "*/*",
Origin: "https://grok.com",
Referer: "https://grok.com/?_s=usage",
}),
body: GRPC_WEB_EMPTY_FRAME,
}, proxyOptions);
if (!response.ok) return null;
const ab = await response.arrayBuffer();
return parseGrokCreditsConfig(Buffer.from(ab));
} catch {
return null;
}
}
async function fetchPlanLabel(accessToken, settingsUrl, proxyOptions) {
if (!settingsUrl) return null;
try {
const response = await proxyAwareFetch(settingsUrl, {
method: "GET",
headers: authHeaders(accessToken, { Accept: "application/json" }),
}, proxyOptions);
if (!response.ok) return null;
const data = await response.json().catch(() => null);
const label = data?.subscription_tier_display;
return typeof label === "string" && label.trim() ? label.trim() : null;
} catch {
return null;
}
}
/**
* @param {string} accessToken - xAI OAuth access token
* @param {object|null} proxyOptions
*/
export async function getXaiUsage(accessToken, proxyOptions = null) {
if (!accessToken) {
return { message: "xAI usage unavailable: no access token. Re-authorize the connection." };
}
const cfg = U("xai") || {};
const billingUrl = cfg.url;
if (!billingUrl) {
return { message: "xAI usage endpoint is not configured." };
}
try {
const [billingResult, weekly, planLabel] = await Promise.all([
fetchBilling(accessToken, billingUrl, proxyOptions),
fetchWeeklyCredits(accessToken, cfg.creditsUrl, proxyOptions),
fetchPlanLabel(accessToken, cfg.settingsUrl, proxyOptions),
]);
if (billingResult.error === "auth") {
return { message: "xAI OAuth token expired or unauthorized. Please re-authorize." };
}
const quotas = {};
let periodStart = null;
let periodEnd = null;
let onDemandCap = 0;
if (billingResult.data) {
const config =
billingResult.data.config && typeof billingResult.data.config === "object"
? billingResult.data.config
: billingResult.data;
const monthlyLimit = moneyVal(config.monthlyLimit);
const used = moneyVal(config.used);
onDemandCap = moneyVal(config.onDemandCap) ?? 0;
periodEnd = parseResetTime(config.billingPeriodEnd);
periodStart = parseResetTime(config.billingPeriodStart);
// Absolute credit counts — do NOT put remaining credits on `remaining`
// (QuotaTable treats remaining as a 0-100 percentage; same pitfall as Qoder).
if (monthlyLimit != null && monthlyLimit > 0) {
const usedSafe = Math.max(0, used ?? 0);
quotas.api_usage = {
used: usedSafe,
total: monthlyLimit,
remainingCredits: Math.max(0, monthlyLimit - usedSafe),
unit: "credits",
resetAt: periodEnd,
unlimited: false,
};
}
if (onDemandCap > 0) {
quotas.on_demand = {
used: 0,
total: onDemandCap,
remainingCredits: onDemandCap,
unit: "credits",
resetAt: periodEnd,
unlimited: false,
};
}
}
// Weekly SuperGrok window — percentage-based like Claude/Codex windows.
if (weekly) {
quotas.weekly = {
used: weekly.usedPercent,
total: 100,
remaining: weekly.remainingPercent,
remainingPercentage: weekly.remainingPercent,
resetAt: weekly.resetAt || null,
unlimited: false,
};
if (!periodStart && weekly.periodStart) periodStart = weekly.periodStart;
}
if (Object.keys(quotas).length === 0) {
const statusHint =
billingResult.error === "http"
? ` Billing API temporarily unavailable (${billingResult.status}).`
: "";
return {
plan: planLabel || "xAI",
message: `xAI connected. No quota allotment reported for this account.${statusHint}`,
periodStart,
periodEnd,
quotas: {},
};
}
return {
plan: planLabel || "xAI",
periodStart,
periodEnd,
onDemandCap,
quotas,
};
} catch (error) {
return { message: `xAI connected. Unable to fetch billing: ${error.message}` };
}
}

View File

@@ -0,0 +1,222 @@
/**
* Zed usage — GET https://cloud.zed.dev/client/users/me
* Auth: Authorization: {user_id} {access_token}
*
* Quota rows are derived from plan.usage (edit_predictions, optional model_requests)
* and subscription_period.ended_at for billing-cycle reset.
*/
import { fetchZedAuthenticatedUser } from "../../shared/zedAuth.js";
import { parseResetTime, toFiniteNumber } from "./shared.js";
/** Map plan_v3 ids to dashboard labels (CodexBar-compatible). */
export function formatZedPlanLabel(rawPlan) {
const raw = String(rawPlan || "").trim();
if (!raw) return "Zed";
switch (raw.toLowerCase()) {
case "zed_free":
return "Zed Free";
case "zed_pro":
return "Zed Pro";
case "zed_pro_trial":
return "Zed Pro Trial";
case "zed_student":
return "Zed Student";
case "zed_business":
return "Zed Business";
default:
return raw
.replace(/_/g, " ")
.split(/\s+/)
.map((word) => word.charAt(0).toUpperCase() + word.slice(1).toLowerCase())
.join(" ");
}
}
/**
* Parse Zed UsageLimit JSON: "unlimited", a number, or { limited: N }.
*/
export function parseZedUsageLimit(limit) {
if (limit == null) return { unlimited: false, total: 0 };
if (limit === "unlimited" || limit?.unlimited === true) {
return { unlimited: true, total: 0 };
}
if (typeof limit === "number" && Number.isFinite(limit)) {
return { unlimited: false, total: Math.max(0, limit) };
}
if (typeof limit === "string") {
const trimmed = limit.trim();
if (trimmed === "unlimited") return { unlimited: true, total: 0 };
const parsed = Number(trimmed);
if (Number.isFinite(parsed)) return { unlimited: false, total: Math.max(0, parsed) };
}
const limited = limit.limited ?? limit.Limited;
if (typeof limited === "number" && Number.isFinite(limited)) {
return { unlimited: false, total: Math.max(0, limited) };
}
return { unlimited: false, total: 0 };
}
/** limit `{ limited: 0 }` on Pro/Student means token billing, not a 0-cap request quota. */
export function isZedTokenBillingModelRequestsLimit(limitRaw) {
const info = parseZedUsageLimit(limitRaw);
return !info.unlimited && info.total === 0;
}
function makeZedQuotaRow(name, usedRaw, limitRaw, resetAt = null) {
const used = Math.max(0, toFiniteNumber(usedRaw, 0));
const limitInfo = parseZedUsageLimit(limitRaw);
if (limitInfo.unlimited) {
return {
used,
total: 0,
remainingPercentage: 100,
resetAt: resetAt || null,
unlimited: true,
};
}
const total = limitInfo.total;
if (total <= 0) {
return {
used,
total: 0,
remainingPercentage: 0,
resetAt: resetAt || null,
unlimited: false,
};
}
const clampedUsed = Math.min(used, total);
const remaining = Math.max(0, total - clampedUsed);
return {
used: clampedUsed,
total,
remainingPercentage: (remaining / total) * 100,
resetAt: resetAt || null,
unlimited: false,
};
}
function usageBucketLimit(bucket) {
if (!bucket || typeof bucket !== "object") return null;
if (bucket.limit != null) return bucket.limit;
return bucket;
}
/**
* Map /client/users/me JSON → { plan, quotas, message } for the dashboard.
*/
export function parseZedAuthenticatedUserUsage(userInfo) {
const plan = userInfo?.plan || {};
const planId =
plan.plan_v3 || plan.plan_v2 || plan.plan || userInfo?.plan_v3 || null;
const resetAt =
parseResetTime(plan.subscription_period?.ended_at) ||
parseResetTime(plan.subscriptionPeriod?.endedAt) ||
null;
const quotas = {};
const usage = plan.usage || {};
const editPredictions = usage.edit_predictions || usage.editPredictions;
if (editPredictions) {
quotas["Edit Predictions"] = makeZedQuotaRow(
"Edit Predictions",
editPredictions.used,
editPredictions.limit,
resetAt,
);
}
const modelRequests = usage.model_requests || usage.modelRequests;
if (modelRequests) {
const limitRaw =
modelRequests.limit != null
? modelRequests.limit
: usageBucketLimit(modelRequests)?.limit;
const limitInfo = parseZedUsageLimit(limitRaw);
// Token-billed plans report model_requests.limit=0 — not a request quota.
if (limitInfo.unlimited || limitInfo.total > 0) {
quotas["Hosted Model Requests"] = makeZedQuotaRow(
"Hosted Model Requests",
modelRequests.used,
limitRaw,
resetAt,
);
}
}
const tokenBillingNote =
modelRequests &&
isZedTokenBillingModelRequestsLimit(
modelRequests.limit ?? usageBucketLimit(modelRequests)?.limit,
)
? "Hosted AI models are billed per token (not request count). Edit Predictions are tracked below. Token spend is on dashboard.zed.dev."
: null;
let planLabel = formatZedPlanLabel(planId);
if (plan.trial_started_at || plan.trialStartedAt) {
if (!/trial/i.test(planLabel)) planLabel = `${planLabel} (Trial active)`;
}
let message = tokenBillingNote;
if (plan.has_overdue_invoices || plan.hasOverdueInvoices) {
message = "This Zed account has overdue invoices. Usage may be blocked until billing is resolved.";
}
return {
plan: planLabel,
quotas,
message,
hasOverdueInvoices: !!(plan.has_overdue_invoices || plan.hasOverdueInvoices),
trialStarted: !!(plan.trial_started_at || plan.trialStartedAt),
planId: planId || null,
resetAt,
};
}
/**
* @param {string|null|undefined} accessToken
* @param {object|null|undefined} providerSpecificData
* @param {object|null|undefined} proxyOptions
*/
export async function getZedUsage(
accessToken = null,
providerSpecificData = {},
proxyOptions = null,
) {
const psd = providerSpecificData || {};
const userId = psd.userId;
if (!accessToken || typeof accessToken !== "string" || !accessToken.trim()) {
return { message: "Zed access token not available. Re-connect Zed to view quota." };
}
if (!userId) {
return { message: "Zed credential is missing user id. Re-connect Zed to view quota." };
}
const credentials = {
accessToken: accessToken.trim(),
providerSpecificData: psd,
};
try {
const userInfo = await fetchZedAuthenticatedUser(credentials, { proxyOptions });
return parseZedAuthenticatedUserUsage(userInfo);
} catch (error) {
const status = error?.status;
if (status === 401 || status === 403) {
return {
message: "Zed authentication failed. Sign in again from the dashboard or Zed editor.",
};
}
return { message: `Zed error: ${error.message || "Failed to fetch quota"}` };
}
}

View File

@@ -54,10 +54,13 @@ export const QODER_MODEL_MAP = {
lite: "lite", lite: "lite",
// Frontier models // Frontier models
qmodel: "qmodel", qmodel: "qmodel",
qfmodel: "qfmodel",
qmodel_latest: "qmodel_latest", qmodel_latest: "qmodel_latest",
qmodel_38max: "qmodel_38max",
dmodel: "dmodel", dmodel: "dmodel",
dfmodel: "dfmodel", dfmodel: "dfmodel",
gm51model: "gm51model", gmodel: "gmodel",
gfmodel: "gfmodel",
kmodel: "kmodel", kmodel: "kmodel",
mmodel: "mmodel", mmodel: "mmodel",
}; };

View File

@@ -172,8 +172,8 @@ function getSystemId(credentials) {
); );
} }
async function fetchJson(url, options) { async function fetchJson(url, options, proxyOptions = null) {
const res = await proxyAwareFetch(url, options); const res = await proxyAwareFetch(url, options, proxyOptions);
const text = await res.text(); const text = await res.text();
let data = null; let data = null;
if (text) { if (text) {
@@ -203,11 +203,15 @@ export async function fetchZedAuthenticatedUser(credentials, options = {}) {
const systemId = getSystemId(credentials); const systemId = getSystemId(credentials);
if (systemId) headers[ZED_HEADERS.systemId] = systemId; if (systemId) headers[ZED_HEADERS.systemId] = systemId;
return fetchJson(zedUrl(config, "cloudBaseUrl", "/client/users/me", ZED_CLOUD_BASE_URL), { return fetchJson(
method: "GET", zedUrl(config, "cloudBaseUrl", "/client/users/me", ZED_CLOUD_BASE_URL),
headers, {
signal: options.signal ?? undefined, method: "GET",
}); headers,
signal: options.signal ?? undefined,
},
options.proxyOptions ?? null,
);
} }
function normalizeOrganizationId(value) { function normalizeOrganizationId(value) {

View File

@@ -58,6 +58,15 @@ export function extractThinking(body) {
return { mode: "level", level: e }; return { mode: "level", level: e };
} }
// OpenAI chat / Responses shape — check effort first (zai sends both thinking object and reasoning.effort)
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
if (typeof effort === "string" && effort) {
const e = effort.toLowerCase();
if (e === "none" || e === "off") return { mode: "none" };
if (e === "auto") return { mode: "auto" };
return { mode: "level", level: e };
}
// Claude shape // Claude shape
const t = body.thinking; const t = body.thinking;
if (t && typeof t === "object") { if (t && typeof t === "object") {
@@ -69,15 +78,6 @@ export function extractThinking(body) {
} }
} }
// OpenAI chat / Responses shape
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
if (typeof effort === "string" && effort) {
const e = effort.toLowerCase();
if (e === "none" || e === "off") return { mode: "none" };
if (e === "auto") return { mode: "auto" };
return { mode: "level", level: e };
}
// Gemini shape (top-level, generationConfig, or request envelope) // Gemini shape (top-level, generationConfig, or request envelope)
const tc = body.thinkingConfig || body.generationConfig?.thinkingConfig || body.request?.generationConfig?.thinkingConfig; const tc = body.thinkingConfig || body.generationConfig?.thinkingConfig || body.request?.generationConfig?.thinkingConfig;
if (tc && typeof tc === "object") { if (tc && typeof tc === "object") {
@@ -105,12 +105,16 @@ export function extractThinking(body) {
// at the call-site where intent is snapshotted before format translation. // at the call-site where intent is snapshotted before format translation.
export const captureThinking = extractThinking; export const captureThinking = extractThinking;
// Resolve thinking format: provider override > capability > derive(targetFormat). const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
function resolveFormat(targetFormat, model, provider) { function resolveFormat(targetFormat, model, provider) {
const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null; const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null;
if (providerFmt) return providerFmt; if (providerFmt) return providerFmt;
const caps = getCapabilitiesForModel(provider, model); const caps = getCapabilitiesForModel(provider, model);
if (caps.thinkingFormat) return caps.thinkingFormat; const isOpenAIWire = targetFormat === "openai" || targetFormat === "openai-responses";
if (caps.thinkingFormat && !(isOpenAIWire && NATIVE_ONLY_FORMATS.has(caps.thinkingFormat))) {
return caps.thinkingFormat;
}
return FORMAT_TO_NATIVE[targetFormat] || "openai"; return FORMAT_TO_NATIVE[targetFormat] || "openai";
} }
@@ -237,14 +241,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
} }
case "claude-adaptive": { case "claude-adaptive": {
if (none && canDisable) { body.thinking = { type: "disabled" }; break; } if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
// output_config.effort alone does NOT turn thinking on: Anthropic requires // Models that can disable thinking need the explicit adaptive switch.
// an explicit thinking:{type:"adaptive"} on Opus 4.6/4.7/4.8 and Sonnet 4.6 // Permanently adaptive models such as Fable 5.1 accept effort directly.
// ("thinking is off unless you explicitly set it"), and Anthropic-compatible if (canDisable) body.thinking = { type: "adaptive" };
// shims (e.g. GitHub Copilot /v1/messages) default thinking off even for else delete body.thinking;
// Sonnet 5. Send both fields — the documented adaptive-thinking shape.
body.thinking = { type: "adaptive" };
const level = toLevel(eff); const level = toLevel(eff);
body.output_config = { effort: level === "xhigh" ? "high" : level }; body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level };
break; break;
} }
case "claude-budget": { case "claude-budget": {
@@ -270,6 +272,18 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
// Z.ai ignores thinking.disabled → must use enable_thinking:false to turn off. // Z.ai ignores thinking.disabled → must use enable_thinking:false to turn off.
if (none && canDisable) { body.enable_thinking = false; delete body.thinking; break; } if (none && canDisable) { body.enable_thinking = false; delete body.thinking; break; }
body.thinking = { type: "enabled" }; body.thinking = { type: "enabled" };
// reasoning_effort is only read by z.ai from GLM-5.2 onward — older GLM ignores it
// (see thinkingEffortSupported in capabilities.js). Skip on unsupported models so we
// don't send a field the API doesn't recognize.
if (caps.thinkingEffortSupported) {
const zaiLvl = toLevel(eff);
// GLM-5.3 only accepts exactly low|high|max (anything else errors); GLM-5.2 accepts
// a wider set but z.ai maps low/medium->high and xhigh->max server-side anyway, so
// this 3-value mapping matches both.
body.reasoning_effort = (zaiLvl === "low" || zaiLvl === "minimal") ? "low"
: (zaiLvl === "high" || zaiLvl === "medium") ? "high"
: "max";
}
break; break;
} }
case "qwen": { case "qwen": {

View File

@@ -151,3 +151,17 @@ export function fixMissingToolResponses(body) {
return body; return body;
} }
// Default `type: "custom"` on Claude-format tools that arrive without one.
// Anthropic's Claude tool schema requires `type` to be explicitly set; strict gateways
// (e.g., MiniMax Anthropic-compatible endpoint, error 2013) reject legacy payloads that
// omit it with HTTP 400. Tools that already carry a truthy `type` (e.g., `computer_use`,
// `bash`, `web_search_20250305`) are passed through untouched.
//
// Spread order matters: `{ ...tool, type: "custom" }` (spread first, override last)
// ensures that falsy `type` values (null, undefined, "") in the original tool don't
// overwrite the default. `{ type: "custom", ...tool }` would let `type: null` survive.
export function defaultClaudeToolType(tools) {
if (!Array.isArray(tools)) return tools;
return tools.map(tool => tool?.type ? tool : { ...tool, type: "custom" });
}

View File

@@ -12,6 +12,18 @@ import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
const CACHE_CONTROL_5M = { type: "ephemeral" }; const CACHE_CONTROL_5M = { type: "ephemeral" };
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" }; const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
// Anthropic rejects a tool carrying BOTH defer_loading:true and cache_control
// ("Tools defer_loading cannot use prompt caching", #3567). MCP clients put
// deferred tools at the tail, which is exactly where the cache anchor lands.
// Anchor on the last tool that CAN be cached instead of dropping caching.
export function lastCacheableToolIndex(tools) {
if (!Array.isArray(tools)) return -1;
for (let i = tools.length - 1; i >= 0; i--) {
if (tools[i]?.defer_loading !== true) return i;
}
return -1;
}
// Check if message has valid non-empty content // Check if message has valid non-empty content
export function hasValidContent(msg) { export function hasValidContent(msg) {
if (typeof msg.content === "string" && msg.content.trim()) return true; if (typeof msg.content === "string" && msg.content.trim()) return true;
@@ -108,11 +120,24 @@ function buildThinkingPlaceholder(provider) {
return block; return block;
} }
// Anthropic validates server_tool_use ids against this pattern and rejects the
// whole request with a 400 when one does not match. A combo that falls back to a
// provider with its own built-in tools (z.ai/glm emits OpenAI-style `call_` ids for
// its analyze_image tool) leaves such blocks in the history, so every later Claude
// turn carries a poisoned id.
const CLAUDE_SERVER_TOOL_USE_ID = /^srvtoolu_[a-zA-Z0-9_]+$/;
function hasForeignServerToolUseId(block) {
return block?.type === CLAUDE_BLOCK.SERVER_TOOL_USE
&& !CLAUDE_SERVER_TOOL_USE_ID.test(String(block.id ?? ""));
}
// Normalize a native Claude passthrough body to match Anthropic Messages API spec. // Normalize a native Claude passthrough body to match Anthropic Messages API spec.
// Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject: // Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject:
// 1. thinking.type "adaptive" → unsupported on Haiku // 1. thinking.type "adaptive" → unsupported on Haiku
// 2. output_config.effort → unsupported on Haiku // 2. output_config.effort → unsupported on Haiku
// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed // 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed
// 4. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright
export function normalizeClaudePassthrough(body, model = "") { export function normalizeClaudePassthrough(body, model = "") {
if (!body || typeof body !== "object") return body; if (!body || typeof body !== "object") return body;
@@ -164,6 +189,7 @@ export function normalizeClaudePassthrough(body, model = "") {
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models, // 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// so foreign signatures leak into history and Anthropic rejects them). // so foreign signatures leak into history and Anthropic rejects them).
const thinkingEnabled = body.thinking?.type === "enabled"; const thinkingEnabled = body.thinking?.type === "enabled";
const droppedServerToolUseIds = new Set();
if (Array.isArray(body.messages)) { if (Array.isArray(body.messages)) {
for (const msg of body.messages) { for (const msg of body.messages) {
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue; if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
@@ -178,6 +204,10 @@ export function normalizeClaudePassthrough(body, model = "") {
} }
continue; continue;
} }
if (hasForeignServerToolUseId(block)) {
if (block.id != null) droppedServerToolUseIds.add(String(block.id));
continue;
}
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true; if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
kept.push(block); kept.push(block);
} }
@@ -188,6 +218,35 @@ export function normalizeClaudePassthrough(body, model = "") {
} }
} }
// A dropped server_tool_use leaves its result behind; Anthropic rejects a
// tool_result that references an id no block declares, so both halves must go.
if (droppedServerToolUseIds.size > 0 && Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (!Array.isArray(msg.content)) continue;
const kept = msg.content.filter(block => !(
(block?.type === CLAUDE_BLOCK.TOOL_RESULT || block?.type === CLAUDE_BLOCK.WEB_SEARCH_TOOL_RESULT)
&& droppedServerToolUseIds.has(String(block.tool_use_id ?? ""))
));
if (kept.length !== msg.content.length) {
msg.content = kept;
}
}
}
// 5. Drop empty text blocks and any message left with no content at all.
// Anthropic rejects `messages.N.content` blocks with empty text (400
// "text content blocks must be non-empty"); a message whose blocks were all
// stripped above must be dropped, not padded with an empty placeholder.
if (Array.isArray(body.messages)) {
body.messages = body.messages.filter(msg => {
if (typeof msg.content === "string") return msg.content.trim().length > 0;
if (!Array.isArray(msg.content)) return true;
msg.content = msg.content.filter(block =>
!(block?.type === CLAUDE_BLOCK.TEXT && !String(block.text ?? "").trim()));
return msg.content.length > 0;
});
}
return body; return body;
} }
@@ -223,7 +282,7 @@ export function anchorClaudeCache(body) {
} }
if (Array.isArray(body.tools)) { if (Array.isArray(body.tools)) {
const last = body.tools.length - 1; const last = lastCacheableToolIndex(body.tools);
body.tools.forEach((tool, i) => { body.tools.forEach((tool, i) => {
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H }; if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
else delete tool.cache_control; else delete tool.cache_control;
@@ -252,33 +311,6 @@ export function anchorClaudeCache(body) {
} }
} }
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// so foreign signatures leak into history and Anthropic rejects them).
const thinkingEnabled = body.thinking?.type === "enabled";
if (Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
let hasToolUse = false;
let hasKeptThinking = false;
const kept = [];
for (const block of msg.content) {
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
if (isValidClaudeSignature(block.signature)) {
hasKeptThinking = true;
kept.push(block);
}
continue;
}
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
kept.push(block);
}
msg.content = kept;
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
msg.content.unshift(buildThinkingPlaceholder("claude"));
}
}
}
return body; return body;
} }
@@ -444,9 +476,10 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
}); });
} }
const lastCacheable = lastCacheableToolIndex(body.tools);
body.tools = body.tools.map((tool, i) => { body.tools = body.tools.map((tool, i) => {
const { cache_control, ...rest } = tool; const { cache_control, ...rest } = tool;
if (i === body.tools.length - 1) { if (i === lastCacheable) {
return { ...rest, cache_control: { type: "ephemeral", ttl: "1h" } }; return { ...rest, cache_control: { type: "ephemeral", ttl: "1h" } };
} }
return rest; return rest;

View File

@@ -14,6 +14,8 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
"uniqueItems", "contains", "uniqueItems", "contains",
// 2020-12 keywords with no Gemini equivalent // 2020-12 keywords with no Gemini equivalent
"unevaluatedProperties", "unevaluatedItems", "contentSchema", "unevaluatedProperties", "unevaluatedItems", "contentSchema",
// Tuple-array keywords; converted to items first, leftovers stripped
"prefixItems", "additionalItems",
// Claude rejects these in VALIDATED mode // Claude rejects these in VALIDATED mode
"default", "examples", "default", "examples",
// JSON Schema meta keywords // JSON Schema meta keywords
@@ -308,6 +310,37 @@ function ensureObjectType(obj) {
for (const v of Object.values(obj)) if (v && typeof v === "object") ensureObjectType(v); for (const v of Object.values(obj)) if (v && typeof v === "object") ensureObjectType(v);
} }
// Convert prefixItems (tuple validation) to items — Gemini cannot express tuples,
// and a type:"array" schema without items is rejected with "missing field"
function convertPrefixItems(obj) {
if (!obj || typeof obj !== "object") return;
if (Array.isArray(obj.prefixItems) && obj.prefixItems.length > 0) {
const variants = obj.prefixItems.filter(s => s && s.type !== "null");
if (!obj.items && variants.length === 1) {
obj.items = variants[0];
} else if (!obj.items && variants.length > 1) {
obj.items = { anyOf: variants };
}
delete obj.prefixItems;
}
for (const value of Object.values(obj)) {
if (value && typeof value === "object") {
convertPrefixItems(value);
}
}
}
// Gemini requires items on every type:"array" schema — fill a permissive placeholder
function ensureArrayItems(obj) {
if (!obj || typeof obj !== "object") return;
if (obj.type === "array" && !obj.items) {
obj.items = { type: "string" };
}
for (const v of Object.values(obj)) if (v && typeof v === "object") ensureArrayItems(v);
}
// Clean JSON Schema for Antigravity API compatibility - removes unsupported keywords recursively // Clean JSON Schema for Antigravity API compatibility - removes unsupported keywords recursively
export function cleanJSONSchemaForAntigravity(schema) { export function cleanJSONSchemaForAntigravity(schema) {
if (!schema || typeof schema !== "object") return schema; if (!schema || typeof schema !== "object") return schema;
@@ -321,11 +354,13 @@ export function cleanJSONSchemaForAntigravity(schema) {
// Phase 2: Flatten complex structures // Phase 2: Flatten complex structures
mergeAllOf(cleaned); mergeAllOf(cleaned);
convertPrefixItems(cleaned);
flattenAnyOfOneOf(cleaned); flattenAnyOfOneOf(cleaned);
flattenTypeArrays(cleaned); flattenTypeArrays(cleaned);
// Phase 2.5: Infer missing type=object when properties exist (Gemini requirement) // Phase 2.5: Infer missing type=object when properties exist (Gemini requirement)
ensureObjectType(cleaned); ensureObjectType(cleaned);
ensureArrayItems(cleaned);
// Phase 3: Remove all unsupported keywords at ALL levels (including inside arrays) // Phase 3: Remove all unsupported keywords at ALL levels (including inside arrays)
removeUnsupportedKeywords(cleaned, UNSUPPORTED_SCHEMA_CONSTRAINTS); removeUnsupportedKeywords(cleaned, UNSUPPORTED_SCHEMA_CONSTRAINTS);

View File

@@ -23,6 +23,59 @@ export function normalizeResponsesInput(input) {
return null; return null;
} }
// Strict Responses upstreams reject overlong call_ids with InputValidationError (#393).
export const MAX_RESPONSES_CALL_ID_LEN = 64;
// Fallback ids share one Date.now() when a batch of items is sanitized in a tight
// loop — a per-process sequence keeps same-millisecond ids unique so
// function_call ↔ function_call_output correlation never collides.
let responsesCallIdSeq = 0;
export function clampResponsesCallId(id) {
if (typeof id !== "string" || !id) return `call_${Date.now()}_${(responsesCallIdSeq += 1)}`;
return id.length > MAX_RESPONSES_CALL_ID_LEN ? id.substring(0, MAX_RESPONSES_CALL_ID_LEN) : id;
}
// Single-stringify: objects → JSON once; valid JSON strings pass through untouched;
// anything else (partial fragments, empty) falls back to "{}" instead of
// double-encoding and tripping upstream InputValidationError.
export function coerceResponsesArguments(value) {
if (value === undefined || value === null || value === "") return "{}";
if (typeof value !== "string") {
try {
return JSON.stringify(value);
} catch {
return "{}";
}
}
try {
JSON.parse(value);
return value;
} catch {
return "{}";
}
}
// function_call_output.output must be a string — never null/object.
export function coerceResponsesOutput(value) {
if (typeof value === "string") return value;
if (value === undefined || value === null) return "";
if (Array.isArray(value)) {
return value.map((c) => {
try {
return c?.text ?? JSON.stringify(c);
} catch {
return String(c);
}
}).join("");
}
try {
return JSON.stringify(value);
} catch {
return String(value);
}
}
/** /**
* Convert OpenAI Responses API format to standard chat completions format * Convert OpenAI Responses API format to standard chat completions format
* Responses API uses: { input: [...], instructions: "..." } * Responses API uses: { input: [...], instructions: "..." }

View File

@@ -1,7 +1,7 @@
import { FORMATS } from "./formats.js"; import { FORMATS } from "./formats.js";
import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js"; import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js";
import { prepareClaudeRequest } from "./formats/claude.js"; import { prepareClaudeRequest } from "./formats/claude.js";
import { cloakClaudeTools } from "../utils/claudeCloaking.js"; import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js";
import { filterToOpenAIFormat } from "./formats/openai.js"; import { filterToOpenAIFormat } from "./formats/openai.js";
import { normalizeThinkingConfig } from "../services/provider.js"; import { normalizeThinkingConfig } from "../services/provider.js";
import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js"; import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js";
@@ -133,7 +133,7 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
result = prepareClaudeRequest(result, provider, apiKey, connectionId, credentials?.rawHeaders, clientSessionId); result = prepareClaudeRequest(result, provider, apiKey, connectionId, credentials?.rawHeaders, clientSessionId);
} }
// Claude cloaking: rename client tools with _cc suffix (anti-ban) // Claude cloaking: rename client tools with CLAUDE_TOOL_SUFFIX (anti-ban)
// quirk: only providers flagged cloakToolsOnOAuth, and only with an OAuth token // quirk: only providers flagged cloakToolsOnOAuth, and only with an OAuth token
if (PROVIDERS[provider]?.quirks?.cloakToolsOnOAuth) { if (PROVIDERS[provider]?.quirks?.cloakToolsOnOAuth) {
const apiKey = credentials?.accessToken || credentials?.apiKey || null; const apiKey = credentials?.accessToken || credentials?.apiKey || null;
@@ -161,9 +161,12 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
// Translate response chunk: target -> openai -> source // Translate response chunk: target -> openai -> source
export function translateResponse(targetFormat, sourceFormat, chunk, state) { export function translateResponse(targetFormat, sourceFormat, chunk, state) {
ensureInitialized(); ensureInitialized();
// If same format, return as-is // If same format, return as-is — except the tool name may still be cloaked:
// translateRequest() suffixes client tools for OAuth-cloaked Claude providers
// even when no format conversion is needed, so streamed tool_use blocks must
// be decloaked here or the client sees an unknown ("_ide"-suffixed) tool.
if (sourceFormat === targetFormat) { if (sourceFormat === targetFormat) {
return [chunk]; return [decloakStreamChunk(chunk, state?.toolNameMap)];
} }
let results = [chunk]; let results = [chunk];

View File

@@ -327,7 +327,6 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
}; };
if (profileArn) payload.profileArn = profileArn; if (profileArn) payload.profileArn = profileArn;
if (systemPrompt) payload.systemPrompt = systemPrompt;
if (additionalModelRequestFields) { if (additionalModelRequestFields) {
payload.additionalModelRequestFields = additionalModelRequestFields; payload.additionalModelRequestFields = additionalModelRequestFields;
} }

View File

@@ -6,12 +6,15 @@
*/ */
import { register } from "../index.js"; import { register } from "../index.js";
import { FORMATS } from "../formats.js"; import { FORMATS } from "../formats.js";
import { normalizeResponsesInput } from "../formats/responsesApi.js"; import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../formats/responsesApi.js";
import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM } from "../schema/index.js"; import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM } from "../schema/index.js";
// Responses API enforces max 64 chars on call_id (#393) const MAX_TOOL_NAME_LEN = 128;
const MAX_CALL_ID_LEN = 64;
const clampCallId = (id) => (typeof id === "string" && id.length > MAX_CALL_ID_LEN ? id.substring(0, MAX_CALL_ID_LEN) : id);
/** /**
* Convert OpenAI Responses API request to OpenAI Chat Completions format * Convert OpenAI Responses API request to OpenAI Chat Completions format
@@ -249,6 +252,23 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
return result; return result;
} }
/**
* Extract plain text from a system/developer message for Responses instructions.
* Array content (text parts) is joined; anything else falls back to "" rather
* than leaking "[object Object]" upstream.
*/
function extractInstructionsText(content) {
if (typeof content === "string") return content;
if (Array.isArray(content)) {
return content.map((c) => {
if (typeof c?.text === "string") return c.text;
if (typeof c?.content === "string") return c.content;
return "";
}).filter(Boolean).join("\n");
}
return "";
}
/** /**
* Ensure object schema always has properties field (required by Codex Responses API) * Ensure object schema always has properties field (required by Codex Responses API)
*/ */
@@ -300,7 +320,16 @@ function buildReasoningInputItem(msg) {
*/ */
export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) { export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) {
// Body already in Responses API format (e.g. Cursor CLI calling /chat/completions with input[]) // Body already in Responses API format (e.g. Cursor CLI calling /chat/completions with input[])
if (body.input) return { ...body, model, stream: true }; if (body.input) {
const out = { ...body, model, stream: true };
if (out.max_output_tokens === undefined) {
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
}
delete out.max_tokens;
delete out.max_completion_tokens;
return out;
}
const result = { const result = {
model, model,
@@ -318,7 +347,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
// Use the first instruction-bearing message as instructions. // Use the first instruction-bearing message as instructions.
// OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt. // OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt.
if (!hasSystemMessage) { if (!hasSystemMessage) {
result.instructions = typeof msg.content === "string" ? msg.content : ""; result.instructions = extractInstructionsText(msg.content);
hasSystemMessage = true; hasSystemMessage = true;
} }
continue; // Skip instruction messages in input continue; // Skip instruction messages in input
@@ -369,26 +398,24 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
// Convert tool calls // Convert tool calls
if (msg.role === ROLE.ASSISTANT && msg.tool_calls) { if (msg.role === ROLE.ASSISTANT && msg.tool_calls) {
for (const tc of msg.tool_calls) { for (const tc of msg.tool_calls) {
// Skip nameless calls — strict Responses upstreams reject them (#444)
const name = typeof tc.function?.name === "string" ? tc.function.name.trim() : "";
if (!name) continue;
result.input.push({ result.input.push({
type: RESPONSES_ITEM.FUNCTION_CALL, type: RESPONSES_ITEM.FUNCTION_CALL,
call_id: clampCallId(tc.id), call_id: clampResponsesCallId(tc.id),
name: tc.function?.name || "_unknown", name: name.slice(0, MAX_TOOL_NAME_LEN),
arguments: tc.function?.arguments || "{}" arguments: coerceResponsesArguments(tc.function?.arguments)
}); });
} }
} }
// Convert tool results - output must be a string for Responses API // Convert tool results - output must be a string for Responses API
if (msg.role === ROLE.TOOL) { if (msg.role === ROLE.TOOL) {
const output = typeof msg.content === "string"
? msg.content
: Array.isArray(msg.content)
? msg.content.map(c => c.text || JSON.stringify(c)).join("")
: JSON.stringify(msg.content);
result.input.push({ result.input.push({
type: RESPONSES_ITEM.FUNCTION_CALL_OUTPUT, type: RESPONSES_ITEM.FUNCTION_CALL_OUTPUT,
call_id: clampCallId(msg.tool_call_id), call_id: clampResponsesCallId(msg.tool_call_id),
output output: coerceResponsesOutput(msg.content)
}); });
} }
} }
@@ -402,21 +429,30 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
if (body.tools && Array.isArray(body.tools)) { if (body.tools && Array.isArray(body.tools)) {
result.tools = body.tools.map(tool => { result.tools = body.tools.map(tool => {
if (tool.type === OPENAI_BLOCK.FUNCTION) { if (tool.type === OPENAI_BLOCK.FUNCTION) {
// Strict upstreams reject nameless/overlong tool declarations
const name = typeof tool.function?.name === "string" ? tool.function.name.trim() : "";
if (!name) return null;
return { return {
type: OPENAI_BLOCK.FUNCTION, type: OPENAI_BLOCK.FUNCTION,
name: tool.function.name, name: name.slice(0, MAX_TOOL_NAME_LEN),
description: String(tool.function.description || ""), description: String(tool.function.description || ""),
parameters: normalizeToolParameters(tool.function.parameters), parameters: normalizeToolParameters(tool.function.parameters),
strict: tool.function.strict strict: tool.function.strict
}; };
} }
return tool; return tool;
}); }).filter(Boolean);
} }
// Pass through other relevant fields // Pass through other relevant fields
if (body.temperature !== undefined) result.temperature = body.temperature; if (body.temperature !== undefined) result.temperature = body.temperature;
if (body.max_tokens !== undefined) result.max_tokens = body.max_tokens; if (body.max_output_tokens !== undefined) {
result.max_output_tokens = body.max_output_tokens;
} else if (body.max_completion_tokens !== undefined) {
result.max_output_tokens = body.max_completion_tokens;
} else if (body.max_tokens !== undefined) {
result.max_output_tokens = body.max_tokens;
}
if (body.top_p !== undefined) result.top_p = body.top_p; if (body.top_p !== undefined) result.top_p = body.top_p;
if (body.reasoning !== undefined) result.reasoning = body.reasoning; if (body.reasoning !== undefined) result.reasoning = body.reasoning;
if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" }; if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" };

View File

@@ -13,160 +13,204 @@ import { register } from "../index.js";
import { FORMATS } from "../formats.js"; import { FORMATS } from "../formats.js";
import { randomUUID } from "crypto"; import { randomUUID } from "crypto";
import { ROLE, OPENAI_BLOCK } from "../schema/index.js"; import { ROLE, OPENAI_BLOCK } from "../schema/index.js";
import { DEFAULT_IMAGE_MIME } from "../schema/index.js";
import { parseDataUri } from "../concerns/image.js";
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js"; import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
function flattenText(content) { function flattenText(content) {
if (content == null) return ""; if (content == null) return "";
if (typeof content === "string") return content; if (typeof content === "string") return content;
if (Array.isArray(content)) { if (Array.isArray(content)) {
const parts = []; const parts = [];
for (const p of content) { for (const p of content) {
if (typeof p === "string") parts.push(p); if (typeof p === "string") parts.push(p);
else if (p && typeof p === "object" && typeof p.text === "string") parts.push(p.text); else if (p && typeof p === "object" && typeof p.text === "string")
} parts.push(p.text);
return parts.join("\n"); }
} return parts.join("\n");
return String(content); }
return String(content);
} }
function toContentBlocks(content) { function toContentBlocks(content) {
if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }]; if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }];
if (typeof content === "string") return [{ type: OPENAI_BLOCK.TEXT, text: content }]; if (typeof content === "string")
if (Array.isArray(content)) { return [{ type: OPENAI_BLOCK.TEXT, text: content }];
const blocks = []; if (Array.isArray(content)) {
for (const part of content) { const blocks = [];
if (typeof part === "string") { for (const part of content) {
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part }); if (typeof part === "string") {
} else if (part && typeof part === "object") { blocks.push({ type: OPENAI_BLOCK.TEXT, text: part });
if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") { } else if (part && typeof part === "object") {
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") {
} else if (part.type === OPENAI_BLOCK.IMAGE_URL || part.type === OPENAI_BLOCK.IMAGE) { blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
blocks.push({ type: OPENAI_BLOCK.TEXT, text: "[image omitted]" }); } else if (
} else if (typeof part.text === "string") { part.type === OPENAI_BLOCK.IMAGE_URL ||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); part.type === OPENAI_BLOCK.IMAGE
} ) {
} // CommandCode `/alpha/generate` accepts {type:"image", image:"<data URI | url>"} —
} // same shape the official command-code CLI sends (verified from CLI source).
return blocks.length ? blocks : [{ type: OPENAI_BLOCK.TEXT, text: "" }]; const src = part.source;
} let raw = part.image_url?.url || src?.data || src?.url || "";
return [{ type: OPENAI_BLOCK.TEXT, text: String(content) }]; let parsed = parseDataUri(raw);
if (!parsed && src?.type === "base64" && src?.data) {
// Claude-style base64 source without a data-URI prefix → wrap it.
raw = `data:${src.media_type || DEFAULT_IMAGE_MIME};base64,${src.data}`;
parsed = parseDataUri(raw);
}
if (parsed) {
blocks.push({
type: "image",
image: `data:${parsed.mimeType};base64,${parsed.base64}`,
});
} else if (raw) {
blocks.push({ type: "image", image: raw });
}
} else if (typeof part.text === "string") {
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
}
}
}
return blocks.length ? blocks : [{ type: OPENAI_BLOCK.TEXT, text: "" }];
}
return [{ type: OPENAI_BLOCK.TEXT, text: String(content) }];
} }
function safeParseJson(s) { function safeParseJson(s) {
if (s == null) return {}; if (s == null) return {};
if (typeof s !== "string") return s; if (typeof s !== "string") return s;
try { return JSON.parse(s); } catch { return {}; } try {
return JSON.parse(s);
} catch {
return {};
}
} }
function convertMessages(messages = []) { function convertMessages(messages = []) {
const out = []; const out = [];
const systemTexts = []; const systemTexts = [];
for (const m of messages) { for (const m of messages) {
if (!m) continue; if (!m) continue;
const role = m.role; const role = m.role;
if (role === ROLE.SYSTEM) { if (role === ROLE.SYSTEM) {
const t = flattenText(m.content); const t = flattenText(m.content);
if (t) systemTexts.push(t); if (t) systemTexts.push(t);
continue; continue;
} }
if (role === ROLE.TOOL) { if (role === ROLE.TOOL) {
const value = typeof m.content === "string" ? m.content : flattenText(m.content); const value =
out.push({ typeof m.content === "string" ? m.content : flattenText(m.content);
role: ROLE.TOOL, out.push({
content: [{ role: ROLE.TOOL,
type: "tool-result", content: [
toolCallId: m.tool_call_id || "", {
toolName: m.name || "", type: "tool-result",
output: { type: "text", value }, toolCallId: m.tool_call_id || "",
}], toolName: m.name || "",
}); output: { type: "text", value },
continue; },
} ],
});
continue;
}
if (role === ROLE.ASSISTANT) { if (role === ROLE.ASSISTANT) {
const blocks = []; const blocks = [];
const text = flattenText(m.content); const text = flattenText(m.content);
if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text }); if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
if (Array.isArray(m.tool_calls)) { if (Array.isArray(m.tool_calls)) {
for (const tc of m.tool_calls) { for (const tc of m.tool_calls) {
const fn = tc.function || {}; const fn = tc.function || {};
blocks.push({ blocks.push({
type: "tool-call", type: "tool-call",
toolCallId: tc.id || "", toolCallId: tc.id || "",
toolName: fn.name || "", toolName: fn.name || "",
input: safeParseJson(fn.arguments), input: safeParseJson(fn.arguments),
}); });
} }
} }
out.push({ role: ROLE.ASSISTANT, content: blocks.length ? blocks : [{ type: OPENAI_BLOCK.TEXT, text: "" }] }); out.push({
continue; role: ROLE.ASSISTANT,
} content: blocks.length
? blocks
: [{ type: OPENAI_BLOCK.TEXT, text: "" }],
});
continue;
}
out.push({ role: ROLE.USER, content: toContentBlocks(m.content) }); out.push({ role: ROLE.USER, content: toContentBlocks(m.content) });
} }
return { messages: out, system: systemTexts.join("\n\n") }; return { messages: out, system: systemTexts.join("\n\n") };
} }
function convertTools(tools) { function convertTools(tools) {
if (!Array.isArray(tools) || tools.length === 0) return undefined; if (!Array.isArray(tools) || tools.length === 0) return undefined;
const result = []; const result = [];
for (const t of tools) { for (const t of tools) {
if (!t) continue; if (!t) continue;
if (t.type === OPENAI_BLOCK.FUNCTION && t.function) { if (t.type === OPENAI_BLOCK.FUNCTION && t.function) {
result.push({ result.push({
name: t.function.name, name: t.function.name,
description: t.function.description, description: t.function.description,
input_schema: t.function.parameters || { type: "object" }, input_schema: t.function.parameters || { type: "object" },
}); });
} else if (t.name && (t.input_schema || t.parameters)) { } else if (t.name && (t.input_schema || t.parameters)) {
result.push({ result.push({
name: t.name, name: t.name,
description: t.description, description: t.description,
input_schema: t.input_schema || t.parameters, input_schema: t.input_schema || t.parameters,
}); });
} }
} }
return result.length ? result : undefined; return result.length ? result : undefined;
} }
export function openaiToCommandCodeRequest(model, body, stream /* , credentials */) { export function openaiToCommandCodeRequest(
const { messages, system } = convertMessages(body.messages); model,
const params = { body,
model, stream /* , credentials */,
messages, ) {
stream: stream !== false, const { messages, system } = convertMessages(body.messages);
max_tokens: body.max_tokens ?? body.max_output_tokens ?? DEFAULT_MAX_TOKENS, const params = {
temperature: body.temperature ?? 0.3, model,
}; messages,
stream: stream !== false,
max_tokens: body.max_tokens ?? body.max_output_tokens ?? DEFAULT_MAX_TOKENS,
temperature: body.temperature ?? 0.3,
};
if (system) params.system = system; if (system) params.system = system;
const tools = convertTools(body.tools); const tools = convertTools(body.tools);
if (tools) params.tools = tools; if (tools) params.tools = tools;
if (body.top_p != null) params.top_p = body.top_p; if (body.top_p != null) params.top_p = body.top_p;
const today = new Date().toISOString().slice(0, 10); const today = new Date().toISOString().slice(0, 10);
return { // environment format mirrors the official command-code CLI (getEnvironmentInfo):
threadId: randomUUID(), // `${platform}-${arch}, Node.js ${version}` (e.g. "darwin-arm64, Node.js v24.16.0").
memory: "", const environment = `${process.platform}-${process.arch}, Node.js ${process.version}`;
config: {
workingDir: process.cwd(), return {
date: today, threadId: randomUUID(),
environment: process.platform, memory: "",
structure: [], config: {
isGitRepo: false, workingDir: process.cwd(),
currentBranch: "", date: today,
mainBranch: "", environment,
gitStatus: "", structure: [],
recentCommits: [], isGitRepo: false,
}, currentBranch: "",
params, mainBranch: "",
}; gitStatus: "",
recentCommits: [],
},
params,
};
} }
register(FORMATS.OPENAI, FORMATS.COMMANDCODE, openaiToCommandCodeRequest, null); register(FORMATS.OPENAI, FORMATS.COMMANDCODE, openaiToCommandCodeRequest, null);

View File

@@ -2,6 +2,7 @@ import { register } from "../index.js";
import { FORMATS } from "../formats.js"; import { FORMATS } from "../formats.js";
import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js"; import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js";
import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js"; import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js";
import { getGeminiThoughtSignatureSync } from "../../services/thoughtSignatureStore.js";
function generateUUID() { function generateUUID() {
return crypto.randomUUID(); return crypto.randomUUID();
} }
@@ -46,7 +47,7 @@ function normalizeGeminiContents(contents) {
} }
// Core: Convert OpenAI request to Gemini format (base for all variants) // Core: Convert OpenAI request to Gemini format (base for all variants)
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) { function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) {
const result = { const result = {
model: model, model: model,
contents: [], contents: [],
@@ -133,18 +134,27 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
if (msg.tool_calls && Array.isArray(msg.tool_calls)) { if (msg.tool_calls && Array.isArray(msg.tool_calls)) {
const toolCallIds = []; const toolCallIds = [];
let firstFunctionCallSeen = false;
for (const tc of msg.tool_calls) { for (const tc of msg.tool_calls) {
if (tc.type !== OPENAI_BLOCK.FUNCTION) continue; if (tc.type !== OPENAI_BLOCK.FUNCTION) continue;
const args = tryParseJSON(tc.function?.arguments || "{}"); const args = tryParseJSON(tc.function?.arguments || "{}");
parts.push({ const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null;
thoughtSignature: signature, // First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig
const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined);
firstFunctionCallSeen = true;
const part = {
functionCall: { functionCall: {
id: tc.id, id: tc.id,
name: sanitizeGeminiFunctionName(tc.function.name), name: sanitizeGeminiFunctionName(tc.function.name),
args: args args: args
} }
}); };
if (callSig) {
part.thoughtSignature = callSig;
}
parts.push(part);
toolCallIds.push(tc.id); toolCallIds.push(tc.id);
} }
@@ -232,13 +242,13 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
} }
// OpenAI -> Gemini (standard API) // OpenAI -> Gemini (standard API)
export function openaiToGeminiRequest(model, body, stream) { export function openaiToGeminiRequest(model, body, stream, credentials = null) {
return openaiToGeminiBase(model, body, stream); return openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_AG_SIGNATURE, credentials?._clientSessionId);
} }
// OpenAI -> Gemini CLI (Cloud Code Assist) // OpenAI -> Gemini CLI (Cloud Code Assist)
export function openaiToGeminiCLIRequest(model, body, stream) { export function openaiToGeminiCLIRequest(model, body, stream, credentials = null) {
const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE); const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE, credentials?._clientSessionId);
// Thinking is normalized centrally by applyThinking (thinkingUnified.js) after translation. // Thinking is normalized centrally by applyThinking (thinkingUnified.js) after translation.
// Clean schema for tools // Clean schema for tools
@@ -335,18 +345,26 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
const parts = []; const parts = [];
if (Array.isArray(msg.content)) { if (Array.isArray(msg.content)) {
let firstToolUseSeen = false;
for (const block of msg.content) { for (const block of msg.content) {
if (block.type === CLAUDE_BLOCK.TEXT) { if (block.type === CLAUDE_BLOCK.TEXT) {
parts.push({ text: block.text }); parts.push({ text: block.text });
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) { } else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
parts.push({ const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null;
thoughtSignature: signature, const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined);
firstToolUseSeen = true;
const part = {
functionCall: { functionCall: {
id: block.id, id: block.id,
name: sanitizeGeminiFunctionName(block.name), name: sanitizeGeminiFunctionName(block.name),
args: block.input || {} args: block.input || {}
} }
}); };
if (callSig) {
part.thoughtSignature = callSig;
}
parts.push(part);
} else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) { } else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) {
let content = block.content; let content = block.content;
if (Array.isArray(content)) { if (Array.isArray(content)) {

View File

@@ -420,7 +420,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
if (profileArn) { if (profileArn) {
payload.profileArn = profileArn; payload.profileArn = profileArn;
} }
if (systemPrompt) payload.systemPrompt = systemPrompt;
if (additionalModelRequestFields) { if (additionalModelRequestFields) {
payload.additionalModelRequestFields = additionalModelRequestFields; payload.additionalModelRequestFields = additionalModelRequestFields;
} }

View File

@@ -25,160 +25,198 @@ import { fallbackToolCallId } from "../concerns/toolCall.js";
import { toOpenAIFinish } from "../concerns/finishReason.js"; import { toOpenAIFinish } from "../concerns/finishReason.js";
function ensureState(state, model) { function ensureState(state, model) {
if (!state.responseId) { if (!state.responseId) {
state.responseId = `chatcmpl-${Date.now()}`; state.responseId = `chatcmpl-${Date.now()}`;
state.created = Math.floor(Date.now() / 1000); state.created = Math.floor(Date.now() / 1000);
state.model = state.model || model || "commandcode"; state.model = state.model || model || "commandcode";
state.chunkIndex = 0; state.chunkIndex = 0;
state.toolIndex = 0; state.toolIndex = 0;
state.toolIndexById = new Map(); state.toolIndexById = new Map();
state.openTools = new Set(); state.openTools = new Set();
state.openText = false; state.openText = false;
state.finishReason = null; state.finishReason = null;
state.usage = null; state.usage = null;
} }
} }
function makeChunk(state, delta, finishReason = null) { function makeChunk(state, delta, finishReason = null) {
return buildChunk( return buildChunk(
{ id: state.responseId, created: state.created, model: state.model }, { id: state.responseId, created: state.created, model: state.model },
delta, delta,
finishReason finishReason,
); );
} }
const mapFinishReason = (reason) => toOpenAIFinish(reason, "commandcode"); const mapFinishReason = (reason) => toOpenAIFinish(reason, "commandcode");
export function commandCodeToOpenAIResponse(chunk, state) { export function commandCodeToOpenAIResponse(chunk, state) {
if (!chunk) return null; if (!chunk) return null;
// Already-OpenAI chunk: pass through // Already-OpenAI chunk: pass through
if (chunk && typeof chunk === "object" && chunk.object === "chat.completion.chunk") { if (
return chunk; chunk &&
} typeof chunk === "object" &&
chunk.object === "chat.completion.chunk"
) {
return chunk;
}
// Parse string lines coming out of upstream // Parse string lines coming out of upstream
let event = chunk; let event = chunk;
if (typeof chunk === "string") { if (typeof chunk === "string") {
const line = chunk.trim(); const line = chunk.trim();
if (!line) return null; if (!line) return null;
// Tolerate raw "data: {...}" framing if the upstream wrapper inserts it // Tolerate raw "data: {...}" framing if the upstream wrapper inserts it
const json = line.startsWith("data:") ? line.slice(5).trim() : line; const json = line.startsWith("data:") ? line.slice(5).trim() : line;
if (!json || json === "[DONE]") return null; if (!json || json === "[DONE]") return null;
try { try {
event = JSON.parse(json); event = JSON.parse(json);
} catch { } catch {
return null; return null;
} }
} }
if (!event || typeof event !== "object" || !event.type) return null; if (!event || typeof event !== "object" || !event.type) return null;
ensureState(state, event.model); ensureState(state, event.model);
const out = []; const out = [];
switch (event.type) { switch (event.type) {
case "text-delta": { case "text-delta": {
const text = event.text || event.delta || ""; const text = event.text || event.delta || "";
if (!text) break; if (!text) break;
const delta = state.chunkIndex === 0 ? { role: ROLE.ASSISTANT, content: text } : { content: text }; const delta =
state.chunkIndex++; state.chunkIndex === 0
state.openText = true; ? { role: ROLE.ASSISTANT, content: text }
out.push(makeChunk(state, delta)); : { content: text };
break; state.chunkIndex++;
} state.openText = true;
case "reasoning-delta": { out.push(makeChunk(state, delta));
const text = event.text || ""; break;
if (!text) break; }
// Map reasoning to OpenAI "reasoning_content" field (used by deepseek-reasoner-style clients). case "reasoning-delta": {
const delta = reasoningDelta(text, state.chunkIndex === 0); const text = event.text || "";
state.chunkIndex++; if (!text) break;
out.push(makeChunk(state, delta)); // Map reasoning to OpenAI "reasoning_content" field (used by deepseek-reasoner-style clients).
break; const delta = reasoningDelta(text, state.chunkIndex === 0);
} state.chunkIndex++;
case "tool-input-start": { out.push(makeChunk(state, delta));
const id = event.id || event.toolCallId || fallbackToolCallId(state.toolIndex); break;
let idx = state.toolIndexById.get(id); }
if (idx == null) { case "tool-input-start": {
idx = state.toolIndex++; const id =
state.toolIndexById.set(id, idx); event.id || event.toolCallId || fallbackToolCallId(state.toolIndex);
} let idx = state.toolIndexById.get(id);
state.openTools.add(id); if (idx == null) {
const delta = { idx = state.toolIndex++;
...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}), state.toolIndexById.set(id, idx);
tool_calls: [{ }
index: idx, state.openTools.add(id);
id, const delta = {
type: OPENAI_BLOCK.FUNCTION, ...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}),
function: { name: event.toolName || "", arguments: "" }, tool_calls: [
}], {
}; index: idx,
state.chunkIndex++; id,
out.push(makeChunk(state, delta)); type: OPENAI_BLOCK.FUNCTION,
break; function: { name: event.toolName || "", arguments: "" },
} },
case "tool-input-delta": { ],
const id = event.id || event.toolCallId; };
const idx = state.toolIndexById.get(id); state.chunkIndex++;
if (idx == null) break; out.push(makeChunk(state, delta));
const delta = { break;
tool_calls: [{ }
index: idx, case "tool-input-delta": {
function: { arguments: event.delta || event.inputTextDelta || "" }, const id = event.id || event.toolCallId;
}], const idx = state.toolIndexById.get(id);
}; if (idx == null) break;
out.push(makeChunk(state, delta)); const delta = {
break; tool_calls: [
} {
case "tool-call": { index: idx,
// Final consolidated tool call — only emit if we never saw tool-input-* deltas. function: { arguments: event.delta || event.inputTextDelta || "" },
const id = event.toolCallId; },
if (state.toolIndexById.has(id)) break; ],
const idx = state.toolIndex++; };
state.toolIndexById.set(id, idx); out.push(makeChunk(state, delta));
const argsStr = typeof event.input === "string" ? event.input : JSON.stringify(event.input ?? {}); break;
const delta = { }
...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}), case "tool-call": {
tool_calls: [{ // Final consolidated tool call — only emit if we never saw tool-input-* deltas.
index: idx, const id = event.toolCallId;
id, if (state.toolIndexById.has(id)) break;
type: OPENAI_BLOCK.FUNCTION, const idx = state.toolIndex++;
function: { name: event.toolName || "", arguments: argsStr }, state.toolIndexById.set(id, idx);
}], const argsStr =
}; typeof event.input === "string"
state.chunkIndex++; ? event.input
out.push(makeChunk(state, delta)); : JSON.stringify(event.input ?? {});
break; const delta = {
} ...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}),
case "finish-step": { tool_calls: [
state.finishReason = mapFinishReason(event.finishReason); {
if (event.usage) state.usage = event.usage; index: idx,
break; id,
} type: OPENAI_BLOCK.FUNCTION,
case "finish": { function: { name: event.toolName || "", arguments: argsStr },
const finishReason = state.finishReason || mapFinishReason(event.finishReason || "stop"); },
const finalChunk = makeChunk(state, {}, finishReason); ],
const totalUsage = event.totalUsage || state.usage; };
const usage = toOpenAIUsage(totalUsage, "commandcode"); state.chunkIndex++;
if (usage) finalChunk.usage = usage; out.push(makeChunk(state, delta));
out.push(finalChunk); break;
break; }
} case "finish-step": {
case "error": { state.finishReason = mapFinishReason(event.finishReason);
state.finishReason = OPENAI_FINISH.STOP; if (event.usage) state.usage = event.usage;
const errVal = event.error ?? event.message ?? "unknown"; break;
const errStr = typeof errVal === "string" ? errVal : JSON.stringify(errVal); }
out.push(makeChunk(state, { content: `\n\n[CommandCode error: ${errStr}]` })); case "finish": {
out.push(makeChunk(state, {}, OPENAI_FINISH.STOP)); const finishReason =
break; state.finishReason || mapFinishReason(event.finishReason || "stop");
} const finalChunk = makeChunk(state, {}, finishReason);
// Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end, const totalUsage = event.totalUsage || state.usage;
// provider-metadata, message-metadata, etc. They carry no client-visible content. const usage = toOpenAIUsage(totalUsage, "commandcode");
default: if (usage) finalChunk.usage = usage;
break; out.push(finalChunk);
} break;
}
case "error": {
// Terminal upstream failure (AI SDK v5 error event) — NOT content. Emit an
// OpenAI-shaped error chunk (chunk.error) so downstream — parseSSEToOpenAIResponse
// for non-streaming, OpenAI SDK clients for streaming — treats the request as
// failed instead of surfacing fake success content like "[CommandCode error: ...]".
state.finishReason = OPENAI_FINISH.STOP;
const errVal = event.error ?? event.message ?? "unknown";
const errStr =
typeof errVal === "string"
? errVal
: typeof errVal?.message === "string"
? errVal.message
: JSON.stringify(errVal);
const errType =
typeof errVal === "string"
? "upstream_error"
: errVal?.type || "upstream_error";
const errChunk = makeChunk(state, {});
errChunk.error = { message: errStr, type: errType };
out.push(errChunk);
out.push(makeChunk(state, {}, OPENAI_FINISH.STOP));
break;
}
// Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end,
// provider-metadata, message-metadata, etc. They carry no client-visible content.
default:
break;
}
return out.length ? out : null; return out.length ? out : null;
} }
register(FORMATS.COMMANDCODE, FORMATS.OPENAI, null, commandCodeToOpenAIResponse); register(
FORMATS.COMMANDCODE,
FORMATS.OPENAI,
null,
commandCodeToOpenAIResponse,
);

View File

@@ -6,6 +6,7 @@ import { toOpenAIUsage } from "../concerns/usage.js";
import { reasoningDelta } from "../concerns/reasoning.js"; import { reasoningDelta } from "../concerns/reasoning.js";
import { encodeDataUri } from "../concerns/image.js"; import { encodeDataUri } from "../concerns/image.js";
import { toOpenAIFinish } from "../concerns/finishReason.js"; import { toOpenAIFinish } from "../concerns/finishReason.js";
import { storeGeminiThoughtSignature } from "../../services/thoughtSignatureStore.js";
// Build chunk meta for current gemini state // Build chunk meta for current gemini state
function chunkMeta(state) { function chunkMeta(state) {
@@ -13,14 +14,18 @@ function chunkMeta(state) {
} }
// Build a tool_call chunk from a gemini functionCall part (shared by sig/non-sig branches) // Build a tool_call chunk from a gemini functionCall part (shared by sig/non-sig branches)
function emitFunctionCall(functionCall, state) { function emitFunctionCall(functionCall, state, signature = null) {
const rawName = functionCall.name; const rawName = functionCall.name;
// Restore original tool name from mapping (AG cloaking) // Restore original tool name from mapping (AG cloaking)
const fcName = state.toolNameMap?.get(rawName) || rawName; const fcName = state.toolNameMap?.get(rawName) || rawName;
const fcArgs = functionCall.args || {}; const fcArgs = functionCall.args || {};
const toolCallIndex = state.functionIndex++; const toolCallIndex = state.functionIndex++;
const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`;
if (signature) {
storeGeminiThoughtSignature(callId, signature, state.sessionId);
}
const toolCall = { const toolCall = {
id: `${fcName}-${Date.now()}-${toolCallIndex}`, id: callId,
index: toolCallIndex, index: toolCallIndex,
type: OPENAI_BLOCK.FUNCTION, type: OPENAI_BLOCK.FUNCTION,
function: { name: fcName, arguments: JSON.stringify(fcArgs) }, function: { name: fcName, arguments: JSON.stringify(fcArgs) },
@@ -57,13 +62,21 @@ export function geminiToOpenAIResponse(chunk, state) {
if (content?.parts) { if (content?.parts) {
for (const part of content.parts) { for (const part of content.parts) {
const hasThoughtSig = part.thoughtSignature || part.thought_signature; const hasThoughtSig = part.thoughtSignature || part.thought_signature;
if (hasThoughtSig && typeof hasThoughtSig === "string") {
state.pendingThoughtSignature = hasThoughtSig;
}
const isThought = part.thought === true; const isThought = part.thought === true;
// Handle thought signature (thinking mode) // Handle thought signature (thinking mode)
if (hasThoughtSig) { if (hasThoughtSig) {
const hasTextContent = part.text !== undefined && part.text !== ""; const hasTextContent = part.text !== undefined && part.text !== "";
const hasFunctionCall = !!part.functionCall; const hasFunctionCall = !!part.functionCall;
// Standalone thoughtSignature part (no text, no functionCall): keep pending for next functionCall
if (!hasTextContent && !hasFunctionCall) {
continue;
}
if (hasTextContent) { if (hasTextContent) {
results.push(buildChunk( results.push(buildChunk(
chunkMeta(state), chunkMeta(state),
@@ -71,9 +84,10 @@ export function geminiToOpenAIResponse(chunk, state) {
null null
)); ));
} }
if (hasFunctionCall) { if (hasFunctionCall) {
results.push(emitFunctionCall(part.functionCall, state)); results.push(emitFunctionCall(part.functionCall, state, hasThoughtSig));
state.pendingThoughtSignature = null;
} }
continue; continue;
} }
@@ -92,7 +106,9 @@ export function geminiToOpenAIResponse(chunk, state) {
// Function call // Function call
if (part.functionCall) { if (part.functionCall) {
results.push(emitFunctionCall(part.functionCall, state)); const sig = state.pendingThoughtSignature || null;
results.push(emitFunctionCall(part.functionCall, state, sig));
state.pendingThoughtSignature = null;
} }
// Inline data (images) // Inline data (images)

View File

@@ -446,6 +446,13 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
state.created = Math.floor(Date.now() / 1000); state.created = Math.floor(Date.now() / 1000);
state.toolCallIndex = 0; state.toolCallIndex = 0;
state.currentToolCallId = null; state.currentToolCallId = null;
// item_id → chat tool_calls index. Deltas carry item_id; keying on it (not
// stream position) keeps parallel calls separate when upstream emits all
// output_item.added events before any done/delta. Lazily created so callers
// that build their own state object (stream.js) need no changes.
state.respToolChatIndex ??= new Map();
// Indices that already received argument deltas (guards done-with-args).
state.respToolArgsEmitted ??= new Set();
} }
// Text content delta // Text content delta
@@ -464,16 +471,29 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
return null; return null;
} }
// Function call started (standard function_call or custom_tool_call) // Function call started (standard function_call or custom_tool_call).
// Index is assigned here (not on done): attributing deltas by stream position
// merges parallel calls into index 0 whenever upstream emits all addeds
// before dones — the client then concatenates N JSON payloads into one
// tool input and fails validation. The server item id is the correlator.
if (eventType === "response.output_item.added" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) { if (eventType === "response.output_item.added" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) {
const item = data.item; const item = data.item;
state.currentToolCallId = item.call_id || fallbackToolCallId(); state.currentToolCallId = item.call_id || fallbackToolCallId();
state.respToolChatIndex ??= new Map();
const key = item.id || data.item_id || state.currentToolCallId;
let idx;
if (key && state.respToolChatIndex.has(key)) {
idx = state.respToolChatIndex.get(key); // duplicate added (retry) — reuse
} else {
idx = state.toolCallIndex++;
if (key) state.respToolChatIndex.set(key, idx);
}
return buildChunk( return buildChunk(
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
{ {
tool_calls: [{ tool_calls: [{
index: state.toolCallIndex, index: idx,
id: state.currentToolCallId, id: state.currentToolCallId,
type: OPENAI_BLOCK.FUNCTION, type: OPENAI_BLOCK.FUNCTION,
function: { name: item.name || "", arguments: "" } function: { name: item.name || "", arguments: "" }
@@ -482,20 +502,39 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
); );
} }
// Function call arguments delta (standard or custom_tool_call variant) // Function call arguments delta (standard or custom_tool_call variant).
// Routed by item_id so interleaved parallel fragments stay on their own call.
if (eventType === "response.function_call_arguments.delta" || eventType === "response.custom_tool_call_input.delta") { if (eventType === "response.function_call_arguments.delta" || eventType === "response.custom_tool_call_input.delta") {
const argsDelta = data.delta || ""; const argsDelta = data.delta || "";
if (!argsDelta) return null; if (!argsDelta) return null;
const known = data.item_id ? state.respToolChatIndex?.get(data.item_id) : undefined;
const idx = known ?? Math.max(0, (state.toolCallIndex || 1) - 1);
state.respToolArgsEmitted ??= new Set();
state.respToolArgsEmitted.add(idx);
return buildChunk( return buildChunk(
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
{ tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsDelta } }] } { tool_calls: [{ index: idx, function: { arguments: argsDelta } }] }
); );
} }
// Function call done (standard or custom_tool_call variant) // Function call done (standard or custom_tool_call variant).
// Index was assigned at added-time; nothing to advance. Some upstreams send
// complete arguments only here (no deltas) — emit them once in that case.
if (eventType === "response.output_item.done" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) { if (eventType === "response.output_item.done" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) {
state.toolCallIndex++; const key = data.item?.id || data.item_id;
const idx = (key && state.respToolChatIndex?.get(key)) ?? Math.max(0, (state.toolCallIndex || 1) - 1);
const fullArgs = data.item?.arguments;
if (typeof fullArgs === "string" && fullArgs) {
state.respToolArgsEmitted ??= new Set();
if (!state.respToolArgsEmitted.has(idx)) {
state.respToolArgsEmitted.add(idx);
return buildChunk(
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
{ tool_calls: [{ index: idx, function: { arguments: fullArgs } }] }
);
}
}
return null; return null;
} }

View File

@@ -20,6 +20,8 @@ export const CLAUDE_BLOCK = {
TOOL_RESULT: "tool_result", TOOL_RESULT: "tool_result",
THINKING: "thinking", THINKING: "thinking",
REDACTED_THINKING: "redacted_thinking", REDACTED_THINKING: "redacted_thinking",
SERVER_TOOL_USE: "server_tool_use",
WEB_SEARCH_TOOL_RESULT: "web_search_tool_result",
}; };
// OpenAI Responses API item types. // OpenAI Responses API item types.

Some files were not shown because too many files have changed in this diff Show More