Compare commits
177 Commits
gitea/feat
...
2a37a4085e
| Author | SHA1 | Date | |
|---|---|---|---|
| 2a37a4085e | |||
| e2f8323ab1 | |||
| 9bd7adc556 | |||
| 34a78f579e | |||
| c8b96a61e7 | |||
| e438a03f96 | |||
| 008e0ef311 | |||
| 5058a402f1 | |||
| 9b27ee2611 | |||
| c5ce1ef140 | |||
| fcd3dcb409 | |||
| 067f18aaa1 | |||
| 88faba150a | |||
| 0dbae80930 | |||
|
|
6fcd27337a | ||
|
|
9be6588cc8 | ||
|
|
1319dea620 | ||
|
|
31df0635aa | ||
|
|
baf3356583 | ||
|
|
24fd165b0d | ||
|
|
a8313cd322 | ||
|
|
f8e8039446 | ||
|
|
e3e3e235f6 | ||
|
|
0afe949387 | ||
|
|
5e59790824 | ||
|
|
16cb40fda1 | ||
|
|
44c7b34837 | ||
|
|
15dfd86416 | ||
|
|
8b0fcf4b16 | ||
|
|
6d96e24bd9 | ||
|
|
3b14bf4a49 | ||
|
|
9c9dd7b191 | ||
|
|
f17a68aaee | ||
|
|
6eaa9f8369 | ||
|
|
65ac9b3cec | ||
|
|
72ec06a81d | ||
|
|
de2da19a9e | ||
|
|
aa0448f7e2 | ||
|
|
41c9e6be87 | ||
|
|
8e04fe1734 | ||
|
|
783e271c16 | ||
|
|
3c17d3406b | ||
|
|
007d372724 | ||
|
|
e45bd73d6e | ||
|
|
c85a5c57ba | ||
|
|
57b3b2c175 | ||
|
|
53a8b5ed55 | ||
|
|
039c4dbc72 | ||
|
|
79918c7830 | ||
|
|
6994cd1f70 | ||
|
|
4f48ab8c7f | ||
|
|
c97963c4fb | ||
|
|
cef5dd4d61 | ||
|
|
d587b2a487 | ||
|
|
7c7fae3955 | ||
|
|
9ba8f37486 | ||
|
|
55628eea02 | ||
|
|
c4a120af8f | ||
|
|
eb00222c4f | ||
|
|
43d4abbcf2 | ||
|
|
e0ba667450 | ||
|
|
0513bf393f | ||
| d826e39008 | |||
| 2897cc3972 | |||
|
|
ccb0842d0a | ||
|
|
68566f53dc | ||
|
|
bc252ea802 | ||
|
|
de680e789f | ||
|
|
6acc3bb965 | ||
|
|
8b9cac180e | ||
|
|
30d0f6d3d8 | ||
|
|
59b7828237 | ||
|
|
d6761c6fb0 | ||
|
|
02ccdc2d22 | ||
|
|
9c58ba645e | ||
|
|
70e8dc4974 | ||
|
|
b94685b80d | ||
|
|
0248dd5348 | ||
|
|
27b37705b3 | ||
|
|
7dfb346667 | ||
|
|
a6a41dfb3c | ||
|
|
2629218b04 | ||
|
|
c9926897ba | ||
|
|
88a8c72d2d | ||
|
|
ba508f2506 | ||
|
|
a077ee85bd | ||
|
|
e567ba800f | ||
|
|
542a088c04 | ||
|
|
9173c29b66 | ||
|
|
eceac9d7ae | ||
| ab9a3c1d43 | |||
|
|
837cfec5a9 | ||
| b1d368d960 | |||
|
|
f89ba32d79 | ||
|
|
9845a1702f | ||
|
|
a625ea9fd8 | ||
|
|
b61c50cbb7 | ||
|
|
baafc74c1f | ||
|
|
bb314118f2 | ||
|
|
2d515c8abc | ||
|
|
74d5fedf79 | ||
|
|
dcf1927f22 | ||
|
|
e1f3399b73 | ||
|
|
f1f9d27061 | ||
|
|
90df008f0c | ||
|
|
d2599ebf17 | ||
|
|
5cdcf67484 | ||
|
|
b25e10160d | ||
|
|
0270f6ea70 | ||
|
|
ce6bdf7fc2 | ||
|
|
a11937cdd6 | ||
|
|
c73c419d09 | ||
|
|
a3b267a5cb | ||
|
|
65c65a0f56 | ||
|
|
7610f28f42 | ||
|
|
0d4d4bc261 | ||
|
|
3a7a878f91 | ||
|
|
e79f9eddb4 | ||
|
|
b9e2611045 | ||
|
|
ddd5509e97 | ||
|
|
288940960a | ||
|
|
d75471bbbc | ||
|
|
0c55d49ab6 | ||
|
|
cfbdf06047 | ||
|
|
a4c5fa4e14 | ||
|
|
20b442b708 | ||
|
|
71cd5b2f23 | ||
|
|
b10b807063 | ||
|
|
081c6f2aff | ||
|
|
19281b5524 | ||
|
|
bbae990b92 | ||
|
|
8c068a1f5c | ||
|
|
97a6708651 | ||
|
|
a3cd7c82bc | ||
|
|
bf7da67859 | ||
|
|
da0149de97 | ||
|
|
1885ad7f64 | ||
|
|
481e7e467b | ||
|
|
008de32c06 | ||
|
|
b6454d84da | ||
|
|
46e6c01a01 | ||
|
|
5041494e1c | ||
|
|
4dadab9d5f | ||
|
|
7f436e2792 | ||
|
|
54e3245ace | ||
|
|
960f8a0379 | ||
|
|
5cc4f222f8 | ||
|
|
cd557a2552 | ||
|
|
ced51ed62f | ||
|
|
cb0135b695 | ||
|
|
abc0add031 | ||
|
|
9102c4c6d8 | ||
|
|
ce6120ce7b | ||
|
|
7afaecd617 | ||
|
|
a5363b83b5 | ||
|
|
b08751c4ea | ||
|
|
76752a4396 | ||
|
|
602ee4054b | ||
|
|
8f81f17b99 | ||
|
|
182c849979 | ||
|
|
0b3c794075 | ||
|
|
a9785a5f70 | ||
|
|
373850ee36 | ||
|
|
749c2e3f9c | ||
|
|
7fa2e7f029 | ||
|
|
8d1db46beb | ||
|
|
9e3866658a | ||
|
|
8a664d619d | ||
|
|
2d94fffe3b | ||
|
|
319caa2d7b | ||
|
|
95bfc64f06 | ||
|
|
eff81b1242 | ||
|
|
b66b5c68ce | ||
|
|
fc8722e897 | ||
|
|
713c563765 | ||
|
|
3d20a4ccd2 | ||
|
|
526235872a |
@@ -14,10 +14,6 @@ NODE_ENV=production
|
||||
API_KEY_SECRET=endpoint-proxy-api-key-secret
|
||||
MACHINE_ID_SALT=endpoint-proxy-salt
|
||||
ENABLE_REQUEST_LOGS=false
|
||||
# Console verbosity: DEBUG | INFO | WARN | ERROR. Default INFO. In production set
|
||||
# ERROR to only print important errors (hides ▶ POST / 📊 DONE / [COMBO] / [CHAT]).
|
||||
# Can also be changed at runtime from dashboard Settings → Logging.
|
||||
# LOG_LEVEL=ERROR
|
||||
OBSERVABILITY_ENABLED=true
|
||||
AUTH_COOKIE_SECURE=false
|
||||
REQUIRE_API_KEY=false
|
||||
|
||||
18
.gitignore
vendored
18
.gitignore
vendored
@@ -1,4 +1,5 @@
|
||||
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
|
||||
|
||||
# dependencies
|
||||
/node_modules
|
||||
/.pnp
|
||||
@@ -8,8 +9,10 @@
|
||||
!.yarn/plugins
|
||||
!.yarn/releases
|
||||
!.yarn/versions
|
||||
|
||||
# testing
|
||||
/coverage
|
||||
|
||||
# next.js
|
||||
/.next/
|
||||
/.next-cli-build/
|
||||
@@ -19,22 +22,28 @@ product
|
||||
# production
|
||||
/build
|
||||
.idea/
|
||||
|
||||
# misc
|
||||
.DS_Store
|
||||
*.pem
|
||||
|
||||
# debug
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
.pnpm-debug.log*
|
||||
|
||||
# env files (can opt-in for committing if needed)
|
||||
.env*
|
||||
!.env.example
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
next-env.d.ts
|
||||
|
||||
.bin/*
|
||||
data/
|
||||
logs/*
|
||||
@@ -52,18 +61,23 @@ Thanks.md
|
||||
PUBLIC.en.md
|
||||
PR/*
|
||||
package-lock.json
|
||||
|
||||
|
||||
#Ignore vscode AI rules
|
||||
.github/instructions/codacy.instructions.md
|
||||
README1.md
|
||||
deploy*.sh
|
||||
ecosystem.config.*
|
||||
|
||||
scripts/agSniffer/*
|
||||
gitbooks/*
|
||||
gitbook/README.md
|
||||
|
||||
# Refactor backup reference (do not bundle/lint)
|
||||
open-sse.old/
|
||||
.graphifyignore
|
||||
graphify-out/*
|
||||
|
||||
# Local-only working dirs (notes, vendored repos, scripts, skills)
|
||||
.claude/
|
||||
.docs/
|
||||
@@ -72,8 +86,6 @@ graphify-out/*
|
||||
.codegraph/
|
||||
.PR/
|
||||
.next-analyze/*
|
||||
|
||||
# CommandCode CLI local state (auth/taste/projects)
|
||||
.commandcode/
|
||||
|
||||
# Pi subagent run artifacts
|
||||
.pi-subagents/
|
||||
|
||||
220
CHANGELOG.md
220
CHANGELOG.md
@@ -1,172 +1,17 @@
|
||||
# v0.5.55 (2026-08-14)
|
||||
# Unreleased
|
||||
|
||||
## Features
|
||||
- **Auth**: native SAML 2.0 SSO alongside OIDC — AuthnRequest generation, ACS
|
||||
assertion handling, SP metadata export, admin config test, replay-protected
|
||||
via a `saml_state` cookie matched against `InResponseTo`
|
||||
- **Providers**: add Alibaba Token Plan (`token-plan.ap-southeast-1`) — the
|
||||
fourth Alibaba key type, Singapore-only and OpenAI-compatible transport only
|
||||
- **Providers**: add `glm-5.3` to GLM Coding and GLM (China)
|
||||
- **Providers**: Kimchi accepts API keys as well as OAuth (dual auth), with a
|
||||
working Test Connection for both modes
|
||||
- **Antigravity**: add Gemini 3.7 Flash and its tiered high/medium/low variants
|
||||
(also in the Gemini registry) with pricing and quota tracking
|
||||
- **TTS**: add Fish Audio — model id travels in an HTTP `model` header, voice
|
||||
is a `reference_id` (preset or cloned voice model)
|
||||
- **OpenCode-Go**: route by request format via declared transports instead of
|
||||
forcing every client into `/messages` — Codex/OpenAI clients no longer pay a
|
||||
lossy Responses→OpenAI→Claude double translation. Per-model `supportedFormats`
|
||||
guard; the bespoke executor is gone (its shared `_lastModel` cache could cross
|
||||
auth headers between concurrent requests)
|
||||
- **Usage**: dedup + cache Claude quota calls (120s TTL keyed by access token,
|
||||
in-flight promise dedup, last-good read on soft failure) to stop multiple
|
||||
tabs tripping 429; manual refresh (↻) sends `force=1` to bypass the cache
|
||||
|
||||
- **Stream error patterns**: per-provider `streamErrorPatterns` setting (UI: provider page → Stream Error Patterns) — HTTP-200 streams whose first bytes match configured patterns (plain text or `/regex/`) are treated as failed requests: fallback works for non-streaming and early stream errors, and late streaming errors are logged as FAILED. Zero overhead when unconfigured.
|
||||
|
||||
## Fixes
|
||||
- **Docker**: ship `sql.js` in the image so the pure-JS DB fallback can start —
|
||||
file tracing carried the package's JS without `dist/sql-wasm.wasm`, so a
|
||||
container with no native driver aborted with ENOENT and never got a database
|
||||
(#3248)
|
||||
- **Usage**: read Gemini `usageMetadata` out of the antigravity `{ response }`
|
||||
envelope — every non-streaming antigravity request logged `IN 0 | OUT 0`
|
||||
(#3260)
|
||||
- **Claude**: re-anchor passthrough cache breakpoints — the client's own
|
||||
`cache_control` markers point at pre-normalization offsets, so the tail was
|
||||
re-cached every request. Last system block and last tool pinned at 1h TTL,
|
||||
last assistant turn at 5m, mid-conversation system messages folded into the
|
||||
neighbouring user turn instead of hoisted into `body.system`
|
||||
- **Combos**: detect images from Hermes and attachment payloads (`images[]`,
|
||||
`experimental_attachments`, message-level `image_url`/`audio_url`, inline
|
||||
`data:` URIs) so the Vision Adapter auto-switch fires for Hermes/Ollama/
|
||||
Vercel AI SDK shapes
|
||||
- **Kiro**: intercept chat via `x-amz-target` — Kiro IDE 1.0.228+ moved
|
||||
`GenerateAssistantResponse` to `POST /` + header, bypassing MITM. Also emit
|
||||
the now-mandatory initial-response frame and map the `auto` model slot
|
||||
- **Kiro**: report real output tokens and stop discarding usable turns
|
||||
- **Qoder**: detect billing blocks at stream start and return a synthetic 403
|
||||
so combo/account fallback triggers instead of leaking the error into chat
|
||||
- **Antigravity**: strip competitive system prompts (Zed IDE's Claude-agent
|
||||
prompt) that Antigravity flags with a 429 Quota Exhausted
|
||||
- **OpenCode**: send the official client fingerprint on free-tier requests so
|
||||
the Console stops classifying traffic as unidentified and rate-limiting it;
|
||||
session id resolves conversation-stable to preserve prompt caching
|
||||
- **Responses**: don't close the message on an empty `tool_calls` array — some
|
||||
providers attach one to every chunk, and the truthy check ended the message
|
||||
on the first content token (#3234)
|
||||
- **Translator**: preserve `prompt_cache_key` when converting chat to responses
|
||||
- **Models**: expose snake_case token limits on `/v1/models`
|
||||
- **Combos**: strip `stream_options` from the Fusion panel fan-out to avoid a
|
||||
DeepSeek 400 (#3024); raise the dashboard model-test probe budget to 1024 and
|
||||
soft-pass reasoning-only responses (#3010)
|
||||
- **Headroom**: the toggle reflects the `headroomEnabled` setting even when the
|
||||
proxy is down — it previously showed OFF while the engine kept calling
|
||||
`/v1/compress`; proxy status stays visible via the status chip
|
||||
- **Hermes**: add the `api_key` parameter to the model block in YAML config
|
||||
- **Providers**: add llm7 to provider test support
|
||||
|
||||
## Docs
|
||||
- **i18n**: add Spanish, French, and Brazilian Portuguese README translations
|
||||
|
||||
## Security
|
||||
- **Real IP**: `x-9r-real-ip` and the Host fallback were trusted from
|
||||
client-controlled headers whenever `custom-server.js` was not in the request
|
||||
path (`npm run start`, `start:bun`), letting a remote caller pose as local to
|
||||
skip API key auth and reach `LOCAL_ONLY_PATHS` (`/api/mcp/*`,
|
||||
`/api/tunnel/enable`, `/api/auth/reset-password`). The server now stamps a
|
||||
per-process `x-9r-peer-token` on every request it sanitizes and only trusts
|
||||
`x-9r-real-ip` behind it — falling back to Host in development and failing
|
||||
closed in production (GHSA-pjm4-8fpg-f9p6). Also fixes IPv6 loopback
|
||||
detection (`::1`, `::ffff:127.0.0.1`) and routes `npm run start` /
|
||||
`start:bun` through `custom-server.js`
|
||||
- **Search**: `resolveBaseUrl()` rejects client-supplied non-public baseUrls
|
||||
(SSRF guard on `/v1/search`)
|
||||
- **Login**: fresh-install remote login with the default password returns 403
|
||||
without issuing a JWT
|
||||
- **Usage**: `/api/usage/request-details` redacts request/response payloads
|
||||
|
||||
# v0.5.50 (2026-08-05)
|
||||
|
||||
## Features
|
||||
- **Providers**: add TokenRouter (300+ models via OpenAI-compatible gateway) with
|
||||
exact per-model pricing for 110 models and `reasoning_effort` thinking config
|
||||
- **Providers**: add Self-hosted STT / TTS / Embedding — point 9Router at your own
|
||||
OpenAI-compatible speech and embedding servers (whisper.cpp, faster-whisper,
|
||||
Kokoro-FastAPI, llama-server, vLLM, Infinity). Unlike the named cloud providers
|
||||
these read `baseUrl` per connection, so one provider can front several machines
|
||||
- **Combos**: default-enable vision/audio capacity adapter (auto-routes to a
|
||||
vision/audio-capable model when the target lacks that capability, falling back
|
||||
to `oc/mimo-v2.5-free`), wired into chat handler routing
|
||||
- **Endpoint**: auto-provision a "Default Key" for first-time users so `/v1`
|
||||
works without a manual dashboard step
|
||||
- **Codex**: support GPT-5.6 Max/Ultra reasoning-level overrides (cx/ routes only)
|
||||
- **Qoder**: support PAT (Personal Access Token) connections end-to-end, alongside
|
||||
OAuth device flow
|
||||
- **CLI tools**: add OpenDesign (manalkaff/opendesign) support
|
||||
- **Headroom**: report effective payload savings (tool schema/history bytes broken
|
||||
out, byte-savings % reflects actual outbound reduction)
|
||||
- **Ollama**: Cloud quota tracker (session + weekly) + proactive background OAuth
|
||||
token refresh scheduler for all providers
|
||||
|
||||
## Fixes
|
||||
- **Providers**: remove Qwen (OAuth flow stopped working reliably)
|
||||
- **Passthrough**: detect codex-tui/Codex Desktop as native Codex client — they
|
||||
were falling through to the translator and losing fields like `reasoning.summary`
|
||||
- **OAuth**: scope antigravity header fixes to loadCodeAssist/onboardUser only
|
||||
- **OAuth**: keep `open` external in the build so xAI/Grok token refresh works on
|
||||
Windows
|
||||
- **OAuth**: declare missing `searchParams` in register-session handler (was a
|
||||
500 instead of JSON on error)
|
||||
- **DB**: `ENABLE_REQUEST_LOGS` env var now overrides the UI setting correctly;
|
||||
observability defaults to off (opt-in)
|
||||
- **Translator**: preserve Codex Responses Lite tool use across chat-native
|
||||
OpenAI-compatible providers
|
||||
- **Translator**: don't drop image-only user messages in `prepareClaudeRequest`
|
||||
- **Translator**: drop JSON Schema keywords Gemini rejects (`uniqueItems`,
|
||||
`contains`, `multipleOf`, `unevaluatedProperties`, `unevaluatedItems`,
|
||||
`contentSchema`)
|
||||
- **Claude**: remove global header cache that leaked one client's identity
|
||||
headers onto another client/account sharing the server; gate `anthropic-beta`
|
||||
by model instead
|
||||
- **Antigravity**: drop retired Gemini 3.0 quota tiers, show Gemini 3.6 Flash
|
||||
usage bars
|
||||
- **Cloudflare AI**: declare API key authentication (dashboard showed "No
|
||||
connections" despite an active key)
|
||||
- **GitHub Copilot**: hold monthly-exhausted accounts until UTC month reset
|
||||
instead of only cooling down 120s
|
||||
- **CodeBuddy**: dodge Tencent CN content filter, add usage tracking, normalize
|
||||
codebuddy-intl messages
|
||||
- **Usage**: stop losing cached prompt tokens in the forced-SSE→JSON path
|
||||
- **Grok CLI**: display the public subscription tier from the OAuth token claim
|
||||
- **Providers**: count apikey connections for Ollama free-tier card; free-tier/
|
||||
apikey providers without `authModes` now default to apikey (were treated
|
||||
oauth-only)
|
||||
- **Build**: include static/public assets in standalone output (login page hung
|
||||
on 404s when run via PM2)
|
||||
- **Server**: support IntelliJ IDEA OpenAI-compatible clients over HTTP (h2c
|
||||
upgrade handling)
|
||||
- **Auth**: redirect already-logged-in sessions away from `/login`
|
||||
- **CLI tools**: enable Apply button for dynamic OpenAI/Anthropic-compatible
|
||||
provider connections
|
||||
- **CLI**: include complete API artifacts in the CLI package
|
||||
- **TTS**: a bare self-hosted model name is the MODEL, not the voice — `kokoro`
|
||||
was parsed as a voice against a default model, 404ing or synthesising with the
|
||||
wrong one
|
||||
- **Embeddings**: self-hosted embeddings no longer fall back to `api.openai.com`
|
||||
when a connection has no `baseUrl` — that silently sent the input text and API
|
||||
key to OpenAI under a provider named "Self-hosted"
|
||||
- **Embeddings**: an adapter that rejects a misconfigured connection now returns
|
||||
400 with the reason instead of escaping the handler uncaught
|
||||
- **Embeddings**: bound the upstream fetch with `FETCH_CONNECT_TIMEOUT_MS` — an
|
||||
endpoint that drops packets never returns headers, so the request previously
|
||||
hung indefinitely
|
||||
|
||||
## Docs
|
||||
- **i18n**: fix port typo, add RTK Token Saver feature descriptions
|
||||
- **CommandCode**: in-stream `{"type":"error"}` events now emit OpenAI error chunks + an executor early-peek → 502 fallback instead of fake success content (`[CommandCode error: ...]`).
|
||||
|
||||
# v0.5.45 (2026-07-30)
|
||||
|
||||
## Features
|
||||
- **TTS**: add Xiaomi MiMo text-to-speech (preset voices 冰糖/茉莉/苏打/白桦/Mia/Chloe/Milo/Dean, style control, language hint dropdown with Auto-detect, i18n for Style label/placeholder)
|
||||
|
||||
- **Providers**: add Poolside (OpenAI-compatible)
|
||||
- **Providers**: add api-airforce, baidu, bazaarlink, bluesminds, kilo-gateway, llm7, morph, sambanova, tencent
|
||||
- **OAuth**: zed / trae / windsurf providers + harden callback proxies
|
||||
@@ -179,6 +24,7 @@
|
||||
- **Usage**: SuperGrok weekly pool via gRPC-web
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Refresh**: rotate `refresh_token` between retry attempts
|
||||
- **Kiro**: canonicalize tool history and route API keys correctly
|
||||
- **Kiro**: normalize dashboard thinking intensity models
|
||||
@@ -193,18 +39,21 @@
|
||||
- **Dashboard**: flex quota rows, thin global scrollbars, no hidden-row overflow
|
||||
|
||||
## Docs
|
||||
|
||||
- **i18n**: expand pt-BR translation to 986 terms
|
||||
- README: Indonesian translation
|
||||
|
||||
# v0.5.40 (2026-07-20)
|
||||
|
||||
## Features
|
||||
|
||||
- **i18n**: add Khmer (km) translations
|
||||
- **CLI tools**: configure Grok Build subagent models
|
||||
- **Kimi**: merge OAuth into dual-auth provider, add K3 / K2.7 models
|
||||
- **Dashboard**: ProviderTopology flow animation
|
||||
|
||||
## Fixes
|
||||
|
||||
- **DB**: resolve better-sqlite3 parameter binding crash
|
||||
- **Translator**: pass `service_tier` through OpenAI → Responses conversion
|
||||
- **Kiro**: map GPT-5.6 reasoning effort fields
|
||||
@@ -215,10 +64,10 @@
|
||||
- **Cursor**: HTTP/2 AgentService support + version bump 3.12.17
|
||||
- **Dashboard**: cut duplicate API/icon spam, lazy-load provider assets
|
||||
|
||||
|
||||
# v0.5.35 (2026-07-16)
|
||||
|
||||
## Features
|
||||
|
||||
- **xAI**: Grok Imagine video generation (`/v1/videos`) + CLI
|
||||
- **CLI tools**: Grok Build setup — choose separate main/general-purpose/explore/plan models and preserve each model's context window
|
||||
- **GitHub Copilot**: route Claude models through Copilot's native `/v1/messages`
|
||||
@@ -229,6 +78,7 @@
|
||||
- **i18n**: Thai (th) + Persian (fa) translations / README
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Providers**: bulk-add API keys no longer overwrite existing keys (gap-fill `Key N`)
|
||||
- **Anthropic**: lowercase `anthropic-version` header to prevent duplication on `/v1/messages`
|
||||
- **Alicode-intl**: use DashScope compatible-mode endpoint so standard keys work
|
||||
@@ -241,14 +91,17 @@
|
||||
- **Translator**: strip `client_metadata` when converting openai-responses → openai
|
||||
|
||||
## Improvements
|
||||
|
||||
- **Perf**: skip inactive background services on startup
|
||||
|
||||
## Docs
|
||||
|
||||
- README: Persian YouTube tutorial
|
||||
|
||||
# v0.5.30 (2026-07-10)
|
||||
|
||||
## Features
|
||||
|
||||
- **Perplexity**: add Agent API provider (#2492)
|
||||
- **Grok CLI**: add Grok CLI / Grok Build provider with OAuth device-code flow (#2502)
|
||||
- **Featherless**: add OpenAI-compatible provider presets
|
||||
@@ -260,6 +113,7 @@
|
||||
- **Proxy-Pools**: auto-rotate strategy for no-auth providers (#2409)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Cloudflare-AI**: support accountId in bulk key import (#2449)
|
||||
- **DB**: backup on schema change, MCP child cleanup, codex models, usage providers OOM
|
||||
- **Codex**: avoid bare-email OAuth dedup (#2477)
|
||||
@@ -276,6 +130,7 @@
|
||||
- **Pricing**: update Claude/Codex model rates and add new models
|
||||
|
||||
## Improvements
|
||||
|
||||
- **i18n(zh-CN)**: complete Chinese translations for all UI strings (#2436)
|
||||
- **API**: caching for tunnel and version status endpoints
|
||||
- **Perf**: faster dev startup and lighter bundle
|
||||
@@ -283,12 +138,14 @@
|
||||
# v0.5.20 (2026-07-07)
|
||||
|
||||
## Features
|
||||
|
||||
- **Thinking**: per-model thinking level picker on provider page — appends `(level)` suffix to copied model names for forced reasoning effort across all formats (openai, claude, gemini, deepseek, kimi, qwen, zai, minimax, hunyuan, step)
|
||||
- **RTK**: add JS-native git-log filter (#2423)
|
||||
- **Caveman**: add targeted upstream-aligned style rules (#2424)
|
||||
- **i18n**: add Farsi (fa) language support (#2385)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Thinking**: strip `(level)` suffix from upstream `body.model` so providers no longer reject requests
|
||||
- **Translator**: preserve developer instructions in openai-responses conversion (#2434)
|
||||
- **count_tokens**: count structured Anthropic blocks (#2419)
|
||||
@@ -302,12 +159,14 @@
|
||||
# v0.5.18 (2026-07-03)
|
||||
|
||||
## Features
|
||||
|
||||
- **Usage**: track cached tokens + correct input/output/cache cost (#2209) — hodtien
|
||||
- **Codex**: show reset credit expiry details (#2290) — Rafli Ahmad Zulfikar
|
||||
- **NVIDIA**: add new models and capabilities — decolua
|
||||
- **ClinePass**: add provider support — sternelee
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Usage**: dedupe streaming request-details log entries — Qin Li
|
||||
- **Claude**: drop foreign thinking signatures in passthrough — decolua
|
||||
- Prevent non-SSE stream pipe crash and cross-IdP account overwrites (#2244) — KunN-21
|
||||
@@ -324,11 +183,13 @@
|
||||
# v0.5.15 (2026-06-29)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Kimchi OAuth provider — Nant361
|
||||
- Refine Qwen vision/video + thinking model patterns — decolua
|
||||
- Opt-in Codex auto-ping quota keep-alive — Emirhan
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Responses**: handle response.done terminal events (#2142) — rifuki
|
||||
- **Headroom**: skip unsafe responses tool history (#2132) — Sutarto Jordan Chrisfivo
|
||||
- **Translator**: map mid-conversation system message to user (claude→openai) — decolua
|
||||
@@ -345,6 +206,7 @@
|
||||
# v0.5.12 (2026-06-26)
|
||||
|
||||
## Features
|
||||
|
||||
- Add token-saver dashboard page — decolua
|
||||
- Add bulk delete for provider connections — teddytkz
|
||||
- Resolve GitHub Copilot model catalog from upstream — caiqinzhou
|
||||
@@ -353,6 +215,7 @@
|
||||
- Overhaul Blackbox provider catalog + WebUI test support — suryacagur
|
||||
|
||||
## Fixes
|
||||
|
||||
- Provider thinking compatibility (DeepSeek/Gemini) — Mink Nguyen
|
||||
- Stop double-counting streaming usage at source — decolua
|
||||
- Usage logging dedupe to reduce stats churn — Mink Nguyen
|
||||
@@ -381,11 +244,13 @@
|
||||
# v0.5.8 (2026-06-21)
|
||||
|
||||
## Features
|
||||
|
||||
- **Antigravity**: native image generation support (image models tagged kind:image, hiển thị trong media-providers UI)
|
||||
- **CodeBuddy CN**: API key auth + credit quota tracker
|
||||
- **CodeBuddy CN**: short model prefix alias "cbcn"
|
||||
|
||||
## Fixes
|
||||
|
||||
- **MiniMax-M3**: enable vision capability
|
||||
- **Headroom**: support Docker sidecar proxy
|
||||
- **Antigravity**: image executor fixes
|
||||
@@ -399,12 +264,14 @@
|
||||
# v0.5.6 (2026-06-20)
|
||||
|
||||
## Features
|
||||
|
||||
- **Ponytail**: minimalist code generation feature
|
||||
- **Headroom**: proxy lifecycle management + dashboard UI (one-click start/stop, install detection, status probing, token saver, claude↔openai shape conversion)
|
||||
- **CodeBuddy CN**: new OAuth provider (copilot.tencent.com) — 15-model catalog, /v2 inference, forced streaming, OpenAI-style reasoning
|
||||
- **OpenCode-Go**: align models with official endpoints; route Qwen 3.7 MiniMax via /v1/messages, GLM/Kimi/DeepSeek/MiMo via /chat/completions
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Anthropic-compatible validation**: use POST /v1/messages (GET /models not spec, false "invalid" for valid keys)
|
||||
- **CLI tools**: tolerate JSONC configs in all 8 settings routes (opencode, openclaw, kilo, droid, cowork, copilot, claude, cline)
|
||||
- **Gemini/Antigravity**: preserve 'pattern' in tool schema translation (glob/grep)
|
||||
@@ -415,6 +282,7 @@
|
||||
# v0.5.4 (2026-06-18)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Kiro**: honor thinking effort budgets
|
||||
- **AG/Kiro/Xiaomi**: provider fixes
|
||||
- **Combo/Fusion**: flatten tool history in panel calls to prevent 503
|
||||
@@ -424,6 +292,7 @@
|
||||
# v0.5.2 (2026-06-17)
|
||||
|
||||
## Features
|
||||
|
||||
- **Combo Fusion strategy** — fans the prompt out to all member models in parallel, then a configurable judge model synthesizes one final answer (quorum-grace, anonymized sources, graceful degradation)
|
||||
- **Per-combo strategy selector** — pick `fallback` / `round-robin` / `fusion` / `capacity` per combo (replaces the old round-robin toggle), with a judge picker for fusion
|
||||
- **Capacity auto-switch** — reorders models per request so images/PDFs route to capable models first
|
||||
@@ -431,6 +300,7 @@
|
||||
- **Claude auto-ping** — warms the 5h quota window right after reset so a fresh window starts immediately (per-connection toggle)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Claude 429**: stop hammering the OAuth usage endpoint — cache resetAt, throttle quota refresh to 3 min, cool down after a 429 (chat unaffected)
|
||||
- **Usage logs always empty**: missing `await` on `getAdapter()` in `getRecentLogs` made `/api/usage/logs` & `/api/usage/request-logs` return nothing
|
||||
- **Executors**: strip params unsupported by the provider/model (drops deprecated `temperature` for claude-opus-4 → Anthropic 400)
|
||||
@@ -442,11 +312,13 @@
|
||||
- **Security**: SSRF hardening on web fetch
|
||||
|
||||
## Internal
|
||||
|
||||
- Large **open-sse / translator refactor** (~40 commits): unified provider/model registry (LiteLLM-style `models[]` + `kind` field, 100 co-located registry files), single-sourced media/OAuth/refresh/token URLs, registry-based dispatch for usage & token-refresh, DRY translator concerns (buildUsage, encodeDataUri, finishReasonMap, chunkBuilder, reasoningDelta…), ESM-safe registry init, large-file splits, dead-code removal, and golden/no-regression test gates
|
||||
|
||||
# v0.4.80 (2026-06-13)
|
||||
|
||||
## Features
|
||||
|
||||
- Vercel AI Gateway: support embeddings, images and credit usage (#1183)
|
||||
- Add MiMo Free no-auth provider (#1789)
|
||||
- Vertex: support ADC `authorized_user` credential
|
||||
@@ -455,6 +327,7 @@
|
||||
- Kiro: enable multi-endpoint failover for GenerateAssistantResponse (#1722)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Security: re-auth on DB export/import + SSRF guard on web fetch
|
||||
- Auth: real client IP rate-limiting + remote default-password guard
|
||||
- Cerebras/Mistral: strip unsupported `client_metadata` from downstream requests (#1742)
|
||||
@@ -473,11 +346,13 @@
|
||||
- Dashboard: show provider node name instead of connection name in topology (#1770) + show explicit `kind="llm"` combos on combos page (#1684)
|
||||
|
||||
## Docs
|
||||
|
||||
- README: add Indonesian 9Router tutorial video (#1709)
|
||||
|
||||
# v0.4.71 (2026-06-06)
|
||||
|
||||
## Features
|
||||
|
||||
- Caveman: add wenyan classical Chinese levels and sync upstream prompts; locale-based visibility on endpoint page
|
||||
- i18n: endpoint exposure notice across multiple languages + Russian README
|
||||
- Antigravity: add gemini-3.5-flash-extra-low (Low) model
|
||||
@@ -486,6 +361,7 @@
|
||||
- MiniMax: add MiniMax-M3 + update Quota Tracker coding/CN (#1631)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Codex: harden streaming timeouts (stall/connect raised to 60s, configurable per-provider), accept `response.done` event, and always emit a terminal `response.failed` + `[DONE]` for Responses passthrough when a stream closes, stalls, or aborts before a terminal event — prevents codex clients from hanging (#1648, #1680, #1688, #1618)
|
||||
- Codex: durable OAuth refresh lifecycle (#1664)
|
||||
- Tunnel: skip virtual interfaces to prevent false netchange watchdog
|
||||
@@ -499,21 +375,25 @@
|
||||
- Model-test: route image/STT probes to their real endpoints, harden STT ping; add opencode-go + xiaomi-tokenplan to connection test (#1576, #1628)
|
||||
|
||||
## Improvements
|
||||
|
||||
- Dashboard: reorganize menu actions across sidebar/header/profile
|
||||
- Translator: add data-driven coverage, bug-exposing cases, and real provider smoke tests
|
||||
|
||||
# v0.4.66 (2026-05-29)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Qoder provider: device-flow OAuth, COSY signing, WAF-bypass body encoding, live model catalog, dashboard quota tracker, 11 models (#1372)
|
||||
- Add new models: Claude Opus 4.8 (Claude Code), GPT 5.4 Mini (Codex)
|
||||
|
||||
## Fixes
|
||||
|
||||
- DeepSeek thinking mode: echo `reasoning_content` back on follow-up/tool-call turns so OpenCode-free and custom providers no longer 400 with "reasoning_content must be passed back" (#1543)
|
||||
- Reasoning injector: match deepseek/kimi model ids case-insensitively (covers custom providers using capitalized model names)
|
||||
- OpenCode suggested-models: include free models without the `-free` suffix, e.g. `big-pickle` (#1535)
|
||||
|
||||
## Improvements
|
||||
|
||||
- Codex: trim sunset models, keep gpt-5.5 / gpt-5.4 / gpt-5.3-codex family, add gpt-5.4-mini
|
||||
- volcengine-ark: refresh model list (add DeepSeek-V4-Flash/Pro, drop EOL entries)
|
||||
- Lower stream stall timeout 35s → 30s for faster hang detection
|
||||
@@ -521,15 +401,18 @@
|
||||
# v0.4.63 (2026-05-26)
|
||||
|
||||
## Fixes
|
||||
|
||||
- GitHub Copilot: never route Gemini/Claude models to the `/responses` endpoint; prevents misleading "does not support Responses API" 400s (#1062)
|
||||
- proxyFetch: restore missing `Readable` import causing runtime `ReferenceError` in DNS-bypass fetch path
|
||||
|
||||
## Improvements
|
||||
|
||||
- Lower stream stall timeout from 60s → 35s for faster hang detection
|
||||
|
||||
# v0.4.62 (2026-05-26)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Codex: auto-retry when upstream drops mid-stream (no more hangs)
|
||||
- Codex: fix random 400/404 errors, tool-calling failures, and unstable prompt cache
|
||||
- MITM: support Antigravity 2.x
|
||||
@@ -541,25 +424,30 @@
|
||||
- Gemini CLI: reuse stored OAuth project IDs for quota checks and show clearer setup guidance when the project is missing (#1271, #1428)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Cloudflare Workers proxy deployer and pool integration (#1360)
|
||||
- Add Deno Deploy relays support and improved proxy pools dashboard layout (#1437)
|
||||
|
||||
## Improvements
|
||||
|
||||
- Refactor Tunnel into dedicated Cloudflare and Tailscale manager modules
|
||||
- Refactor tokenRefresh service with in-flight dedup to prevent refresh_token_reused errors
|
||||
|
||||
# v0.4.59 (2026-05-21)
|
||||
|
||||
## Fixes
|
||||
|
||||
- OAuth: fix login flow on Windows
|
||||
|
||||
# v0.4.58 (2026-05-21)
|
||||
|
||||
## Features
|
||||
|
||||
- xAI Grok provider (OAuth, API key, image)
|
||||
- Provider limits: paginated accounts with page size controls
|
||||
|
||||
## Fixes
|
||||
|
||||
- Tailscale: fix connection status on Windows (#1300)
|
||||
- Tunnel: fix false "checking" when tunnel URL is reachable
|
||||
- Stream: fix pipe errors on client disconnect/abort
|
||||
@@ -567,11 +455,13 @@
|
||||
# v0.4.55 (2026-05-18)
|
||||
|
||||
## Features
|
||||
|
||||
- Xiaomi MiMo Token Plan: region selector (Singapore / China / Europe) — keys are cluster-specific
|
||||
- Antigravity: risk confirmation dialog before first connection
|
||||
- Gemini CLI: surface upstream retry delay on 429 errors
|
||||
|
||||
## Fixes
|
||||
|
||||
- MITM: cannot kill process on macOS under sudo (lsof not found in PATH)
|
||||
- Stream: false-positive stall timeout on Claude reasoning / Kiro responses
|
||||
- Tunnel: cannot re-enable after disable (stuck state)
|
||||
@@ -580,16 +470,19 @@
|
||||
- Antigravity OAuth: metadata now matches the official client
|
||||
|
||||
## Improvements
|
||||
|
||||
- Gemini CLI: bump engine to 0.34.0
|
||||
- Re-hide `qwen` (OAuth EOL) and `iflow` (not ready) providers
|
||||
|
||||
# v0.4.52 (2026-05-17)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Vercel AI Gateway provider support (#1183)
|
||||
- rtk: Kiro format tool result compression — handle conversationState.history & currentMessage, preserve error results, ~13.6% savings (#1194)
|
||||
|
||||
## Fixes
|
||||
|
||||
- openclaw: normalize agent.model object form `{primary, fallbacks}` before .startsWith → fix TypeError & 'not configured' status (#1216)
|
||||
- Usage Details pagination: stay inside mobile viewport <640px (#1218)
|
||||
- Fix test model error
|
||||
@@ -599,6 +492,7 @@
|
||||
# v0.4.50 (2026-05-16)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Fix duplicate tray icon on macOS when hiding to tray
|
||||
- Fix tray not showing in background mode on macOS
|
||||
- Fix hide to tray broken on Windows/Linux
|
||||
@@ -607,11 +501,13 @@
|
||||
# v0.4.49 (2026-05-16)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Kiro provider support: full request/response translation, live model listing, reasoning content support
|
||||
- Add `buildOutput` RTK filter with autodetect for npm/yarn/cargo build logs
|
||||
- Add MITM warning notification in tray and dashboard
|
||||
|
||||
## Improvements
|
||||
|
||||
- Add modalities (input/output) to model configuration for OpenCode
|
||||
- Fix tray hide-to-tray: keep current process alive instead of spawning detached child (fixes macOS NSStatusItem ghost icon)
|
||||
- Fix tray kill: graceful shutdown with SIGTERM/SIGKILL escalation
|
||||
@@ -620,9 +516,11 @@
|
||||
- Update i18n across 32 languages
|
||||
|
||||
## Fixes
|
||||
|
||||
- Fix model check (test-models) blocked by dashboardGuard: pass machineId-based CLI token in internal self-calls
|
||||
|
||||
# v0.4.46 (2026-05-15)
|
||||
|
||||
## Breaking Changes
|
||||
|
||||
- Tunnel public URL changed — old tunnel links no longer work, please reconnect to get the new URL
|
||||
@@ -37,9 +37,6 @@ COPY --from=builder /app/src/mitm ./src/mitm
|
||||
COPY --from=builder /app/node_modules/node-forge ./node_modules/node-forge
|
||||
# Ensure `next` is available at runtime in case tracing did not include it.
|
||||
COPY --from=builder /app/node_modules/next ./node_modules/next
|
||||
# sql.js loads dist/sql-wasm.wasm by path at runtime; tracing only follows JS imports,
|
||||
# so the last-resort DB driver would abort with ENOENT on the missing binary.
|
||||
COPY --from=builder /app/node_modules/sql.js ./node_modules/sql.js
|
||||
|
||||
RUN mkdir -p /app/data && chown -R node:node /app && \
|
||||
mkdir -p /app/data-home && chown node:node /app/data-home && \
|
||||
|
||||
68
README.md
68
README.md
@@ -17,7 +17,7 @@
|
||||
|
||||
[🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Setup](#-setup-guide) • [🌐 Website](https://9router.com)
|
||||
|
||||
[🇧🇷 Português (Brasil)](./i18n/README.pt-BR.md) • [🇻🇳 Tiếng Việt](./i18n/README.vi.md) • [🇨🇳 中文](./i18n/README.zh-CN.md) • [🇯🇵 日本語](./i18n/README.ja-JP.md) • [🇷🇺 Русский](./i18n/README.ru.md) • [🇹🇭 ไทย](./i18n/README.th.md) • [🇮🇷 فارسی](./i18n/README.fa_IR.md) • [🇮🇩 Indonesia](./i18n/README.id-ID.md) • [🇪🇸 Español](./i18n/README.es.md) • [🇫🇷 Français](./i18n/README.fr.md)
|
||||
[🇻🇳 Tiếng Việt](./i18n/README.vi.md) • [🇨🇳 中文](./i18n/README.zh-CN.md) • [🇯🇵 日本語](./i18n/README.ja-JP.md) • [🇷🇺 Русский](./i18n/README.ru.md) • [🇹🇭 ไทย](./i18n/README.th.md) • [🇮🇷 فارسی](./i18n/README.fa_IR.md) • [🇮🇩 Indonesia](./i18n/README.id-ID.md)
|
||||
|
||||
</div>
|
||||
|
||||
@@ -285,32 +285,6 @@ Default URLs:
|
||||
<b>Kilo Code</b>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/opendesign.png" width="60" alt="OpenDesign"/><br/>
|
||||
<b>OpenDesign</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/jcode.png" width="60" alt="jcode"/><br/>
|
||||
<b>jcode</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/grok-cli.png" width="60" alt="Grok Build"/><br/>
|
||||
<b>Grok Build</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/devin-cli.png" width="60" alt="Devin CLI"/><br/>
|
||||
<b>Devin CLI</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/deepseek-tui.png" width="60" alt="DeepSeek TUI"/><br/>
|
||||
<b>DeepSeek TUI</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/qwen.png" width="60" alt="Qwen Code"/><br/>
|
||||
<b>Qwen Code</b>
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
@@ -467,46 +441,6 @@ Default URLs:
|
||||
<p><i>...and 20+ more providers including Nebius, Chutes, Hyperbolic, and custom OpenAI/Anthropic compatible endpoints</i></p>
|
||||
</div>
|
||||
|
||||
### 🏠 Self-hosted Providers
|
||||
|
||||
For speech and embeddings served from **your own** machine — whisper.cpp,
|
||||
faster-whisper, Speaches, Kokoro-FastAPI, openedai-speech, llama.cpp/llama-server,
|
||||
vLLM, Infinity, text-embeddings-inference, or anything else that speaks the OpenAI
|
||||
shape.
|
||||
|
||||
| Provider | Endpoint used | Typical server |
|
||||
| --- | --- | --- |
|
||||
| **Self-hosted STT** | `/v1/audio/transcriptions` | whisper.cpp, faster-whisper |
|
||||
| **Self-hosted TTS** | `/v1/audio/speech` | Kokoro-FastAPI, openedai-speech |
|
||||
| **Self-hosted Embedding** | `/v1/embeddings` | llama-server, vLLM, Infinity |
|
||||
|
||||
Every other speech provider is a named cloud service with a fixed endpoint. These
|
||||
three read their address from **each connection**, so one provider can front
|
||||
several machines and load-balance across them like any other.
|
||||
|
||||
Set it on the connection as `providerSpecificData.baseUrl`:
|
||||
|
||||
| Provider | Give it | Result |
|
||||
| --- | --- | --- |
|
||||
| Self-hosted STT | the full URL — `http://host:8080/v1/audio/transcriptions` | used as-is |
|
||||
| Self-hosted TTS | the server root — `http://host:8880` | `+ /v1/audio/speech` |
|
||||
| Self-hosted Embedding | the **OpenAI base**, `/v1` included — `http://host:8080/v1` | `+ /embeddings` |
|
||||
|
||||
> **Mind the `/v1` on embeddings.** The adapter appends `/embeddings`, so
|
||||
> `http://host:8080` resolves to `http://host:8080/embeddings` and misses the
|
||||
> OpenAI route — llama-server answers **501**. Give it the same base URL an OpenAI
|
||||
> client would use. A full `.../v1/embeddings` is also accepted, so a value pasted
|
||||
> from a `curl` example works too.
|
||||
|
||||
The API key is not checked by most local servers, but the field must be non-empty:
|
||||
it is what gives the connection a credentials record, and `baseUrl` lives there.
|
||||
Any placeholder works.
|
||||
|
||||
Self-hosted Embedding has **no cloud fallback by design** — a connection saved
|
||||
without a `baseUrl` is reported as a configuration error rather than quietly
|
||||
falling back to `api.openai.com`, which would send your input text and API key to
|
||||
a third party under a provider named "Self-hosted".
|
||||
|
||||
---
|
||||
|
||||
## 💡 Key Features
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "9router",
|
||||
"version": "0.5.55",
|
||||
"version": "0.5.45",
|
||||
"description": "9Router CLI - Start and manage 9Router server",
|
||||
"bin": {
|
||||
"9router": "./cli.js"
|
||||
|
||||
@@ -81,96 +81,28 @@ function copyRecursive(src, dest) {
|
||||
}
|
||||
}
|
||||
|
||||
function resolveStandaloneBuild(appDir, buildDistDir) {
|
||||
const legacyStandaloneRoot = path.join(appDir, ".next", "standalone");
|
||||
const resolvedStandaloneRoot = path.join(buildDistDir, "standalone");
|
||||
let standaloneRoot = fs.existsSync(resolvedStandaloneRoot)
|
||||
? resolvedStandaloneRoot
|
||||
: legacyStandaloneRoot;
|
||||
console.log("📦 Building 9Router CLI package with Next.js...\n");
|
||||
|
||||
// Next.js 16 nests standalone output under the project name when
|
||||
// NEXT_TRACING_ROOT_MODE=workspace, e.g. standalone/9router/server.js.
|
||||
const pkgName = path.basename(appDir);
|
||||
const nestedRoot = path.join(standaloneRoot, pkgName);
|
||||
if (fs.existsSync(path.join(nestedRoot, "server.js")) && !fs.existsSync(path.join(standaloneRoot, "server.js"))) {
|
||||
console.log(`ℹ️ Detected nested standalone output: ${pkgName}/`);
|
||||
standaloneRoot = nestedRoot;
|
||||
}
|
||||
fs.mkdirSync(buildHomeDir, { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Roaming"), { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Local"), { recursive: true });
|
||||
|
||||
const standaloneApp = fs.existsSync(path.join(standaloneRoot, "server.js"))
|
||||
? standaloneRoot
|
||||
: path.join(standaloneRoot, "app");
|
||||
if (!fs.existsSync(standaloneApp)) {
|
||||
throw new Error(
|
||||
"Next.js standalone build not found under .next/standalone; " +
|
||||
"expected either .next/standalone/server.js or .next/standalone/app/",
|
||||
);
|
||||
}
|
||||
|
||||
return { standaloneApp, standaloneRoot };
|
||||
}
|
||||
|
||||
function copyStandaloneBuild(appDir, buildDistDir, cliAppDir) {
|
||||
const { standaloneApp, standaloneRoot } = resolveStandaloneBuild(appDir, buildDistDir);
|
||||
copyRecursive(standaloneApp, cliAppDir);
|
||||
|
||||
// Older nested-app layout stores traced node_modules at standalone root.
|
||||
const standaloneNodeModules = path.join(standaloneRoot, "node_modules");
|
||||
if (standaloneApp !== standaloneRoot && fs.existsSync(standaloneNodeModules)) {
|
||||
copyRecursive(standaloneNodeModules, path.join(cliAppDir, "node_modules"));
|
||||
}
|
||||
}
|
||||
|
||||
function mergeServerArtifacts(buildDistDir, cliAppDir) {
|
||||
const serverSrc = path.join(buildDistDir, "server");
|
||||
const serverDest = path.join(cliAppDir, buildDistDirName, "server");
|
||||
if (!fs.existsSync(serverSrc)) {
|
||||
throw new Error(`Complete Next.js server build not found: ${serverSrc}`);
|
||||
}
|
||||
copyRecursive(serverSrc, serverDest);
|
||||
}
|
||||
|
||||
function assertRequiredApiArtifacts(cliAppDir) {
|
||||
const requiredArtifacts = [
|
||||
"app/api/v1/chat/completions/route.js",
|
||||
"app/api/v1/messages/route.js",
|
||||
];
|
||||
const serverDir = path.join(cliAppDir, buildDistDirName, "server");
|
||||
const missingArtifacts = requiredArtifacts
|
||||
.map((artifact) => path.join(serverDir, artifact))
|
||||
.filter((artifact) => !fs.existsSync(artifact));
|
||||
|
||||
if (missingArtifacts.length > 0) {
|
||||
throw new Error(
|
||||
`Required CLI API route artifact${missingArtifacts.length === 1 ? " is" : "s are"} missing:\n` +
|
||||
missingArtifacts.join("\n"),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
function buildCliPackage() {
|
||||
console.log("📦 Building 9Router CLI package with Next.js...\n");
|
||||
|
||||
fs.mkdirSync(buildHomeDir, { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Roaming"), { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Local"), { recursive: true });
|
||||
|
||||
// Step 0: Sync version from app/cli/package.json to app/package.json
|
||||
console.log("0️⃣ Syncing version to app/package.json...");
|
||||
const cliPkg = JSON.parse(fs.readFileSync(path.join(cliDir, "package.json"), "utf8"));
|
||||
const appPkgPath = path.join(appDir, "package.json");
|
||||
const appPkg = JSON.parse(fs.readFileSync(appPkgPath, "utf8"));
|
||||
if (appPkg.version !== cliPkg.version) {
|
||||
// Step 0: Sync version from app/cli/package.json to app/package.json
|
||||
console.log("0️⃣ Syncing version to app/package.json...");
|
||||
const cliPkg = JSON.parse(fs.readFileSync(path.join(cliDir, "package.json"), "utf8"));
|
||||
const appPkgPath = path.join(appDir, "package.json");
|
||||
const appPkg = JSON.parse(fs.readFileSync(appPkgPath, "utf8"));
|
||||
if (appPkg.version !== cliPkg.version) {
|
||||
appPkg.version = cliPkg.version;
|
||||
fs.writeFileSync(appPkgPath, JSON.stringify(appPkg, null, 2) + "\n");
|
||||
console.log(`✅ Version synced: ${cliPkg.version}\n`);
|
||||
} else {
|
||||
} else {
|
||||
console.log(`✅ Version already synced: ${cliPkg.version}\n`);
|
||||
}
|
||||
}
|
||||
|
||||
// Step 1: Build app with Next.js (workspace tracing root → traced node_modules in standalone).
|
||||
console.log("1️⃣ Building Next.js app...");
|
||||
try {
|
||||
// Step 1: Build app with Next.js (workspace tracing root → traced node_modules in standalone).
|
||||
console.log("1️⃣ Building Next.js app...");
|
||||
try {
|
||||
execSync("npm run build", {
|
||||
stdio: "inherit",
|
||||
cwd: appDir,
|
||||
@@ -185,48 +117,65 @@ function buildCliPackage() {
|
||||
}
|
||||
});
|
||||
console.log("✅ Next.js build completed\n");
|
||||
} catch (error) {
|
||||
} catch (error) {
|
||||
console.error("❌ Next.js build failed");
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
// Step 2: Clean old app/cli/app if exists
|
||||
console.log("2️⃣ Cleaning old app/cli/app...");
|
||||
if (fs.existsSync(cliAppDir)) {
|
||||
// Step 2: Clean old app/cli/app if exists
|
||||
console.log("2️⃣ Cleaning old app/cli/app...");
|
||||
if (fs.existsSync(cliAppDir)) {
|
||||
fs.rmSync(cliAppDir, { recursive: true, force: true });
|
||||
}
|
||||
console.log("✅ Cleaned\n");
|
||||
}
|
||||
console.log("✅ Cleaned\n");
|
||||
|
||||
// Step 3: Copy Next.js standalone build to app/cli/app.
|
||||
// Newer Next.js standalone output writes server.js/package.json plus .next/, src/, and
|
||||
// node_modules/ directly under .next/standalone. Older builds may still use a nested app/.
|
||||
console.log("3️⃣ Copying Next.js standalone build to app/cli/app...");
|
||||
try {
|
||||
copyStandaloneBuild(appDir, buildDistDir, cliAppDir);
|
||||
} catch (error) {
|
||||
// Step 3: Copy Next.js standalone build to app/cli/app.
|
||||
// Newer Next.js standalone output writes server.js/package.json plus .next/, src/, and
|
||||
// node_modules/ directly under .next/standalone. Older builds may still use a nested app/.
|
||||
console.log("3️⃣ Copying Next.js standalone build to app/cli/app...");
|
||||
const standaloneRoot = path.join(appDir, ".next", "standalone");
|
||||
const standaloneRootResolved = path.join(buildDistDir, "standalone");
|
||||
let standaloneRootToUse = fs.existsSync(standaloneRootResolved) ? standaloneRootResolved : standaloneRoot;
|
||||
// Next.js 16 nests standalone output under the project name when NEXT_TRACING_ROOT_MODE=workspace
|
||||
// e.g. .next-cli-build/standalone/9router/server.js
|
||||
const pkgName = path.basename(appDir);
|
||||
const nestedRoot = path.join(standaloneRootToUse, pkgName);
|
||||
if (fs.existsSync(path.join(nestedRoot, "server.js")) && !fs.existsSync(path.join(standaloneRootToUse, "server.js"))) {
|
||||
console.log(`ℹ️ Detected nested standalone output: ${pkgName}/`);
|
||||
standaloneRootToUse = nestedRoot;
|
||||
}
|
||||
const standaloneApp = fs.existsSync(path.join(standaloneRootToUse, "server.js"))
|
||||
? standaloneRootToUse
|
||||
: path.join(standaloneRootToUse, "app");
|
||||
if (!fs.existsSync(standaloneApp)) {
|
||||
console.error("❌ Next.js standalone build not found under .next/standalone");
|
||||
console.error("Expected either .next/standalone/server.js or .next/standalone/app/");
|
||||
process.exit(1);
|
||||
}
|
||||
console.log("✅ Copied standalone build\n");
|
||||
}
|
||||
copyRecursive(standaloneApp, cliAppDir);
|
||||
|
||||
// Step 3a: Copy custom server (injects real socket IP, strips spoofable XFF).
|
||||
const customServerSrc = path.join(appDir, "custom-server.js");
|
||||
if (fs.existsSync(customServerSrc)) {
|
||||
// Older nested-app layout stores traced node_modules at standalone root.
|
||||
const standaloneNodeModules = path.join(standaloneRootToUse, "node_modules");
|
||||
if (standaloneApp !== standaloneRootToUse && fs.existsSync(standaloneNodeModules)) {
|
||||
copyRecursive(standaloneNodeModules, path.join(cliAppDir, "node_modules"));
|
||||
}
|
||||
console.log("✅ Copied standalone build\n");
|
||||
|
||||
// Step 3a: Copy custom server (injects real socket IP, strips spoofable XFF).
|
||||
const customServerSrc = path.join(appDir, "custom-server.js");
|
||||
if (fs.existsSync(customServerSrc)) {
|
||||
fs.copyFileSync(customServerSrc, path.join(cliAppDir, "custom-server.js"));
|
||||
console.log("✅ Copied custom-server.js\n");
|
||||
} else {
|
||||
console.error("❌ custom-server.js not found — without it no request can be proven local,");
|
||||
console.error(" so the packaged CLI would demand an API key for its own dashboard and /v1.");
|
||||
process.exit(1);
|
||||
}
|
||||
} else {
|
||||
console.warn("⚠️ custom-server.js not found — server will run without real-IP injection\n");
|
||||
}
|
||||
|
||||
// Step 3b: Ensure sql.js (pure JS fallback) bundled in app/cli/app/node_modules.
|
||||
// Strip better-sqlite3 (native) — it lives in ~/.9router/runtime to avoid
|
||||
// Windows EBUSY during global CLI updates. node:sqlite (Node ≥22.5) is also
|
||||
// available as a no-install middle tier.
|
||||
console.log("3️⃣ b Configuring SQLite drivers...");
|
||||
function ensureModuleInBundle(pkg) {
|
||||
// Step 3b: Ensure sql.js (pure JS fallback) bundled in app/cli/app/node_modules.
|
||||
// Strip better-sqlite3 (native) — it lives in ~/.9router/runtime to avoid
|
||||
// Windows EBUSY during global CLI updates. node:sqlite (Node ≥22.5) is also
|
||||
// available as a no-install middle tier.
|
||||
console.log("3️⃣ b Configuring SQLite drivers...");
|
||||
function ensureModuleInBundle(pkg) {
|
||||
const dest = path.join(cliAppDir, "node_modules", pkg);
|
||||
if (fs.existsSync(dest)) {
|
||||
console.log(`✅ ${pkg} already bundled`);
|
||||
@@ -244,111 +193,89 @@ function buildCliPackage() {
|
||||
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
||||
copyRecursive(src, dest);
|
||||
console.log(`✅ Bundled ${pkg}`);
|
||||
}
|
||||
ensureModuleInBundle("sql.js");
|
||||
// `open` is external (see serverExternalPackages in next.config.mjs), so it must exist in
|
||||
// the bundle's node_modules or every importer throws MODULE_NOT_FOUND at runtime. Output
|
||||
// tracing normally copies it; this is the same belt-and-braces guard used for sql.js.
|
||||
ensureModuleInBundle("open");
|
||||
const betterDir = path.join(cliAppDir, "node_modules", "better-sqlite3");
|
||||
if (fs.existsSync(betterDir)) {
|
||||
}
|
||||
ensureModuleInBundle("sql.js");
|
||||
const betterDir = path.join(cliAppDir, "node_modules", "better-sqlite3");
|
||||
if (fs.existsSync(betterDir)) {
|
||||
fs.rmSync(betterDir, { recursive: true, force: true });
|
||||
console.log("✅ Stripped better-sqlite3 (lives in ~/.9router/runtime)");
|
||||
}
|
||||
console.log("");
|
||||
}
|
||||
console.log("");
|
||||
|
||||
// Step 4: Copy static files
|
||||
console.log("4️⃣ Copying static files...");
|
||||
const staticSrc = path.join(appDir, ".next", "static");
|
||||
const staticSrcResolved = path.join(buildDistDir, "static");
|
||||
const staticDest = path.join(cliAppDir, buildDistDirName, "static");
|
||||
if (fs.existsSync(staticSrcResolved) || fs.existsSync(staticSrc)) {
|
||||
// Step 4: Copy static files
|
||||
console.log("4️⃣ Copying static files...");
|
||||
const staticSrc = path.join(appDir, ".next", "static");
|
||||
const staticSrcResolved = path.join(buildDistDir, "static");
|
||||
const staticDest = path.join(cliAppDir, buildDistDirName, "static");
|
||||
if (fs.existsSync(staticSrcResolved) || fs.existsSync(staticSrc)) {
|
||||
copyRecursive(fs.existsSync(staticSrcResolved) ? staticSrcResolved : staticSrc, staticDest);
|
||||
console.log("✅ Copied static files\n");
|
||||
} else {
|
||||
} else {
|
||||
console.log("⏭️ No static files found\n");
|
||||
}
|
||||
}
|
||||
|
||||
// Step 5: Copy public folder if exists
|
||||
console.log("5️⃣ Copying public folder...");
|
||||
const publicSrc = path.join(appDir, "public");
|
||||
const publicDest = path.join(cliAppDir, "public");
|
||||
if (fs.existsSync(publicSrc)) {
|
||||
// Step 5: Copy public folder if exists
|
||||
console.log("5️⃣ Copying public folder...");
|
||||
const publicSrc = path.join(appDir, "public");
|
||||
const publicDest = path.join(cliAppDir, "public");
|
||||
if (fs.existsSync(publicSrc)) {
|
||||
copyRecursive(publicSrc, publicDest);
|
||||
console.log("✅ Copied public folder\n");
|
||||
} else {
|
||||
} else {
|
||||
console.log("⏭️ No public folder found\n");
|
||||
}
|
||||
}
|
||||
|
||||
// Step 6: Copy vendor-chunks (required for production)
|
||||
console.log("6️⃣ Copying vendor-chunks...");
|
||||
const vendorChunksSrc = path.join(appDir, ".next", "server", "vendor-chunks");
|
||||
const vendorChunksSrcResolved = path.join(buildDistDir, "server", "vendor-chunks");
|
||||
const vendorChunksDest = path.join(cliAppDir, buildDistDirName, "server", "vendor-chunks");
|
||||
if (fs.existsSync(vendorChunksSrcResolved) || fs.existsSync(vendorChunksSrc)) {
|
||||
// Step 6: Copy vendor-chunks (required for production)
|
||||
console.log("6️⃣ Copying vendor-chunks...");
|
||||
const vendorChunksSrc = path.join(appDir, ".next", "server", "vendor-chunks");
|
||||
const vendorChunksSrcResolved = path.join(buildDistDir, "server", "vendor-chunks");
|
||||
const vendorChunksDest = path.join(cliAppDir, buildDistDirName, "server", "vendor-chunks");
|
||||
if (fs.existsSync(vendorChunksSrcResolved) || fs.existsSync(vendorChunksSrc)) {
|
||||
copyRecursive(fs.existsSync(vendorChunksSrcResolved) ? vendorChunksSrcResolved : vendorChunksSrc, vendorChunksDest);
|
||||
console.log("✅ Copied vendor-chunks\n");
|
||||
} else {
|
||||
} else {
|
||||
console.log("⏭️ No vendor-chunks found\n");
|
||||
}
|
||||
}
|
||||
|
||||
// Step 6b: Merge the complete generated server tree. Next.js standalone output
|
||||
// is trace-pruned and can omit route modules or chunks loaded dynamically.
|
||||
console.log("6️⃣ b Copying complete server artifacts...");
|
||||
mergeServerArtifacts(buildDistDir, cliAppDir);
|
||||
assertRequiredApiArtifacts(cliAppDir);
|
||||
console.log("✅ Copied complete server artifacts\n");
|
||||
|
||||
// Step 7: Copy MITM server files (not bundled by Next.js standalone)
|
||||
console.log("7️⃣ Copying MITM server files...");
|
||||
const mitmSrc = path.join(appDir, "src", "mitm");
|
||||
const mitmDest = path.join(cliAppDir, "src", "mitm");
|
||||
if (fs.existsSync(mitmSrc)) {
|
||||
// Step 7: Copy MITM server files (not bundled by Next.js standalone)
|
||||
console.log("7️⃣ Copying MITM server files...");
|
||||
const mitmSrc = path.join(appDir, "src", "mitm");
|
||||
const mitmDest = path.join(cliAppDir, "src", "mitm");
|
||||
if (fs.existsSync(mitmSrc)) {
|
||||
copyRecursive(mitmSrc, mitmDest);
|
||||
console.log("✅ Copied MITM files\n");
|
||||
} else {
|
||||
} else {
|
||||
console.log("⏭️ No MITM files found\n");
|
||||
}
|
||||
}
|
||||
|
||||
// Step 7b: Copy standalone updater (headless Node process for install progress)
|
||||
console.log("7️⃣ b Copying updater files...");
|
||||
const updaterSrc = path.join(appDir, "src", "lib", "updater");
|
||||
const updaterDest = path.join(cliAppDir, "src", "lib", "updater");
|
||||
if (fs.existsSync(updaterSrc)) {
|
||||
// Step 7b: Copy standalone updater (headless Node process for install progress)
|
||||
console.log("7️⃣ b Copying updater files...");
|
||||
const updaterSrc = path.join(appDir, "src", "lib", "updater");
|
||||
const updaterDest = path.join(cliAppDir, "src", "lib", "updater");
|
||||
if (fs.existsSync(updaterSrc)) {
|
||||
copyRecursive(updaterSrc, updaterDest);
|
||||
console.log("✅ Copied updater files\n");
|
||||
} else {
|
||||
} else {
|
||||
console.log("⏭️ No updater files found\n");
|
||||
}
|
||||
}
|
||||
|
||||
// Step 8: Build MITM server (config driven - see app/cli/scripts/buildMitm.js)
|
||||
console.log("8️⃣ Building MITM server...");
|
||||
try {
|
||||
// Step 8: Build MITM server (config driven - see app/cli/scripts/buildMitm.js)
|
||||
console.log("8️⃣ Building MITM server...");
|
||||
try {
|
||||
execSync("node scripts/buildMitm.js", { stdio: "inherit", cwd: cliDir });
|
||||
console.log("✅ MITM server build completed\n");
|
||||
} catch (error) {
|
||||
} catch (error) {
|
||||
console.error("❌ MITM build failed");
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
console.log("✨ CLI package build completed!");
|
||||
console.log(`📁 Output: ${cliAppDir}`);
|
||||
console.log("✨ CLI package build completed!");
|
||||
console.log(`📁 Output: ${cliAppDir}`);
|
||||
|
||||
try {
|
||||
try {
|
||||
const { execSync: exec } = require("child_process");
|
||||
const size = exec(`du -sh "${cliAppDir}"`, { encoding: "utf8" }).trim();
|
||||
console.log(`📊 Package size: ${size.split("\t")[0]}`);
|
||||
} catch (e) {
|
||||
} catch (e) {
|
||||
// Silent fail on size check
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
assertRequiredApiArtifacts,
|
||||
copyStandaloneBuild,
|
||||
mergeServerArtifacts,
|
||||
};
|
||||
|
||||
if (require.main === module) {
|
||||
buildCliPackage();
|
||||
}
|
||||
|
||||
@@ -53,9 +53,6 @@ const PROVIDER_MODELS = {
|
||||
{ id: "glm-4.7" },
|
||||
],
|
||||
ag: [
|
||||
{ id: "gemini-3.7-flash-high" },
|
||||
{ id: "gemini-3.7-flash-medium" },
|
||||
{ id: "gemini-3.7-flash-low" },
|
||||
{ id: "gemini-3.6-flash-high" },
|
||||
{ id: "gemini-3.6-flash-medium" },
|
||||
{ id: "gemini-3.6-flash-low" },
|
||||
|
||||
111
custom-server.js
111
custom-server.js
@@ -1,51 +1,7 @@
|
||||
const http = require("http");
|
||||
const path = require("path");
|
||||
const fs = require("fs");
|
||||
const crypto = require("crypto");
|
||||
const { pathToFileURL } = require("url");
|
||||
|
||||
const origCreate = http.createServer.bind(http);
|
||||
|
||||
// Per-process secret proving x-9r-real-ip was stamped below rather than sent by the client.
|
||||
// A bare `next start` / `next dev` never loads this file, so it cannot produce a matching
|
||||
// header even though the env var is inherited by child processes. Named like x-9r-cli-token
|
||||
// so the request-detail header sanitizer redacts it too.
|
||||
const PEER_TOKEN = crypto.randomBytes(24).toString("hex");
|
||||
process.env.NINEROUTER_PEER_TOKEN = PEER_TOKEN;
|
||||
|
||||
let backgroundRefreshStarted = false;
|
||||
|
||||
function startBackgroundTokenRefreshFromCustomServer() {
|
||||
if (backgroundRefreshStarted) return;
|
||||
backgroundRefreshStarted = true;
|
||||
// Prefer source path (repo / standalone that still has src). Fail-open if missing
|
||||
// — initializeApp also starts the same scheduler when the Next app boots.
|
||||
const modPath = path.join(__dirname, "src", "sse", "services", "backgroundTokenRefresh.js");
|
||||
import(pathToFileURL(modPath).href)
|
||||
.then((m) => {
|
||||
try {
|
||||
m.startBackgroundTokenRefresh();
|
||||
} catch (e) {
|
||||
console.error("[BackgroundTokenRefresh] start failed:", e && e.message ? e.message : e);
|
||||
}
|
||||
const stop = () => {
|
||||
try {
|
||||
m.stopBackgroundTokenRefresh();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
};
|
||||
process.once("SIGINT", stop);
|
||||
process.once("SIGTERM", stop);
|
||||
})
|
||||
.catch((e) => {
|
||||
// Expected in published CLI standalone (src/ not on disk). App bootstrap covers it.
|
||||
if (process.env.DEBUG_BACKGROUND_TOKEN_REFRESH) {
|
||||
console.error("[BackgroundTokenRefresh] import failed:", e && e.message ? e.message : e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Wrap Next standalone HTTP server: derive client IP from the TCP socket
|
||||
// (unspoofable) and strip client-supplied forwarding headers so downstream
|
||||
// rate-limiting keys on the real peer address instead of attacker-controlled XFF.
|
||||
@@ -66,74 +22,11 @@ http.createServer = (...args) => {
|
||||
delete req.headers["x-9r-real-ip"];
|
||||
delete req.headers["x-forwarded-for"];
|
||||
delete req.headers["x-9r-via-proxy"];
|
||||
delete req.headers["x-9r-peer-token"];
|
||||
req.headers["x-9r-real-ip"] = ip;
|
||||
req.headers["x-9r-peer-token"] = PEER_TOKEN;
|
||||
if (viaProxy) req.headers["x-9r-via-proxy"] = "1";
|
||||
return handler(req, res);
|
||||
};
|
||||
const server = origCreate(...rest, wrapped);
|
||||
server.once("listening", () => {
|
||||
startBackgroundTokenRefreshFromCustomServer();
|
||||
});
|
||||
const origEmit = server.emit;
|
||||
// JBR 25 sends h2c upgrades that the HTTP/1.1 server would otherwise close.
|
||||
server.emit = function (event, ...eventArgs) {
|
||||
const [req, socket, head] = eventArgs;
|
||||
if (event !== "upgrade" || String(req.headers.upgrade || "").toLowerCase() !== "h2c") {
|
||||
return origEmit.call(this, event, ...eventArgs);
|
||||
}
|
||||
|
||||
const contentLength = Number(req.headers["content-length"] || 0);
|
||||
if (!Number.isSafeInteger(contentLength) || contentLength < 0) {
|
||||
socket.destroy();
|
||||
return true;
|
||||
}
|
||||
const chunks = [head];
|
||||
let received = head.length;
|
||||
const serve = () => {
|
||||
// Replay the upgraded request through the existing HTTP/1.1 handler.
|
||||
const replay = new http.IncomingMessage(socket);
|
||||
Object.assign(replay, { method: req.method, url: req.url, headers: req.headers, complete: true });
|
||||
if (received) replay.push(Buffer.concat(chunks, received).subarray(0, contentLength));
|
||||
replay.push(null);
|
||||
const res = new http.ServerResponse(replay);
|
||||
res.shouldKeepAlive = false;
|
||||
res.assignSocket(socket);
|
||||
res.once("finish", () => socket.end());
|
||||
Promise.resolve().then(() => wrapped(replay, res)).catch((error) => {
|
||||
console.error("Failed to downgrade h2c request", error);
|
||||
socket.destroy();
|
||||
});
|
||||
};
|
||||
if (received >= contentLength) serve();
|
||||
else {
|
||||
socket.on("data", function readBody(chunk) {
|
||||
chunks.push(chunk);
|
||||
received += chunk.length;
|
||||
if (received < contentLength) return;
|
||||
socket.off("data", readBody);
|
||||
serve();
|
||||
});
|
||||
socket.resume();
|
||||
}
|
||||
delete req.headers.upgrade;
|
||||
delete req.headers["http2-settings"];
|
||||
req.headers.connection = "close";
|
||||
return true;
|
||||
};
|
||||
return server;
|
||||
return origCreate(...rest, wrapped);
|
||||
};
|
||||
|
||||
if (require.main === module) {
|
||||
const standalone = path.join(__dirname, "server.js");
|
||||
if (fs.existsSync(standalone)) {
|
||||
require(standalone);
|
||||
} else {
|
||||
// Repo checkout has no standalone build next to us. `next start` builds its HTTP
|
||||
// server in-process, so the wrapper above still sanitizes every request.
|
||||
const nextBin = require.resolve("next/dist/bin/next");
|
||||
process.argv = [process.argv[0], nextBin, "start", ...process.argv.slice(2)];
|
||||
require(nextBin);
|
||||
}
|
||||
}
|
||||
require("./server.js");
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 103 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 15 KiB |
@@ -1,328 +0,0 @@
|
||||
# GPT-5.6 Codex Reasoning Overrides Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Preserve Codex-advertised Max and Ultra overrides for GPT-5.6 Sol and Terra, preserve Max for Luna, and convert Luna Ultra to Max without changing Kiro or generic OpenAI-format behavior.
|
||||
|
||||
**Architecture:** Keep the supported reasoning matrix in the existing `getThinkingLevels(provider, model)` resolver and reuse that result in both translation and Codex executor normalization. The dashboard already consumes this resolver, so no UI component change is required. Unsupported top-end levels remain safely normalized, with Luna Ultra selecting Luna's supported Max level.
|
||||
|
||||
**Tech Stack:** JavaScript ES modules, Next.js, Vitest, Codex Responses transport.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Apply the new overrides only to the OpenAI Codex provider (`codex`, exposed as `cx/`).
|
||||
- Sol and Terra support `max` and `ultra`; Luna supports `max` but not `ultra`.
|
||||
- Convert Luna `ultra` requests to `max` in both translated and native passthrough request paths.
|
||||
- Preserve existing Kiro and generic OpenAI-compatible normalization.
|
||||
- Do not add runtime model-catalog fetching, dependencies, pricing changes, or unrelated refactors.
|
||||
- Write each behavior test first and observe the expected failure before changing production code.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Provider-scoped GPT-5.6 level matrix
|
||||
|
||||
**Files:**
|
||||
- Modify: `tests/unit/thinking-levels-gpt56-sol.test.js`
|
||||
- Modify: `open-sse/providers/thinkingLevels.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getThinkingLevels(provider, model)` and existing capability metadata.
|
||||
- Produces: `getThinkingLevels(provider, model): string[] | null` with Codex-only GPT-5.6 level overrides.
|
||||
|
||||
- [ ] **Step 1: Replace the Sol-only assertions with the complete behavior matrix**
|
||||
|
||||
Use literal expected arrays so each model/provider contract is independently checked:
|
||||
|
||||
```js
|
||||
it.each([
|
||||
["gpt-5.6-sol", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-terra", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-luna", ["none", "minimal", "low", "medium", "high", "xhigh", "max"]],
|
||||
["gpt-5.6-sol-review", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-terra-review", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-luna-review", ["none", "minimal", "low", "medium", "high", "xhigh", "max"]],
|
||||
])("returns Codex levels for %s", (model, expected) => {
|
||||
expect(getThinkingLevels("codex", model)).toEqual(expected);
|
||||
});
|
||||
|
||||
it("does not expose Codex-only GPT-5.6 overrides on Kiro", () => {
|
||||
expect(getThinkingLevels("kiro", "gpt-5.6-sol")).toEqual([
|
||||
"none", "minimal", "low", "medium", "high", "xhigh",
|
||||
]);
|
||||
});
|
||||
```
|
||||
|
||||
Keep the older Codex-model assertion to protect the existing `gpt-5.3-codex` behavior.
|
||||
|
||||
- [ ] **Step 2: Run the level test and verify it fails for the missing matrix/provider scoping**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/thinking-levels-gpt56-sol.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because Sol lacks Ultra, Terra/Luna lack Max, and Kiro currently inherits Sol Max.
|
||||
|
||||
- [ ] **Step 3: Add provider-aware pattern matching and the three Codex model rules**
|
||||
|
||||
Update `PATTERN_THINKING` entries to accept an optional `provider` field and match it in `getThinkingLevels`:
|
||||
|
||||
```js
|
||||
const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
||||
|
||||
const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] },
|
||||
];
|
||||
|
||||
const hit = PATTERN_THINKING.find((entry) =>
|
||||
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
|
||||
);
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Re-run the level test and verify it passes**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/thinking-levels-gpt56-sol.test.js
|
||||
```
|
||||
|
||||
Expected: 1 test file passed with no failures.
|
||||
|
||||
- [ ] **Step 5: Commit the capability matrix**
|
||||
|
||||
```bash
|
||||
git add open-sse/providers/thinkingLevels.js tests/unit/thinking-levels-gpt56-sol.test.js
|
||||
git commit -m "feat(codex): expose GPT-5.6 reasoning overrides"
|
||||
```
|
||||
|
||||
### Task 2: Model-aware shared thinking translation
|
||||
|
||||
**Files:**
|
||||
- Modify: `tests/translator/thinking-unified.test.js`
|
||||
- Modify: `open-sse/translator/concerns/thinkingUnified.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getThinkingLevels(provider, cleanModel): string[] | null` from Task 1.
|
||||
- Produces: `parseSuffix(model)` support for `ultra` and `applyThinking(...)` output that preserves supported Codex levels.
|
||||
|
||||
- [ ] **Step 1: Add failing suffix and translation tests**
|
||||
|
||||
Add a literal parser assertion:
|
||||
|
||||
```js
|
||||
expect(parseSuffix("gpt-5.6-sol(ultra)")).toEqual({
|
||||
cleanModel: "gpt-5.6-sol",
|
||||
override: { mode: "level", level: "ultra" },
|
||||
});
|
||||
```
|
||||
|
||||
Add table-driven Codex assertions using direct request fields:
|
||||
|
||||
```js
|
||||
it.each([
|
||||
["gpt-5.6-sol", "max", "max"],
|
||||
["gpt-5.6-sol", "ultra", "ultra"],
|
||||
["gpt-5.6-terra", "max", "max"],
|
||||
["gpt-5.6-terra", "ultra", "ultra"],
|
||||
["gpt-5.6-luna", "max", "max"],
|
||||
["gpt-5.6-luna", "ultra", "max"],
|
||||
])("normalizes Codex %s effort %s to %s", (model, effort, expected) => {
|
||||
const out = apply("openai-responses", model, { reasoning: { effort } }, "codex");
|
||||
expect(out.reasoning_effort).toBe(expected);
|
||||
});
|
||||
```
|
||||
|
||||
Add a parenthesized override assertion and Kiro isolation assertion:
|
||||
|
||||
```js
|
||||
expect(apply("openai-responses", "gpt-5.6-sol(ultra)", {}, "codex").reasoning_effort).toBe("ultra");
|
||||
expect(apply("openai", "gpt-5.6-sol", { reasoning_effort: "max" }, "kiro").reasoning_effort).toBe("xhigh");
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run the translator test and verify it fails for Ultra parsing and preserved Max/Ultra**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/translator/thinking-unified.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because Ultra suffixes are ignored and OpenAI translation clamps Max to XHigh.
|
||||
|
||||
- [ ] **Step 3: Implement supported-level normalization in the shared translator**
|
||||
|
||||
Import `getThinkingLevels`. Recognize `ultra` explicitly in `parseSuffix` without adding it to the budget map. Resolve supported levels once in `applyThinking` and pass them to `applyFormat`.
|
||||
|
||||
Use this normalization rule for the OpenAI format:
|
||||
|
||||
```js
|
||||
function normalizeOpenAILevel(level, supportedLevels) {
|
||||
if (level !== "max" && level !== "ultra") return level;
|
||||
if (supportedLevels?.includes(level)) return level;
|
||||
if (level === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
return "xhigh";
|
||||
}
|
||||
```
|
||||
|
||||
Keep `none`, automatic effort, budget conversion, and every non-OpenAI format unchanged.
|
||||
|
||||
- [ ] **Step 4: Re-run the translator and generic OpenAI clamp tests**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/translator/thinking-unified.test.js tests/unit/thinking-effort-openai-max-clamp.test.js
|
||||
```
|
||||
|
||||
Expected: 2 test files passed; generic OpenAI Max still becomes XHigh.
|
||||
|
||||
- [ ] **Step 5: Commit shared translation support**
|
||||
|
||||
```bash
|
||||
git add open-sse/translator/concerns/thinkingUnified.js tests/translator/thinking-unified.test.js
|
||||
git commit -m "feat(codex): preserve supported reasoning efforts"
|
||||
```
|
||||
|
||||
### Task 3: Codex native passthrough normalization
|
||||
|
||||
**Files:**
|
||||
- Modify: `tests/unit/codex-fast-capacity.test.js`
|
||||
- Modify: `open-sse/executors/codex.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getThinkingLevels("codex", upstreamModel): string[] | null` from Task 1.
|
||||
- Produces: `CodexExecutor.transformRequest(...)` payloads with model-supported upstream `reasoning.effort` values.
|
||||
|
||||
- [ ] **Step 1: Add failing Codex executor behavior tests**
|
||||
|
||||
Add a separate `describe("Codex reasoning normalization", ...)` block with real `transformRequest` calls:
|
||||
|
||||
```js
|
||||
it.each([
|
||||
["gpt-5.6-sol", "max", "max"],
|
||||
["gpt-5.6-sol", "ultra", "ultra"],
|
||||
["gpt-5.6-terra", "max", "max"],
|
||||
["gpt-5.6-terra", "ultra", "ultra"],
|
||||
["gpt-5.6-luna", "max", "max"],
|
||||
["gpt-5.6-luna", "ultra", "max"],
|
||||
])("normalizes %s effort %s to %s", (model, effort, expected) => {
|
||||
const body = new CodexExecutor().transformRequest(model, {
|
||||
model,
|
||||
input: "hi",
|
||||
reasoning: { effort },
|
||||
}, true, {});
|
||||
expect(body.reasoning.effort).toBe(expected);
|
||||
});
|
||||
|
||||
it("resolves review models before applying the reasoning matrix", () => {
|
||||
const body = new CodexExecutor().transformRequest("gpt-5.6-terra-review", {
|
||||
model: "gpt-5.6-terra-review",
|
||||
input: "hi",
|
||||
reasoning_effort: "ultra",
|
||||
}, true, {});
|
||||
expect(body.model).toBe("gpt-5.6-terra");
|
||||
expect(body.reasoning.effort).toBe("ultra");
|
||||
});
|
||||
```
|
||||
|
||||
Keep the existing GPT-5.5 Max-to-XHigh fast-tier test.
|
||||
|
||||
- [ ] **Step 2: Run the executor test and verify supported values fail by being clamped**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/codex-fast-capacity.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because current normalization maps supported Max to XHigh and does not map Luna Ultra to Max.
|
||||
|
||||
- [ ] **Step 3: Make Codex normalization model-aware**
|
||||
|
||||
Import `getThinkingLevels` and replace the global Max clamp with:
|
||||
|
||||
```js
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
}
|
||||
```
|
||||
|
||||
Call it only after `body.model` has resolved review aliases to their upstream base model. Pass `body.model` for both `reasoning_effort` and existing `reasoning.effort` request shapes.
|
||||
|
||||
- [ ] **Step 4: Re-run the executor and focused feature suites**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/codex-fast-capacity.test.js tests/unit/thinking-levels-gpt56-sol.test.js tests/translator/thinking-unified.test.js tests/unit/thinking-effort-openai-max-clamp.test.js
|
||||
```
|
||||
|
||||
Expected: 4 test files passed with no failures.
|
||||
|
||||
- [ ] **Step 5: Commit native Codex normalization**
|
||||
|
||||
```bash
|
||||
git add open-sse/executors/codex.js tests/unit/codex-fast-capacity.test.js
|
||||
git commit -m "feat(codex): forward GPT-5.6 max and ultra efforts"
|
||||
```
|
||||
|
||||
### Task 4: Full verification and pull request
|
||||
|
||||
**Files:**
|
||||
- Verify all changed production, test, design, and plan files.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: completed Tasks 1-3.
|
||||
- Produces: verified branch pushed to `origin` and a pull request targeting `decolua/9router:master`.
|
||||
|
||||
- [ ] **Step 1: Run all focused regression tests**
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/thinking-levels-gpt56-sol.test.js tests/translator/thinking-unified.test.js tests/unit/thinking-effort-openai-max-clamp.test.js tests/unit/codex-fast-capacity.test.js
|
||||
```
|
||||
|
||||
Expected: all selected test files and tests pass.
|
||||
|
||||
- [ ] **Step 2: Run the complete unit test suite**
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit tests/translator
|
||||
```
|
||||
|
||||
Expected: all test files pass with zero failed tests.
|
||||
|
||||
- [ ] **Step 3: Run the production build**
|
||||
|
||||
```bash
|
||||
npm run build
|
||||
```
|
||||
|
||||
Expected: Next.js production build exits with status 0.
|
||||
|
||||
- [ ] **Step 4: Verify repository hygiene and requirement coverage**
|
||||
|
||||
```bash
|
||||
git diff --check upstream/master...HEAD
|
||||
git status --short --branch
|
||||
git log --oneline upstream/master..HEAD
|
||||
```
|
||||
|
||||
Expected: no whitespace errors, no uncommitted source changes, and only scoped feature commits.
|
||||
|
||||
- [ ] **Step 5: Push the feature branch and open the pull request**
|
||||
|
||||
```bash
|
||||
git push -u origin codex/gpt-5-6-reasoning-overrides
|
||||
gh pr create --repo decolua/9router --base master --head seakleangnhak:codex/gpt-5-6-reasoning-overrides --title "feat(codex): support GPT-5.6 Max and Ultra overrides" --body $'## Summary\n- expose Max and Ultra for Codex GPT-5.6 Sol and Terra\n- expose Max for Codex GPT-5.6 Luna and normalize Luna Ultra to Max\n- keep Kiro and generic OpenAI-compatible reasoning behavior unchanged\n\n## Verification\n- `npx vitest run tests/unit tests/translator`\n- `npm run build`'
|
||||
```
|
||||
|
||||
The pull request body must summarize the Codex-only support matrix, Luna Ultra-to-Max fallback, Kiro isolation, and fresh test/build evidence.
|
||||
@@ -1,122 +0,0 @@
|
||||
# GPT-5.6 Codex Reasoning Overrides Design
|
||||
|
||||
## Goal
|
||||
|
||||
Expose and preserve the reasoning levels currently advertised by the OpenAI
|
||||
Codex model catalog for GPT-5.6 Sol, Terra, and Luna when they are routed
|
||||
through the `codex` provider (`cx/`).
|
||||
|
||||
The supported override matrix is:
|
||||
|
||||
| Model family | Max | Ultra |
|
||||
| --- | --- | --- |
|
||||
| GPT-5.6 Sol | Yes | Yes |
|
||||
| GPT-5.6 Terra | Yes | Yes |
|
||||
| GPT-5.6 Luna | Yes | No |
|
||||
|
||||
The same matrix applies to 9router's virtual `-review` variants because they
|
||||
resolve to the corresponding upstream base model.
|
||||
|
||||
## Scope
|
||||
|
||||
This change is limited to OpenAI Codex (`cx/`) routes. Kiro (`kr/`) and other
|
||||
OpenAI-format providers retain their existing reasoning-level behavior even
|
||||
when they expose models with the same GPT-5.6 names.
|
||||
|
||||
The change covers the complete local request path:
|
||||
|
||||
1. The provider page advertises only the levels supported by each Codex model.
|
||||
2. A copied model suffix such as `gpt-5.6-sol(ultra)` is parsed as a reasoning
|
||||
override.
|
||||
3. The shared thinking translator preserves a supported Codex override while
|
||||
retaining the existing `xhigh` fallback for unsupported OpenAI levels.
|
||||
4. The Codex executor sends supported `max` and `ultra` values unchanged to the
|
||||
upstream Codex Responses endpoint.
|
||||
|
||||
## Current Behavior
|
||||
|
||||
`gpt-5.6-luna` and the other GPT-5.6 models already exist in the Codex model
|
||||
registry. The capability picker has a global Sol-only `max` pattern, which also
|
||||
affects providers such as Kiro unintentionally. The shared OpenAI translator
|
||||
and Codex executor then convert `max` to `xhigh`, so the advertised override is
|
||||
not preserved end to end. `ultra` is not recognized as a model suffix.
|
||||
|
||||
## Design
|
||||
|
||||
### Provider-scoped level resolution
|
||||
|
||||
Extend the existing model-pattern overrides in
|
||||
`open-sse/providers/thinkingLevels.js` with an optional provider constraint.
|
||||
Add three Codex-only GPT-5.6 patterns in most-specific order:
|
||||
|
||||
- Sol: existing levels plus `max` and `ultra`.
|
||||
- Terra: existing levels plus `max` and `ultra`.
|
||||
- Luna: existing levels plus `max`.
|
||||
|
||||
Matching remains wildcard-based so virtual `-review` variants inherit the
|
||||
base model's levels. Provider matching prevents these overrides from changing
|
||||
Kiro or other providers.
|
||||
|
||||
### Shared translation
|
||||
|
||||
Teach the suffix parser to recognize `ultra` as a discrete level without
|
||||
assigning it a synthetic token budget. When applying the OpenAI wire format,
|
||||
reuse the resolved per-provider model levels:
|
||||
|
||||
- Preserve `max` or `ultra` when the target provider/model explicitly supports
|
||||
the requested level.
|
||||
- Convert `ultra` to `max` for GPT-5.6 Luna, preserving the highest level Luna
|
||||
supports.
|
||||
- Convert other unsupported `max` or `ultra` requests to `xhigh`, preserving
|
||||
the existing safe fallback for generic OpenAI-compatible providers.
|
||||
- Leave all existing lower levels and `none` handling unchanged.
|
||||
|
||||
This keeps one capability source for the dashboard and translation behavior
|
||||
instead of duplicating the GPT-5.6 matrix.
|
||||
|
||||
### Codex executor
|
||||
|
||||
Make Codex reasoning normalization model-aware. After virtual review models
|
||||
are resolved to their upstream base model, preserve a requested level when
|
||||
the Codex capability resolver lists it. Continue converting unsupported
|
||||
`max` or `ultra` values to `xhigh`, except that Luna converts `ultra` to its
|
||||
supported `max` level.
|
||||
|
||||
Do not add `max` to the executor's legacy hyphen-suffix parser because
|
||||
`gpt-5.1-codex-max` is an actual model identifier. Dashboard overrides use the
|
||||
existing parenthesized suffix and the shared translator removes that suffix
|
||||
before executor dispatch.
|
||||
|
||||
## Error and Compatibility Behavior
|
||||
|
||||
- `cx/gpt-5.6-luna(ultra)` becomes `max` rather than sending an unsupported
|
||||
level upstream.
|
||||
- Non-GPT-5.6 Codex models retain their current supported levels and fallback
|
||||
behavior.
|
||||
- Kiro GPT-5.6 routes no longer inherit the Codex Sol-only picker override and
|
||||
continue using Kiro's existing effort normalization.
|
||||
- Direct request fields and parenthesized model overrides follow the same
|
||||
model-aware rules.
|
||||
|
||||
## Testing
|
||||
|
||||
Use test-driven development with focused unit coverage:
|
||||
|
||||
1. Level resolver tests for Sol, Terra, Luna, their review variants, an older
|
||||
Codex model, and Kiro isolation.
|
||||
2. Shared translator tests proving `max` and `ultra` survive only for supported
|
||||
Codex model/provider combinations, Luna `ultra` becomes `max`, and other
|
||||
unsupported combinations become `xhigh`.
|
||||
3. Codex executor tests proving native and translated request shapes preserve
|
||||
supported values after upstream model resolution.
|
||||
4. Existing thinking translation and Codex executor suites to guard generic
|
||||
OpenAI clamping and fast-tier behavior.
|
||||
5. Project lint/build checks in proportion to the changed JavaScript modules.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Runtime fetching or caching of the Codex model catalog.
|
||||
- Adding these levels to Kiro or another provider.
|
||||
- Changing model pricing, quotas, defaults, or service tiers.
|
||||
- Adding Codex Ultra's multi-agent orchestration behavior inside 9router;
|
||||
9router only forwards the catalog-advertised reasoning override.
|
||||
1445
i18n/README.es.md
1445
i18n/README.es.md
File diff suppressed because it is too large
Load Diff
1445
i18n/README.fr.md
1445
i18n/README.fr.md
File diff suppressed because it is too large
Load Diff
1526
i18n/README.pt-BR.md
1526
i18n/README.pt-BR.md
File diff suppressed because it is too large
Load Diff
@@ -1,15 +1,21 @@
|
||||
Dưới đây là bản dịch tiếng Việt của tài liệu Markdown, giữ nguyên toàn bộ cú pháp và cấu trúc kỹ thuật.
|
||||
|
||||
<div align="center">
|
||||
<img src="../images/9router.png?1" alt="Bảng điều khiển 9Router" width="800"/>
|
||||
|
||||
# 9Router - Free AI Router & Token Saver
|
||||
# 9Router - Free AI Router
|
||||
|
||||
**Không bao giờ ngừng code. Tiết kiệm 20-40% token với RTK + tự động dự phòng sang các mô hình AI MIỄN PHÍ & giá rẻ.**
|
||||
**Không bao giờ ngừng code. Tự động định tuyến tới các mô hình AI MIỄN PHÍ & giá rẻ với cơ chế dự phòng thông minh.**
|
||||
|
||||
**Kết nối tất cả công cụ AI Code (Claude Code, Codex, Cursor, Cline, Copilot, Antigravity...) tới 40+ Nhà cung cấp AI & 100+ Mô hình.**
|
||||
**Nhà cung cấp AI Miễn cho OpenClaw.**
|
||||
|
||||
<p align="center">
|
||||
<img src="../public/providers/openclaw.png" alt="OpenClaw" width="80"/>
|
||||
</p>
|
||||
|
||||
[](https://www.npmjs.com/package/9router)
|
||||
[](https://www.npmjs.com/package/9router)
|
||||
[](https://github.com/decolua/9router/blob/main/LICENSE)
|
||||
[](https://github.com/decolua/9router/blob/main/LICENSE)
|
||||
|
||||
[🚀 Bắt đầu nhanh](#-quick-start) • [💡 Tính năng](#-key-features) • [📖 Cài đặt](#-setup-guide) • [🌐 Website](https://9router.com)
|
||||
</div>
|
||||
@@ -18,21 +24,19 @@
|
||||
|
||||
## 🤔 Tại sao chọn 9Router?
|
||||
|
||||
**Ngừng lãng phí tiền bạc, token và không bao giờ lo chạm giới hạn (rate limit):**
|
||||
**Ngừng lãng phí tiền bạc và gặp phải giới hạn:**
|
||||
|
||||
- ❌ Hạn mức gói đăng ký hết hạn mỗi tháng mà không dùng hết
|
||||
- ❌ Giới hạn tốc độ (rate limit) làm gián đoạn công việc mid-coding
|
||||
- ❌ Kết quả của công cụ (git diff, grep, ls...) ngốn rất nhiều token
|
||||
- ❌ Chi phí API đắt đỏ ($20-50/tháng cho từng nhà cung cấp)
|
||||
- ❌ Phải chuyển đổi thủ công giữa các nhà cung cấp AI
|
||||
- ❌ Giới hạn tốc độ (rate limit) ngăn bạn giữaừng khi code
|
||||
- ❌ Các API đắt đỏ ($20-50/tháng cho mỗi nhà cung cấp)
|
||||
- ❌ Phải chuyển đổi thủ công giữa các nhà cung cấp
|
||||
|
||||
**9Router giải quyết vấn đề này:**
|
||||
|
||||
- ✅ **RTK Token Saver** - Tự động nén nội dung `tool_result`, tiết kiệm 20-40% token trên mỗi request
|
||||
- ✅ **Tối đa hóa gói đăng ký** - Theo dõi hạn mức, tận dụng triệt để trước khi reset
|
||||
- ✅ **Tự động dự phòng (Auto Fallback)** - Gói đăng ký → Giá rẻ → Miễn phí, không lo downtime
|
||||
- ✅ **Đa tài khoản (Multi-account)** - Xoay vòng (round-robin) các tài khoản cho mỗi nhà cung cấp
|
||||
- ✅ **Phổ quát (Universal)** - Hoạt động với Claude Code, Codex, Cursor, Cline, Antigravity và mọi công cụ CLI
|
||||
- ✅ **Tối đa hóa gói đăng ký** - Theo dõi hạn mức, sử dụng từng bit trước khi reset
|
||||
- ✅ **Tự động dự phòng** - Gói đăng ký → Giá rẻ → Miễn phí, thời gian chết bằng không
|
||||
- ✅ **Đa tài khoản** - Vòng tròn (round-robin) các tài khoản của mỗi nhà cung cấp
|
||||
- ✅ **Phổ quát** - Hoạt động với Claude Code, Codex, Gemini CLI, Cursor, Cline, bất kỳ công cụ CLI nào
|
||||
|
||||
---
|
||||
|
||||
@@ -40,26 +44,25 @@
|
||||
|
||||
```
|
||||
┌─────────────┐
|
||||
│ Công cụ │ (Claude Code, Codex, OpenClaw, Cursor, Cline, Antigravity...)
|
||||
│ CLI AI │
|
||||
│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
|
||||
│ Tool │
|
||||
└──────┬──────┘
|
||||
│ http://localhost:20128/v1
|
||||
↓
|
||||
┌─────────────────────────────────────────────┐
|
||||
┌────────────────────────────────────────┐
|
||||
│ 9Router (Smart Router) │
|
||||
│ • RTK Token Saver (nén tool_result token) │
|
||||
│ • Dịch chuyển định dạng (OpenAI ↔ Claude) │
|
||||
│ • Quota tracking (theo dõi hạn mức) │
|
||||
│ • Tự động làm mới OAuth Token │
|
||||
└──────┬──────────────────────────────────────┘
|
||||
│ • Format translation (OpenAI ↔ Claude) │
|
||||
│ • Quota tracking │
|
||||
│ • Auto token refresh │
|
||||
└──────┬──────────────────────────────────┘
|
||||
│
|
||||
├─→ [Tier 1: GÓI ĐĂNG KÝ] Claude Code, Codex, GitHub Copilot
|
||||
│ ↓ hết hạn mức quota
|
||||
├─→ [Tier 2: GIÁ RẺ] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ ↓ chạm ngân sách
|
||||
└─→ [Tier 3: MIỄN PHÍ] Kiro AI, OpenCode Free, Vertex AI ($300 credits)
|
||||
├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
|
||||
│ ↓ quota exhausted
|
||||
├─→ [Tier 2: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ budget limit
|
||||
└─→ [Tier 3: FREE] iFlow, Qwen, Kiro (unlimited)
|
||||
|
||||
Kết quả: Không bao giờ ngừng code, chi phí tối thiểu + tiết kiệm 20-40% token qua RTK
|
||||
Result: Never stop coding, minimal cost
|
||||
```
|
||||
|
||||
---
|
||||
@@ -73,26 +76,26 @@ npm install -g 9router
|
||||
9router
|
||||
```
|
||||
|
||||
🎉 Bảng điều khiển (Dashboard) sẽ tự động mở tại `http://localhost:20128`
|
||||
🎉 Bảng điều khiển mở tại `http://localhost:20128`
|
||||
|
||||
**2. Kết nối nhà cung cấp MIỄN PHÍ (không cần đăng ký):**
|
||||
|
||||
Bảng điều khiển → Providers → Kết nối **Kiro AI** (~50 credits/tháng miễn phí: Claude 4.5 + GLM-5 + MiniMax) hoặc **OpenCode Free** (không cần auth) → Xong!
|
||||
Bảng điều khiển → Providers -> Kết nối **ude Code** hoặc **Antigravity** -> Đăng nhập OAuth -> Xong!
|
||||
|
||||
**3. Sử dụng trong công cụ CLI của bạn:**
|
||||
|
||||
```
|
||||
Cài đặt Claude Code/Codex/OpenClaw/Cursor/Cline/Antigravity:
|
||||
Cài đặt Claude Code/Codex/Gemini CLI/OpenClaw/Cursor/Cline:
|
||||
Endpoint: http://localhost:20128/v1
|
||||
API Key: [sao chép từ bảng điều khiển]
|
||||
Model: kr/claude-sonnet-4.5
|
||||
Model: if/kimi-k2-thinking
|
||||
```
|
||||
|
||||
**Thế là xong!** Bắt đầu code ngay với các mô hình AI MIỄN PHÍ.
|
||||
**Xong rồi!** Bắt đầu code với các mô hình AI MIỄN PHÍ.
|
||||
|
||||
**Phương án khác: chạy từ mã nguồn (repository này):**
|
||||
**Phương án khác: chạy từ nguồn (k lưu trữ này):**
|
||||
|
||||
Gói kho lưu trữ này là riêng tư (`9router-app`), vì vậy việc chạy từ nguồn/Docker là cách phát triển cục bộ mặc định.
|
||||
Gói kho lưu trữ này là riêng tư (`9router-app`), vì vậy việc thực thi nguồn/Docker là đường dẫn phát triển cục bộ dự kiến.
|
||||
|
||||
```bash
|
||||
cp .env.example .env
|
||||
@@ -108,12 +111,11 @@ PORT=20128 HOSTNAME=0.0.0.0 NEXT_PUBLIC_BASE_URL=http://localhost:20128 npm run
|
||||
```
|
||||
|
||||
URL mặc định:
|
||||
- Bảng điều khiển Dashboard: `http://localhost:20128/dashboard`
|
||||
- Bảng điều khiển: `http://localhost:20128/dashboard`
|
||||
- API tương thích OpenAI: `http://localhost:20128/v1`
|
||||
|
||||
---
|
||||
|
||||
|
||||
## 🎥 Hướng dẫn Video
|
||||
|
||||
<div align="center">
|
||||
|
||||
@@ -42,26 +42,25 @@
|
||||
|
||||
```
|
||||
┌─────────────┐
|
||||
│ Your CLI │ (Claude Code, Codex, OpenClaw, Cursor, Cline, Antigravity...)
|
||||
│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
|
||||
│ Tool │
|
||||
└──────┬──────┘
|
||||
│ http://localhost:20128/v1
|
||||
│ http://localhost:201281
|
||||
↓
|
||||
┌─────────────────────────────────────────────┐
|
||||
┌─────────────────────────────────────────┐
|
||||
│ 9Router (Smart Router) │
|
||||
│ • RTK Token Saver (节省 20-40% Token) │
|
||||
│ • 格式转换 (OpenAI ↔ Claude) │
|
||||
│ • 配额追踪 (Quota tracking) │
|
||||
│ • 自动刷新 OAuth Token │
|
||||
└──────┬──────────────────────────────────────┘
|
||||
│ • Format translation (OpenAI ↔ Claude) │
|
||||
│ • Quota tracking │
|
||||
│ • Auto token refresh │
|
||||
└──────┬──────────────────────────────────┘
|
||||
│
|
||||
├─→ [Tier 1: 订阅] Claude Code, Codex, GitHub Copilot
|
||||
│ ↓ 配额用尽
|
||||
├─→ [Tier 2: 低价] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ ↓ 触及预算上限
|
||||
└─→ [Tier 3: 免费] Kiro AI, OpenCode Free, Vertex AI ($300 credits)
|
||||
├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
|
||||
│ ↓ quota exhausted
|
||||
├─→ [Tier 2: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ ↓ budget limit
|
||||
└─→ [Tier 3: FREE] iFlow, Qwen, Kiro (unlimited)
|
||||
|
||||
结果:永不停歇的编程体验,最低成本 + 通过 RTK 节省 20-40% Token
|
||||
Result: Never stop coding, minimal cost
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
@@ -13,14 +13,7 @@ const proxyClientMaxBodySize = process.env.NINEROUTER_PROXY_CLIENT_MAX_BODY_SIZE
|
||||
const nextConfig = {
|
||||
distDir: process.env.NEXT_DIST_DIR || ".next",
|
||||
output: "standalone",
|
||||
// `open` must stay external. It derives its own directory from `import.meta.url`, and
|
||||
// webpack replaces that with the absolute path of the BUILD machine as a string literal.
|
||||
// A release built on macOS therefore ships `file:///Users/.../open/index.js`, which
|
||||
// `fileURLToPath` rejects on Windows ("File URL path must be absolute" — no drive
|
||||
// letter). That throw happens at module scope, so every consumer of `open` dies on
|
||||
// import — including xAI/Grok token refresh, which loads the OAuth service that imports
|
||||
// it. Keeping it external preserves the real `import.meta.url` at runtime.
|
||||
serverExternalPackages: ["better-sqlite3", "sql.js", "node:sqlite", "bun:sqlite", "open"],
|
||||
serverExternalPackages: ["better-sqlite3", "sql.js", "node:sqlite", "bun:sqlite"],
|
||||
turbopack: {
|
||||
root: tracingRoot
|
||||
},
|
||||
|
||||
@@ -37,3 +37,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
|
||||
- `registry/index.js` is an auto-generated static import list; regenerate it (don't hand-edit) after adding a `registry/{id}.js`. REGISTRY_TEMPLATE is excluded by design.
|
||||
- Special binary/protobuf formats (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — handle in their executor.
|
||||
- `rtk/` + `headroom.js` mutate the request body in-place and are **fail-open**: any error returns null and leaves the body untouched — never throw out of them. RTK skips `is_error`/`status:"error"` tool results to preserve traces.
|
||||
- **HTTP 200 in-stream errors**: some upstreams signal failure INSIDE a 200 stream (AI SDK v5 `{"type":"error"}` events, error text in content). HTTP-level success checks miss these → no fallback, `Status: success` in logs. Three hook points + one config escape hatch:
|
||||
1. **Translator** — never map an error event to content. Emit an OpenAI-shaped `chunk.error = { message, type }` + terminal chunk (`translator/response/commandcode-to-openai.js` is the worked example). Downstream `parseSSEToOpenAIResponse` already detects `chunk?.error`.
|
||||
2. **Executor early-peek** — for streaming fallback, read the first events BEFORE returning the response; an error → non-ok Response (`executors/commandcode.js` `peekForUpstreamError`).
|
||||
3. **Config escape hatch (no code)** — per-provider `streamErrorPatterns` setting (UI: provider page → Stream Error Patterns). Patterns matched against the first ~8KB of the stream and the assembled non-streaming content; see `utils/streamErrorPeek.js` + `utils/streamErrorPatterns.js`.
|
||||
|
||||
@@ -156,13 +156,6 @@ export const LOAD_CODE_ASSIST_HEADERS = {
|
||||
"Client-Metadata": JSON.stringify({ ideType: IDE_TYPE.ANTIGRAVITY, platform: getPlatformEnum(), pluginType: PLUGIN_TYPE.GEMINI }),
|
||||
};
|
||||
|
||||
// Real Antigravity IDE doesn't send X-Goog-Api-Client/Client-Metadata on loadCodeAssist/onboardUser —
|
||||
// Google's backend fingerprints those and silently refuses to provision a cloudaicompanionProject.
|
||||
export const ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT,
|
||||
};
|
||||
|
||||
export const LOAD_CODE_ASSIST_METADATA = {
|
||||
ideType: IDE_TYPE.ANTIGRAVITY,
|
||||
platform: getPlatformEnum(),
|
||||
@@ -183,6 +176,7 @@ export const OAUTH_ENDPOINTS = {
|
||||
google: { token: "https://oauth2.googleapis.com/token", auth: "https://accounts.google.com/o/oauth2/auth" },
|
||||
openai: { token: PROVIDER_OAUTH["codex"]?.tokenUrl, auth: PROVIDER_OAUTH["codex"]?.authorizeUrl },
|
||||
anthropic: { token: PROVIDER_OAUTH["claude"]?.tokenUrl, auth: "https://api.anthropic.com/v1/oauth/authorize" }, // ≠ claude.authorizeUrl (claude.ai login) — keep
|
||||
qwen: { token: PROVIDER_OAUTH["qwen"]?.tokenUrl, auth: PROVIDER_OAUTH["qwen"]?.deviceCodeUrl },
|
||||
iflow: { token: PROVIDER_OAUTH["iflow"]?.tokenUrl, auth: PROVIDER_OAUTH["iflow"]?.authorizeUrl },
|
||||
github: { token: PROVIDER_OAUTH["github"]?.tokenUrl, auth: PROVIDER_OAUTH["github"]?.authorizeUrl, deviceCode: PROVIDER_OAUTH["github"]?.deviceCodeUrl },
|
||||
};
|
||||
|
||||
@@ -2,7 +2,7 @@ import { PROVIDERS } from "./providers.js";
|
||||
import REGISTRY from "../providers/registry/index.js";
|
||||
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
|
||||
import { PROVIDER_MODELS } from "../providers/index.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js";
|
||||
export { PROVIDER_MODELS };
|
||||
|
||||
@@ -54,14 +54,6 @@ export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
return modelTargetFormat(findModel(models, modelId, aliasOrId));
|
||||
}
|
||||
|
||||
// Declared upstream formats for a model (registry `supportedFormats`). Drives the
|
||||
// per-model guard on the sourceFormat-matched transport; null when undeclared.
|
||||
export function getModelSupportedFormats(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelSupportedFormats(findModel(models, modelId, aliasOrId));
|
||||
}
|
||||
|
||||
export function getModelType(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
|
||||
@@ -33,21 +33,6 @@ const GEMINI_VOICES = [
|
||||
"Vindemiatrix", "Sadachbia", "Sadaltager", "Sulafat",
|
||||
].map((id) => ({ id, name: id, type: "tts" }));
|
||||
|
||||
// Xiaomi MiMo preset voices (from https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5).
|
||||
// Voice id is passed via `audio.voice`; `mimo_default` = default (冰糖 on CN cluster, Mia elsewhere).
|
||||
// Voices are language-independent — the spoken language is a separate hint, not bound to the voice.
|
||||
const MIMO_VOICES = [
|
||||
{ id: "mimo_default", name: "mimo_default" },
|
||||
{ id: "冰糖", name: "冰糖" },
|
||||
{ id: "茉莉", name: "茉莉" },
|
||||
{ id: "苏打", name: "苏打" },
|
||||
{ id: "白桦", name: "白桦" },
|
||||
{ id: "Mia", name: "Mia" },
|
||||
{ id: "Chloe", name: "Chloe" },
|
||||
{ id: "Milo", name: "Milo" },
|
||||
{ id: "Dean", name: "Dean" },
|
||||
].map((v) => ({ type: "tts", ...v }));
|
||||
|
||||
// ── TTS Config (config-driven, single source of truth) ─────────────────────
|
||||
export const TTS_MODELS_CONFIG = {
|
||||
openai: {
|
||||
@@ -122,14 +107,6 @@ export const TTS_MODELS_CONFIG = {
|
||||
},
|
||||
allVoices: GEMINI_VOICES,
|
||||
},
|
||||
"xiaomi-mimo": {
|
||||
models: [
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", type: "tts" },
|
||||
],
|
||||
voices: {
|
||||
"mimo-v2.5-tts": MIMO_VOICES,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
// ── Helper: get voices for a specific model ────────────────────────────────
|
||||
|
||||
@@ -245,18 +245,6 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
// Strip tools/toolConfig (handled separately) and blacklisted fields that Google rejects
|
||||
const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {};
|
||||
stripBlacklisted(requestWithoutTools);
|
||||
|
||||
// Rewrite competitive system prompts (e.g. Zed IDE's Claude prompt) to prevent Antigravity from
|
||||
// flagging the request and immediately blocking it with a 429 Quota Exhausted response.
|
||||
if (requestWithoutTools.systemInstruction?.parts) {
|
||||
const oldText = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
|
||||
for (const part of requestWithoutTools.systemInstruction.parts) {
|
||||
if (typeof part.text === "string" && part.text.includes(oldText)) {
|
||||
part.text = part.text.split(oldText).join("");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const generationConfig = { ...(requestWithoutTools.generationConfig || {}) };
|
||||
if (generationConfig.maxOutputTokens > MAX_ANTIGRAVITY_OUTPUT_TOKENS) {
|
||||
generationConfig.maxOutputTokens = MAX_ANTIGRAVITY_OUTPUT_TOKENS;
|
||||
|
||||
@@ -2,8 +2,8 @@ import { HTTP_STATUS, RETRY_CONFIG, DEFAULT_RETRY_CONFIG, resolveRetryEntry, FET
|
||||
import { shouldRefreshCredentials } from "../services/oauthCredentialManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
|
||||
/**
|
||||
* BaseExecutor - Base class for provider executors
|
||||
@@ -31,7 +31,7 @@ export class BaseExecutor {
|
||||
if (this.provider?.startsWith?.("openai-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || OPENAI_COMPAT_BASE;
|
||||
const normalized = baseUrl.replace(/\/$/, "");
|
||||
const path = resolveOpenAICompatibleApiType(this.provider, credentials) === "responses" ? "/responses" : "/chat/completions";
|
||||
const path = this.provider.includes("responses") ? "/responses" : "/chat/completions";
|
||||
return `${normalized}${path}`;
|
||||
}
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
@@ -127,19 +127,20 @@ export class BaseExecutor {
|
||||
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
|
||||
const url = this.buildUrl(model, stream, urlIndex, credentials);
|
||||
const transformedBody = this.transformRequest(model, body, stream, credentials);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model);
|
||||
const headers = this.buildHeaders(credentials, stream, url);
|
||||
|
||||
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
|
||||
|
||||
// Abort if upstream doesn't return response headers within connection timeout
|
||||
const connectCtrl = new AbortController();
|
||||
const timeoutMs = this.config?.timeoutMs || FETCH_CONNECT_TIMEOUT_MS;
|
||||
const timeoutMs = await resolveProviderTimeoutMs(this.provider, this.config?.timeoutMs, FETCH_CONNECT_TIMEOUT_MS);
|
||||
const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs);
|
||||
const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal;
|
||||
let fetchT0 = 0;
|
||||
|
||||
try {
|
||||
const bodyStr = JSON.stringify(transformedBody);
|
||||
const fetchT0 = Date.now();
|
||||
fetchT0 = Date.now();
|
||||
dbg("FETCH", `${this.provider.toUpperCase()} → ${url} | body=${bodyStr.length}B | connectTimeout=${timeoutMs}ms`);
|
||||
const response = await proxyAwareFetch(url, {
|
||||
method: "POST",
|
||||
@@ -165,6 +166,11 @@ export class BaseExecutor {
|
||||
clearTimeout(connectTimer);
|
||||
lastError = error;
|
||||
const isConnectTimeout = connectCtrl.signal.aborted && error.name === "AbortError";
|
||||
// Error diagnostic — only logs on actual upstream failure. Distinguishes
|
||||
// undici connect timeout (UND_ERR_CONNECT_TIMEOUT), DNS (ENOTFOUND),
|
||||
// refused (ECONNREFUSED) vs our own connectCtrl abort (AbortError).
|
||||
const cause = error?.cause || {};
|
||||
console.log(`[FETCH-DIAG] ${this.provider} fetch error | name=${error.name} | code=${error.code ?? cause?.code ?? "none"} | msg=${String(error.message).slice(0, 120)} | connectTimeout=${timeoutMs}ms | elapsed=${Date.now() - fetchT0}ms`);
|
||||
dbg("FETCH", `${this.provider.toUpperCase()} ✖ ${error.name}: ${error.message}${isConnectTimeout ? " (connect timeout)" : ""}`);
|
||||
// Connect timeout is internal — convert to retryable network error, don't propagate AbortError
|
||||
if (error.name === "AbortError" && !isConnectTimeout) throw error;
|
||||
|
||||
@@ -18,35 +18,6 @@ export class CodeBuddyExecutor extends DefaultExecutor {
|
||||
const transformed = super.transformRequest(model, body, stream, credentials);
|
||||
transformed.stream = true;
|
||||
|
||||
// Tencent's content filter flags CLI agent system prompts ("You are Claude
|
||||
// Code, Anthropic's official CLI...") as prompt injection / sensitive content
|
||||
// and rejects the whole request. Detect agent system prompts (length catch-all
|
||||
// + identity-marker regex) and replace them with a neutral one, while leaving
|
||||
// legitimate user system prompts untouched. content may be a string or typed
|
||||
// blocks ([{type:"text",text}]) depending on the incoming client format, so
|
||||
// flatten before matching and preserve the original shape on replacement.
|
||||
const NEUTRAL_PROMPT = "You are a helpful AI assistant that helps with software engineering tasks.";
|
||||
const AGENT_PATTERN = /you are claude code|claude.?code.+official.+cli|anthropic.+official.+cli|anxthxropic.+official.+cli|you are (?:cursor|windsurf|cline|aider|continue|copilot|cody)|you are an? (?:ai )?(?:coding |code )?agent|cc_entrypoint\s*=\s*(?:cli|vscode|jetbrains|gui)|claude.?code.+issues|give feedback.+claude.?code|you are .{0,30}(?:powerful )?ai agent|orchestration capabilities|OhMyOpenCode|<agent-identity>|<Role>|<Behavior_Instructions>/i;
|
||||
const flatten = (content) =>
|
||||
typeof content === "string"
|
||||
? content
|
||||
: Array.isArray(content)
|
||||
? content.map((b) => (b && typeof b.text === "string" ? b.text : "")).join("\n")
|
||||
: "";
|
||||
if (Array.isArray(transformed.messages)) {
|
||||
transformed.messages = transformed.messages.map((message) => {
|
||||
if (!message || message.role !== "system") return message;
|
||||
const text = flatten(message.content);
|
||||
if (!text) return message;
|
||||
if (text.length > 2000 || AGENT_PATTERN.test(text)) {
|
||||
return typeof message.content === "string"
|
||||
? { ...message, content: NEUTRAL_PROMPT }
|
||||
: { ...message, content: [{ type: "text", text: NEUTRAL_PROMPT }] };
|
||||
}
|
||||
return message;
|
||||
});
|
||||
}
|
||||
|
||||
// CodeBuddy only surfaces model reasoning when the request carries the CLI's
|
||||
// OpenAI-style params: reasoning_effort + reasoning_summary:"auto". 9router's
|
||||
// thinking pipeline sets reasoning_effort only when the client asks, and never
|
||||
|
||||
@@ -23,20 +23,6 @@ export class CodeBuddyIntlExecutor extends DefaultExecutor {
|
||||
} else if (eff) {
|
||||
transformed.reasoning_summary = "auto";
|
||||
}
|
||||
|
||||
// CodeBuddy rejects plain OpenAI shape (11101 invalid request): needs a
|
||||
// leading system prompt + user content as typed blocks, not a bare string.
|
||||
const source = Array.isArray(transformed.messages) ? transformed.messages : [];
|
||||
transformed.messages = [{ role: "system", content: "You are CodeBuddy Code." }];
|
||||
for (const message of source) {
|
||||
if (!message || typeof message !== "object" || ["system", "developer"].includes(message.role)) continue;
|
||||
if (message.role === "user" && typeof message.content === "string") {
|
||||
transformed.messages.push({ ...message, content: [{ type: "text", text: message.content }] });
|
||||
} else {
|
||||
transformed.messages.push({ ...message });
|
||||
}
|
||||
}
|
||||
|
||||
return transformed;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,7 +8,6 @@ import {
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
@@ -125,12 +124,8 @@ function resolveCacheSessionId(body, credentials) {
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
function normalizeReasoningEffort(value) {
|
||||
return value === "max" ? "xhigh" : value;
|
||||
}
|
||||
|
||||
function findNestedMessage(value, depth = 0) {
|
||||
@@ -445,10 +440,10 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || 'low');
|
||||
const effort = normalizeReasoningEffort(body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort);
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { randomUUID } from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { commandCodeToOpenAIResponse } from "../translator/response/commandcode-to-openai.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
|
||||
@@ -14,13 +15,19 @@ import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
* We translate each event to an OpenAI chat.completion.chunk and emit it as SSE so
|
||||
* both the streaming and non-streaming (forced SSE → JSON) downstream handlers in
|
||||
* 9router can consume it without further format translation.
|
||||
*
|
||||
* Terminal upstream failures arrive as `{"type":"error"}` events inside the HTTP
|
||||
* 200 stream, so a plain `response.ok` check cannot see them. We peek the first
|
||||
* events before committing the response (see peekForUpstreamError) so a stream
|
||||
* that starts with an error fails fast — the normal `!response.ok` path then
|
||||
* triggers account/model fallback instead of streaming fake success content.
|
||||
*/
|
||||
export class CommandCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("commandcode", PROVIDERS.commandcode);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
transformRequest(_model, body, _stream, _credentials) {
|
||||
body.stream = true;
|
||||
return body;
|
||||
}
|
||||
@@ -42,11 +49,195 @@ export class CommandCodeExecutor extends BaseExecutor {
|
||||
async execute(opts) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
result.response = wrapNdjsonAsOpenAISse(result.response, opts.model);
|
||||
result.response = await peekForUpstreamError(result.response, opts.model, {
|
||||
signal: opts.signal,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
// How long to hold the response open while peeking the first upstream events.
|
||||
// An upstream error event ("Network connection lost") is emitted at stream
|
||||
// start, so the peek is fast; the bound just prevents a slow-started stream
|
||||
// from being held hostage. Env: COMMANDCODE_PEEK_TIMEOUT_MS.
|
||||
const PEEK_TIMEOUT_MS = (() => {
|
||||
const raw = process.env.COMMANDCODE_PEEK_TIMEOUT_MS;
|
||||
const n = raw ? parseInt(raw, 10) : NaN;
|
||||
return Number.isFinite(n) && n > 0 ? n : 10 * 1000;
|
||||
})();
|
||||
|
||||
// Event types that count as "the stream has started producing". Everything
|
||||
// else (start, start-step, reasoning-start, text-start, ...) is metadata and
|
||||
// does not end the peek.
|
||||
const MEANINGFUL_EVENT_TYPES = new Set([
|
||||
"text-delta",
|
||||
"reasoning-delta",
|
||||
"tool-input-start",
|
||||
"tool-input-delta",
|
||||
"tool-input-end",
|
||||
"tool-call",
|
||||
"finish-step",
|
||||
"finish",
|
||||
]);
|
||||
|
||||
function makeAbortError(reason) {
|
||||
const error = new Error(reason?.message || reason || "Request aborted");
|
||||
error.name = "AbortError";
|
||||
return error;
|
||||
}
|
||||
|
||||
function tryParseEvent(line) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) return null;
|
||||
const json = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
if (!json || json === "[DONE]") return null;
|
||||
try {
|
||||
return JSON.parse(json);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function formatErrorValue(errVal) {
|
||||
const errStr =
|
||||
typeof errVal === "string"
|
||||
? errVal
|
||||
: typeof errVal?.message === "string"
|
||||
? errVal.message
|
||||
: JSON.stringify(errVal);
|
||||
const errType =
|
||||
typeof errVal === "string"
|
||||
? "upstream_error"
|
||||
: errVal?.type || "upstream_error";
|
||||
return { message: errStr, type: errType };
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the first upstream events before committing the response.
|
||||
*
|
||||
* - `{"type":"error"}` as the first meaningful event → return a 502 Response so
|
||||
* chatCore's `!response.ok` path parses the error and triggers fallback.
|
||||
* - Otherwise → re-emit the buffered bytes + the rest of the stream through the
|
||||
* normal NDJSON → OpenAI SSE wrapper and return it untouched in spirit.
|
||||
*
|
||||
* Bounded by `timeoutMs` (default PEEK_TIMEOUT_MS): if no meaningful event
|
||||
* arrives in time, or the request signal aborts, we commit whatever we have and
|
||||
* let the regular stream pipeline (stall detection, abort handling) take over.
|
||||
*/
|
||||
export async function peekForUpstreamError(
|
||||
originalResponse,
|
||||
model,
|
||||
{ signal = null, timeoutMs = PEEK_TIMEOUT_MS } = {},
|
||||
) {
|
||||
const reader = originalResponse.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
const abortController = new AbortController();
|
||||
const forwardAbort = () => abortController.abort(signal?.reason);
|
||||
if (signal?.aborted) abortController.abort(signal?.reason);
|
||||
else if (signal)
|
||||
signal.addEventListener("abort", forwardAbort, { once: true });
|
||||
|
||||
// Raw bytes for lossless re-emission; decoded text is only used for line
|
||||
// parsing / error detection. Never re-encode decoded text: TextDecoder
|
||||
// holds a split multi-byte char internally and flush() would replace it
|
||||
// with U+FFFD, corrupting the stream.
|
||||
const rawChunks = [];
|
||||
let peeked = "";
|
||||
let errorEvent = null;
|
||||
let committed = false;
|
||||
|
||||
const readWithTimeout = (ms) => {
|
||||
if (abortController.signal.aborted) {
|
||||
return Promise.reject(makeAbortError(abortController.signal.reason));
|
||||
}
|
||||
const timeoutPromise = new Promise((_, reject) => {
|
||||
const t = setTimeout(() => reject(new Error("peek timeout")), ms);
|
||||
t.unref?.();
|
||||
});
|
||||
const abortPromise = new Promise((_, reject) => {
|
||||
abortController.signal.addEventListener(
|
||||
"abort",
|
||||
() => reject(makeAbortError(abortController.signal.reason)),
|
||||
{ once: true },
|
||||
);
|
||||
});
|
||||
return Promise.race([reader.read(), timeoutPromise, abortPromise]);
|
||||
};
|
||||
|
||||
try {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (!errorEvent && !committed && Date.now() < deadline) {
|
||||
const { done, value } = await readWithTimeout(
|
||||
Math.max(deadline - Date.now(), 1),
|
||||
);
|
||||
if (done) break;
|
||||
rawChunks.push(value);
|
||||
peeked += decoder.decode(value, { stream: true });
|
||||
const lines = peeked.split("\n");
|
||||
// The last segment may be a partial line — only parse complete ones.
|
||||
for (const line of lines.slice(0, -1)) {
|
||||
const event = tryParseEvent(line);
|
||||
if (!event?.type) continue;
|
||||
if (event.type === "error") {
|
||||
errorEvent = event;
|
||||
break;
|
||||
}
|
||||
if (MEANINGFUL_EVENT_TYPES.has(event.type)) {
|
||||
committed = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// timeout / abort / read failure during the peek → commit whatever we have;
|
||||
// the downstream stream pipeline (stall detection, abort handling) takes over.
|
||||
}
|
||||
|
||||
if (signal) signal.removeEventListener("abort", forwardAbort);
|
||||
|
||||
if (errorEvent) {
|
||||
await reader.cancel("commandcode early error detected").catch(() => {});
|
||||
const { message, type } = formatErrorValue(
|
||||
errorEvent.error ?? errorEvent.message ?? "unknown",
|
||||
);
|
||||
return new Response(JSON.stringify({ error: { message, type } }), {
|
||||
status: HTTP_STATUS.BAD_GATEWAY,
|
||||
statusText: message.slice(0, 200),
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
const remaining = new ReadableStream({
|
||||
start(controller) {
|
||||
(async () => {
|
||||
try {
|
||||
// Re-emit RAW bytes (never re-encoded decoded text) so split
|
||||
// multi-byte UTF-8 sequences survive the peek untouched.
|
||||
for (const c of rawChunks) controller.enqueue(c);
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
controller.enqueue(value);
|
||||
}
|
||||
controller.close();
|
||||
} catch (err) {
|
||||
controller.error(err);
|
||||
}
|
||||
})();
|
||||
},
|
||||
cancel() {
|
||||
reader.cancel("commandcode stream cancelled").catch(() => {});
|
||||
},
|
||||
});
|
||||
|
||||
const combined = new Response(remaining, {
|
||||
status: originalResponse.status,
|
||||
statusText: originalResponse.statusText,
|
||||
headers: originalResponse.headers,
|
||||
});
|
||||
return wrapNdjsonAsOpenAISse(combined, model);
|
||||
}
|
||||
|
||||
function wrapNdjsonAsOpenAISse(originalResponse, model) {
|
||||
const decoder = new TextDecoder();
|
||||
const encoder = new TextEncoder();
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS, PROVIDER_OAUTH } from "../config/providers.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { OAUTH_ENDPOINTS, buildKimiHeaders } from "../config/appConstants.js";
|
||||
import { buildClineHeaders } from "../shared/clineAuth.js";
|
||||
import { getCachedClaudeHeaders } from "../utils/claudeHeaderCache.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
|
||||
@@ -42,6 +42,21 @@ const HEADER_HOOKS = {
|
||||
kimiHeaders: (h, c) => Object.assign(h, buildKimiHeaders(c?.providerSpecificData?.deviceId)),
|
||||
clineHeaders: (h, c) => Object.assign(h, buildClineHeaders(c.apiKey || c.accessToken)),
|
||||
kilocodeOrg: (h, c) => { if (c.providerSpecificData?.orgId) h["X-Kilocode-OrganizationID"] = c.providerSpecificData.orgId; },
|
||||
claudeOverlay: (h) => {
|
||||
const cached = getCachedClaudeHeaders();
|
||||
if (!cached) return;
|
||||
for (const lcKey of Object.keys(cached)) {
|
||||
const titleKey = lcKey.replace(/(^|-)([a-z])/g, (_, sep, ch) => sep + ch.toUpperCase());
|
||||
if (lcKey === "anthropic-beta") {
|
||||
const staticBetaStr = h[titleKey] || h[lcKey] || "";
|
||||
const flags = new Set(staticBetaStr.split(",").map(f => f.trim()).filter(Boolean));
|
||||
for (const f of cached[lcKey].split(",").map(f => f.trim()).filter(Boolean)) flags.add(f);
|
||||
cached[lcKey] = Array.from(flags).join(",");
|
||||
}
|
||||
if (titleKey !== lcKey && h[titleKey] !== undefined) delete h[titleKey];
|
||||
}
|
||||
Object.assign(h, cached);
|
||||
},
|
||||
};
|
||||
|
||||
// Config-driven OAuth refresh grants — derived from registry oauth.refresh.
|
||||
@@ -110,7 +125,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
if (this.provider?.startsWith?.("openai-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || OPENAI_COMPAT_BASE;
|
||||
const normalized = baseUrl.replace(/\/$/, "");
|
||||
const path = resolveOpenAICompatibleApiType(this.provider, credentials) === "responses" ? "/responses" : "/chat/completions";
|
||||
const path = this.provider.includes("responses") ? "/responses" : "/chat/completions";
|
||||
return `${normalized}${path}`;
|
||||
}
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
@@ -146,18 +161,14 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
return BEARER;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const rt = credentials?.runtimeTransport;
|
||||
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
|
||||
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
|
||||
// Hooks run BEFORE auth so dynamic overlays can't clobber the token.
|
||||
// Hooks run BEFORE auth so dynamic overlays (claude cached headers) can't clobber the token.
|
||||
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
|
||||
applyAuth(headers, desc, credentials);
|
||||
|
||||
if (this.provider === "claude" && model) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
|
||||
}
|
||||
|
||||
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || "";
|
||||
@@ -211,6 +222,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
const refreshers = {
|
||||
claude: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
codex: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
qwen: () => this.refreshWithForm(OAUTH_ENDPOINTS.qwen.token, { grant_type: "refresh_token", refresh_token: credentials.refreshToken, client_id: PROVIDERS.qwen.clientId }, proxyOptions),
|
||||
iflow: () => this.refreshIflow(credentials.refreshToken, proxyOptions),
|
||||
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
|
||||
|
||||
@@ -9,7 +9,9 @@ import { KimchiExecutor } from "./kimchi.js";
|
||||
import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
import { QwenExecutor } from "./qwen.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
@@ -39,7 +41,9 @@ const executors = {
|
||||
cu: new CursorExecutor(), // Alias for cursor
|
||||
vertex: new VertexExecutor("vertex"),
|
||||
"vertex-partner": new VertexExecutor("vertex-partner"),
|
||||
qwen: new QwenExecutor(),
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
@@ -83,7 +87,9 @@ export { CodexExecutor } from "./codex.js";
|
||||
export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
export { DefaultExecutor } from "./default.js";
|
||||
export { QwenExecutor } from "./qwen.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
|
||||
@@ -144,12 +144,6 @@ function normalizeStopReason(value) {
|
||||
return reason || null;
|
||||
}
|
||||
|
||||
// Of the reasons stopDisposition() folds into "terminal_incomplete", only these
|
||||
// mean "usable as far as it got, then the budget ran out" -- the case
|
||||
// finish_reason "length" exists for. cancelled / pause_turn are abandoned turns
|
||||
// whose partial content must stay private, so they are deliberately absent.
|
||||
const KIRO_TRUNCATION_STOP_REASONS = new Set(["model_context_window_exceeded", "max_tokens"]);
|
||||
|
||||
function stopDisposition(stopReason, hasToolCalls) {
|
||||
if (["malformed_model_output", "invalid_model_output"].includes(stopReason)) return "retryable_protocol_failure";
|
||||
if (["cancelled", "pause_turn", "model_context_window_exceeded"].includes(stopReason)) return "terminal_incomplete";
|
||||
@@ -717,12 +711,7 @@ export class KiroExecutor extends BaseExecutor {
|
||||
};
|
||||
const emitTools = (controller) => {
|
||||
for (const tool of state.tools.values()) {
|
||||
// Validate per tool, not per turn: one unusable fragment used to throw out
|
||||
// of emitTools and take every other complete tool call in the same turn
|
||||
// with it, which the client saw as a turn that answered nothing.
|
||||
let input;
|
||||
try {
|
||||
input = parsedToolInput(tool);
|
||||
const input = parsedToolInput(tool);
|
||||
if (tool.name === "tool_call") {
|
||||
if (typeof input.name !== "string" || !input.name.trim()) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool name");
|
||||
@@ -731,12 +720,6 @@ export class KiroExecutor extends BaseExecutor {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool arguments");
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
state.droppedTools = (state.droppedTools || 0) + 1;
|
||||
state.toolValidationError ||= error.message;
|
||||
console.error(`[Kiro] dropping unusable tool call ${tool.id} (${tool.name}): ${error.message}`);
|
||||
continue;
|
||||
}
|
||||
const index = state.toolCounter++;
|
||||
emitDelta(controller, {
|
||||
tool_calls: [{
|
||||
@@ -746,26 +729,14 @@ export class KiroExecutor extends BaseExecutor {
|
||||
function: { name: tool.name, arguments: "" }
|
||||
}]
|
||||
});
|
||||
const serializedInput = JSON.stringify(input);
|
||||
emitDelta(controller, {
|
||||
tool_calls: [{ index, function: { arguments: serializedInput } }]
|
||||
tool_calls: [{ index, function: { arguments: JSON.stringify(input) } }]
|
||||
});
|
||||
// Tool arguments are billed output like any other completion bytes. They
|
||||
// were never added to totalContentLength, so the /4 estimator in finish()
|
||||
// reported OUT 0 -- or the Math.max floor of 1 -- for every turn whose
|
||||
// entire answer was a tool call.
|
||||
state.totalContentLength += tool.name.length + serializedInput.length;
|
||||
state.hasToolCalls = true;
|
||||
}
|
||||
state.tools.clear();
|
||||
state.bufferedToolBytes = 0;
|
||||
// A declared tool turn that emitted no usable call is only fatal when the
|
||||
// turn produced nothing else. Throwing unconditionally here escaped
|
||||
// emitTools() with provenance "invalid_tool_call", which the integrity gate
|
||||
// re-derived into a repair retry -- discarding text the client had already
|
||||
// been promised.
|
||||
if (state.stopReason === "tool_use" && !state.hasToolCalls &&
|
||||
!state.hasText && !state.hasReasoning && !state.hasCode) {
|
||||
if (state.stopReason === "tool_use" && !state.hasToolCalls) {
|
||||
throw new Error("Kiro tool_use stop reason did not include a complete tool call");
|
||||
}
|
||||
};
|
||||
@@ -825,6 +796,7 @@ export class KiroExecutor extends BaseExecutor {
|
||||
emitDelta(controller, { content: event.payload.content });
|
||||
} else if (eventType === "toolUseEvent") {
|
||||
state.sawToolUse = true;
|
||||
if (state.toolValidationError) return true;
|
||||
const values = Array.isArray(event.payload) ? event.payload : [event.payload];
|
||||
if (!values[0]) throw new Error("Kiro toolUseEvent is empty");
|
||||
for (const value of values) {
|
||||
@@ -952,10 +924,9 @@ export class KiroExecutor extends BaseExecutor {
|
||||
} catch (error) {
|
||||
const bufferExceeded = error.code === "KIRO_BUFFER_EXCEEDED";
|
||||
if (!bufferExceeded) {
|
||||
// Keep whatever is already buffered: the rejected fragment belongs to
|
||||
// one tool, and clearing the map dropped the complete calls too.
|
||||
state.toolValidationError ||= error.message;
|
||||
console.error(`[Kiro] tool fragment rejected, keeping ${state.tools.size} buffered tool(s): ${error.message}`);
|
||||
state.tools.clear();
|
||||
state.bufferedToolBytes = 0;
|
||||
continue;
|
||||
}
|
||||
fail(
|
||||
@@ -987,16 +958,7 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
state.transportState = "clean_eof";
|
||||
const declaredDisposition = stopDisposition(state.stopReason, state.sawToolUse);
|
||||
// model_context_window_exceeded / max_tokens map to terminal_incomplete. When
|
||||
// they arrive after the model already streamed content, fail() threw away a
|
||||
// complete-enough answer; a truncated turn is what finish_reason "length" is
|
||||
// for. chunkIndex > 0 means at least one delta already reached the client.
|
||||
const declaredTruncatedAfterOutput = declaredDisposition === "terminal_incomplete" &&
|
||||
KIRO_TRUNCATION_STOP_REASONS.has(state.stopReason) && state.chunkIndex > 0;
|
||||
if (declaredTruncatedAfterOutput) {
|
||||
console.error(`[Kiro] truncated after ${state.chunkIndex} chunk(s) (stop_reason=${state.stopReason}); keeping output`);
|
||||
}
|
||||
if (!declaredTruncatedAfterOutput && ["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(declaredDisposition)) {
|
||||
if (["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(declaredDisposition)) {
|
||||
const code = declaredDisposition === "retryable_protocol_failure"
|
||||
? "kiro_retryable_protocol_failure"
|
||||
: declaredDisposition === "terminal_refusal"
|
||||
@@ -1013,6 +975,16 @@ export class KiroExecutor extends BaseExecutor {
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (state.toolValidationError) {
|
||||
fail(
|
||||
controller,
|
||||
"invalid_tool_call",
|
||||
"invalid_kiro_tool_call",
|
||||
state.toolValidationError,
|
||||
{ transport_state: state.transportState, stop_disposition: "retryable_protocol_failure" }
|
||||
);
|
||||
return;
|
||||
}
|
||||
try {
|
||||
emitTools(controller);
|
||||
} catch (error) {
|
||||
@@ -1025,22 +997,6 @@ export class KiroExecutor extends BaseExecutor {
|
||||
);
|
||||
return;
|
||||
}
|
||||
// Fail only when the turn has nothing usable left. emitTools() validates
|
||||
// per tool and drops just the unusable ones, so this has to run AFTER it:
|
||||
// before, the rejected tool was still buffered and tools.size was never 0.
|
||||
// A turn that also produced text keeps that text -- the dropped call is
|
||||
// logged, not fatal.
|
||||
if (state.toolValidationError && !state.hasToolCalls &&
|
||||
!state.hasText && !state.hasReasoning && !state.hasCode) {
|
||||
fail(
|
||||
controller,
|
||||
"invalid_tool_call",
|
||||
"invalid_kiro_tool_call",
|
||||
state.toolValidationError,
|
||||
{ transport_state: state.transportState, stop_disposition: "retryable_protocol_failure" }
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const hasOutput = state.hasText || state.hasReasoning || state.hasCode || state.hasToolCalls;
|
||||
if (!hasOutput && !state.explicitStop) {
|
||||
@@ -1055,13 +1011,7 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
const disposition = stopDisposition(state.stopReason, state.hasToolCalls);
|
||||
// Same reasoning as declaredTruncatedAfterOutput above.
|
||||
const truncatedAfterOutput = disposition === "terminal_incomplete" &&
|
||||
KIRO_TRUNCATION_STOP_REASONS.has(state.stopReason) && state.chunkIndex > 0;
|
||||
if (truncatedAfterOutput) {
|
||||
console.error(`[Kiro] truncated after ${state.chunkIndex} chunk(s) (stop_reason=${state.stopReason}); closing as length`);
|
||||
}
|
||||
if (!truncatedAfterOutput && ["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(disposition)) {
|
||||
if (["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(disposition)) {
|
||||
const code = disposition === "retryable_protocol_failure"
|
||||
? "kiro_retryable_protocol_failure"
|
||||
: disposition === "terminal_refusal"
|
||||
@@ -1091,9 +1041,7 @@ export class KiroExecutor extends BaseExecutor {
|
||||
total_tokens: prompt + completion
|
||||
};
|
||||
}
|
||||
const finishReason = truncatedAfterOutput
|
||||
? "length"
|
||||
: state.hasToolCalls
|
||||
const finishReason = state.hasToolCalls
|
||||
? "tool_calls"
|
||||
: disposition === "length"
|
||||
? "length"
|
||||
@@ -1104,11 +1052,7 @@ export class KiroExecutor extends BaseExecutor {
|
||||
options.onTerminalState?.(diagnostics({
|
||||
terminal_provenance: state.terminalProvenance || "clean_eventstream_eof",
|
||||
transport_state: state.transportState,
|
||||
// Report what this exit actually did, not the raw disposition. The
|
||||
// integrity gate re-derives its verdict from stop_disposition, so
|
||||
// reporting "terminal_incomplete" for a turn we deliberately kept made
|
||||
// it discard the very bytes we just released to the client.
|
||||
stop_disposition: truncatedAfterOutput ? "length" : disposition
|
||||
stop_disposition: disposition
|
||||
}));
|
||||
};
|
||||
|
||||
|
||||
49
open-sse/executors/opencode-go.js
Normal file
49
open-sse/executors/opencode-go.js
Normal file
@@ -0,0 +1,49 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
|
||||
// Models that use /zen/go/v1/messages (Anthropic/Claude format + x-api-key auth)
|
||||
const MESSAGES_FORMAT_MODELS = new Set([
|
||||
"minimax-m3",
|
||||
"minimax-m2.7",
|
||||
"minimax-m2.5",
|
||||
"qwen3.7-max",
|
||||
"qwen3.7-plus",
|
||||
"qwen3.6-plus",
|
||||
]);
|
||||
|
||||
const BASE = "https://opencode.ai/zen/go/v1";
|
||||
|
||||
export class OpenCodeGoExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode-go", PROVIDERS["opencode-go"]);
|
||||
}
|
||||
|
||||
// buildUrl runs before buildHeaders in BaseExecutor.execute, cache model here
|
||||
buildUrl(model) {
|
||||
this._lastModel = model;
|
||||
return MESSAGES_FORMAT_MODELS.has(model)
|
||||
? `${BASE}/messages`
|
||||
: `${BASE}/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const key = credentials?.apiKey || credentials?.accessToken;
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
|
||||
if (MESSAGES_FORMAT_MODELS.has(this._lastModel)) {
|
||||
headers["x-api-key"] = key;
|
||||
headers["anthropic-version"] = ANTHROPIC_API_VERSION;
|
||||
} else {
|
||||
headers["Authorization"] = `Bearer ${key}`;
|
||||
}
|
||||
|
||||
if (stream) headers["Accept"] = "text/event-stream";
|
||||
return headers;
|
||||
}
|
||||
|
||||
transformRequest(model, body) {
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
}
|
||||
@@ -1,43 +1,16 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
const OPENCODE_UA = "opencode";
|
||||
// Models that use /zen/v1/messages (claude format)
|
||||
const MESSAGES_MODELS = new Set();
|
||||
|
||||
function generateRequestId() {
|
||||
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
}
|
||||
|
||||
function generateSessionId() {
|
||||
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
}
|
||||
|
||||
// Normalize any resolved id into opencode's ses_ format (stable per-conversation)
|
||||
function toOpencodeSession(id) {
|
||||
const stripped = String(id || "").replace(/^ses_/, "").replace(/-/g, "");
|
||||
return stripped ? `ses_${stripped}` : null;
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials) {
|
||||
return toOpencodeSession(resolveSessionId({
|
||||
headers: credentials?.rawHeaders,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
}));
|
||||
}
|
||||
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode", PROVIDERS.opencode);
|
||||
this._currentSessionId = null;
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
this._currentSessionId = resolveOpencodeSession(body, credentials);
|
||||
transformRequest(model, body) {
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
@@ -48,23 +21,12 @@ export class OpenCodeExecutor extends BaseExecutor {
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const lower = {};
|
||||
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
|
||||
|
||||
const downstreamUa = lower["user-agent"] || "";
|
||||
const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode");
|
||||
|
||||
buildHeaders() {
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer public",
|
||||
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
|
||||
"x-opencode-client": lower["x-opencode-client"] || "desktop",
|
||||
"x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(),
|
||||
"x-opencode-request": lower["x-opencode-request"] || generateRequestId(),
|
||||
"x-opencode-project": lower["x-opencode-project"] || "global",
|
||||
"Accept": stream ? "text/event-stream" : "*/*",
|
||||
"x-opencode-client": "desktop",
|
||||
"Accept": "text/event-stream"
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,13 +30,16 @@ import { PROVIDERS } from "../config/providers.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import {
|
||||
QODER_CHAT_URL_ENCODED,
|
||||
QODER_CHAT_BASE_ALT,
|
||||
QODER_CHAT_SIG_PATH,
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
QODER_USERINFO_URL,
|
||||
QODER_MODEL_MAP,
|
||||
QODER_IDE_VERSION,
|
||||
QODER_CLIENT_TYPE,
|
||||
} from "../shared/qoder/constants.js";
|
||||
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
|
||||
import { getQoderModelConfig, resolveQoderModels } from "../services/qoderModels.js";
|
||||
|
||||
/**
|
||||
* Hoist role:"system" messages out of the messages array (Qoder rejects
|
||||
@@ -215,52 +218,6 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a qoder error message indicates a billing/quota block.
|
||||
* Signatures: code 112 (quota exhausted), code 10605 (queue throttle), pricingUrl field.
|
||||
*/
|
||||
function isBillingBlock(inner) {
|
||||
if (!inner || typeof inner !== "string") return false;
|
||||
const lowerMsg = inner.toLowerCase();
|
||||
// Match: {"code":"112",...}, {"code":"10605",...}, or pricingUrl field
|
||||
return /\"code\"\s*:\s*\"(112|10605)\"/.test(inner) || lowerMsg.includes("pricingurl");
|
||||
}
|
||||
|
||||
/**
|
||||
* Peek the first SSE frame to detect billing errors before piping.
|
||||
* Returns { isBilling, statusVal, message, consumed } — `consumed` is every
|
||||
* byte read so far (including the peeked line) so the caller can re-process
|
||||
* it and nothing is dropped from the stream.
|
||||
*/
|
||||
async function peekFirstQoderFrame(reader, decoder) {
|
||||
let consumed = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) return { isBilling: false, consumed, upstreamDone: true };
|
||||
|
||||
consumed += decoder.decode(value, { stream: true });
|
||||
const nl = consumed.indexOf("\n");
|
||||
if (nl === -1) continue; // need a full line first
|
||||
|
||||
const line = consumed.slice(0, nl).replace(/\r$/, "").trim();
|
||||
if (!line.startsWith("data:")) continue;
|
||||
|
||||
const data = line.slice(5).trimStart();
|
||||
if (data === "[DONE]") return { isBilling: false, consumed };
|
||||
|
||||
let envelope;
|
||||
try { envelope = JSON.parse(data); } catch { return { isBilling: false, consumed }; }
|
||||
|
||||
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
|
||||
const inner = typeof envelope.body === "string" ? envelope.body : "";
|
||||
|
||||
if (statusVal !== 200 && isBillingBlock(inner)) {
|
||||
return { isBilling: true, statusVal, message: inner || `qoder billing block (${statusVal})` };
|
||||
}
|
||||
return { isBilling: false, consumed };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap the upstream's `{statusCodeValue, body}` SSE envelope into plain
|
||||
* OpenAI SSE chunks the rest of the chatCore pipeline understands.
|
||||
@@ -275,33 +232,15 @@ async function peekFirstQoderFrame(reader, decoder) {
|
||||
* [DONE]/error frame (agent keepalive). Non-streaming clients drain via
|
||||
* response.text() which hangs until the socket closes — so on terminal
|
||||
* events we cancel the upstream reader and close our stream immediately.
|
||||
*
|
||||
* NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl).
|
||||
* If detected, return 403 response so chatCore marks connection unavailable
|
||||
* and triggers combo fallback instead of leaking error text into chat.
|
||||
*/
|
||||
async function wrapQoderSSE(response, model) {
|
||||
function wrapQoderSSE(response, model) {
|
||||
if (!response.ok || !response.body) return response;
|
||||
|
||||
const decoder = new TextDecoder();
|
||||
const reader = response.body.getReader();
|
||||
|
||||
// Peek first frame to detect billing block
|
||||
const peek = await peekFirstQoderFrame(reader, decoder);
|
||||
if (peek?.isBilling) {
|
||||
// Billing block detected — return 403 so chatCore fails this connection
|
||||
await reader.cancel().catch(() => {});
|
||||
return new Response(
|
||||
JSON.stringify({ error: { message: peek.message, code: peek.statusVal } }),
|
||||
{ status: 403, headers: { "Content-Type": "application/json" } }
|
||||
);
|
||||
}
|
||||
|
||||
// Normal flow: re-process every byte the peek consumed, then continue.
|
||||
let buffer = peek.consumed || "";
|
||||
const upstreamDrained = peek.upstreamDone === true;
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
let doneEmitted = false;
|
||||
const reader = response.body.getReader();
|
||||
|
||||
// Process one already-extracted SSE line (no trailing newline).
|
||||
const processLine = (line, controller) => {
|
||||
@@ -351,28 +290,7 @@ async function wrapQoderSSE(response, model) {
|
||||
// enqueueing would never be re-invoked, hanging consumers like .text().
|
||||
async start(controller) {
|
||||
try {
|
||||
// Drain whatever the peek already pulled off the socket first.
|
||||
let nlSeed;
|
||||
while ((nlSeed = buffer.indexOf("\n")) !== -1) {
|
||||
const line = buffer.slice(0, nlSeed);
|
||||
buffer = buffer.slice(nlSeed + 1);
|
||||
processLine(line, controller);
|
||||
if (doneEmitted) {
|
||||
await reader.cancel().catch(() => {});
|
||||
controller.close();
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (upstreamDrained) {
|
||||
// Peek hit end-of-stream: flush any trailing partial line.
|
||||
buffer += decoder.decode();
|
||||
if (buffer.length > 0) {
|
||||
processLine(buffer, controller);
|
||||
buffer = "";
|
||||
}
|
||||
}
|
||||
|
||||
while (!doneEmitted && !upstreamDrained) {
|
||||
while (!doneEmitted) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) {
|
||||
buffer += decoder.decode();
|
||||
@@ -425,18 +343,98 @@ async function wrapQoderSSE(response, model) {
|
||||
});
|
||||
}
|
||||
|
||||
// ── PAT (Personal Access Token) → job-token exchange ───────────────────────
|
||||
// PATs (pt-...) cannot sign COSY requests directly. Exchange them for a
|
||||
// short-lived job token (jt-...) via /api/v1/jobToken/exchange (plain JSON,
|
||||
// not COSY-signed), then resolve the userId from userinfo. Mirrors the
|
||||
// official qodercli flow. Cached per-PAT until near-expiry.
|
||||
const PAT_PREFIX = "pt-";
|
||||
const PAT_REFRESH_BUFFER_MS = 5 * 60 * 1000;
|
||||
const patJobCache = new Map();
|
||||
|
||||
export function isQoderPat(token) {
|
||||
return typeof token === "string" && token.startsWith(PAT_PREFIX);
|
||||
}
|
||||
|
||||
async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
{
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
"Cosy-Version": QODER_IDE_VERSION,
|
||||
"Cosy-ClientType": QODER_CLIENT_TYPE,
|
||||
},
|
||||
body: JSON.stringify({ personal_token: pat }),
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
throw new Error(`qoder PAT exchange failed: ${res.status} ${text.slice(0, 200)}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
if (!data.token) throw new Error("qoder PAT exchange returned no job token");
|
||||
|
||||
let expiresAt = Date.now() + 24 * 60 * 60 * 1000;
|
||||
if (data.expires_at) {
|
||||
const parsed = Date.parse(data.expires_at);
|
||||
if (!Number.isNaN(parsed)) expiresAt = parsed;
|
||||
} else if (typeof data.expires_in === "number" && data.expires_in > 0) {
|
||||
expiresAt = Date.now() + data.expires_in;
|
||||
}
|
||||
return { jobToken: data.token, jobRefreshToken: data.refresh_token || "", expiresAt };
|
||||
}
|
||||
|
||||
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null) {
|
||||
try {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_USERINFO_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${jobToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
},
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
const info = await res.json().catch(() => ({}));
|
||||
return info.id || info.userId || info.user_id || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Exchange a PAT for a job token + userId, caching until near-expiry so repeat
|
||||
* chat requests don't re-exchange. Returns { accessToken, userId }.
|
||||
*/
|
||||
async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
|
||||
const cached = patJobCache.get(pat);
|
||||
if (cached && cached.expiresAt - Date.now() > PAT_REFRESH_BUFFER_MS) {
|
||||
return cached;
|
||||
}
|
||||
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal);
|
||||
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal);
|
||||
const entry = { accessToken: jobToken, userId, expiresAt };
|
||||
patJobCache.set(pat, entry);
|
||||
return entry;
|
||||
}
|
||||
|
||||
export class QoderExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("qoder", PROVIDERS.qoder);
|
||||
}
|
||||
|
||||
buildUrl(credentials) {
|
||||
// Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt-
|
||||
// with "Login expired" (403). Device tokens (dt-...) stay on api3.
|
||||
const raw = credentials?.apiKey || credentials?.accessToken;
|
||||
if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) {
|
||||
return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
|
||||
}
|
||||
buildUrl() {
|
||||
return QODER_CHAT_URL_ENCODED;
|
||||
}
|
||||
|
||||
@@ -446,24 +444,36 @@ export class QoderExecutor extends BaseExecutor {
|
||||
// - COSY headers built from the *encoded* body bytes
|
||||
// - response stream re-wrapped from {statusCodeValue, body} to OpenAI SSE
|
||||
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
|
||||
const url = this.buildUrl();
|
||||
|
||||
// PAT (pt-...) → exchange for short-lived job token + resolve userId so
|
||||
// downstream COSY signing + catalog fetch work. Device tokens (dt-...) and
|
||||
// job tokens (jt-...) skip this and are used directly.
|
||||
const rawToken = credentials?.apiKey || credentials?.accessToken;
|
||||
if (isQoderPat(rawToken)) {
|
||||
try {
|
||||
credentials = await resolveQoderCredentials(credentials, proxyOptions, signal);
|
||||
const resolved = await resolvePatCredential(rawToken, proxyOptions, signal);
|
||||
credentials = {
|
||||
...credentials,
|
||||
accessToken: resolved.accessToken,
|
||||
apiKey: undefined,
|
||||
providerSpecificData: {
|
||||
authMethod: "pat",
|
||||
...(credentials?.providerSpecificData || {}),
|
||||
userId: resolved.userId || credentials?.providerSpecificData?.userId || "",
|
||||
machineId: credentials?.providerSpecificData?.machineId || "",
|
||||
},
|
||||
};
|
||||
} catch (err) {
|
||||
log?.error?.("QODER", `PAT exchange failed: ${err.message}`);
|
||||
const fakeResp = new Response(
|
||||
JSON.stringify({ error: { message: `qoder PAT exchange failed: ${err.message}` } }),
|
||||
{ status: 401, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
return { response: fakeResp, url: this.buildUrl(credentials), headers: {}, transformedBody: body };
|
||||
return { response: fakeResp, url, headers: {}, transformedBody: body };
|
||||
}
|
||||
}
|
||||
|
||||
const url = this.buildUrl(credentials);
|
||||
const psd = credentials?.providerSpecificData || {};
|
||||
if (!psd.userId) {
|
||||
// No user id → no way to sign. Surface a 401 so the dashboard nudges
|
||||
@@ -536,7 +546,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
};
|
||||
|
||||
// Abort if upstream doesn't return response headers within connect timeout.
|
||||
const timeoutMs = this.config?.timeoutMs || FETCH_CONNECT_TIMEOUT_MS;
|
||||
const timeoutMs = await resolveProviderTimeoutMs(this.provider, this.config?.timeoutMs, FETCH_CONNECT_TIMEOUT_MS);
|
||||
const connectCtrl = new AbortController();
|
||||
const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs);
|
||||
const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal;
|
||||
@@ -557,7 +567,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
return { response, url, headers, transformedBody: payload };
|
||||
}
|
||||
|
||||
const wrapped = await wrapQoderSSE(response, `qoder/${qoderKey}`);
|
||||
const wrapped = wrapQoderSSE(response, `qoder/${qoderKey}`);
|
||||
return { response: wrapped, url, headers, transformedBody: payload };
|
||||
}
|
||||
|
||||
@@ -581,5 +591,6 @@ export const __test__ = {
|
||||
normalizeMessages,
|
||||
wrapQoderSSE,
|
||||
buildQoderRequestBody,
|
||||
isBillingBlock,
|
||||
isQoderPat,
|
||||
resolvePatCredential,
|
||||
};
|
||||
|
||||
129
open-sse/executors/qwen.js
Normal file
129
open-sse/executors/qwen.js
Normal file
@@ -0,0 +1,129 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS } from "../config/appConstants.js";
|
||||
|
||||
/** portal.qwen.ai — static fingerprint matching stable Qwen Code release */
|
||||
const QWEN_USER_AGENT = "QwenCode/0.12.3 (linux; x64)";
|
||||
const QWEN_STAINLESS = {
|
||||
os: "Linux",
|
||||
arch: "x64",
|
||||
lang: "js",
|
||||
runtime: "node",
|
||||
runtimeVersion: "v18.19.1",
|
||||
packageVersion: "5.11.0",
|
||||
retryCount: "1"
|
||||
};
|
||||
const QWEN_DEFAULT_SYSTEM_MESSAGE = {
|
||||
role: "system",
|
||||
content: [{ type: "text", text: "", cache_control: { type: "ephemeral" } }]
|
||||
};
|
||||
|
||||
function ensureQwenSystemMessage(body) {
|
||||
if (!body || typeof body !== "object") return body;
|
||||
const next = { ...body };
|
||||
if (Array.isArray(next.messages)) {
|
||||
next.messages = [QWEN_DEFAULT_SYSTEM_MESSAGE, ...next.messages];
|
||||
} else {
|
||||
next.messages = [QWEN_DEFAULT_SYSTEM_MESSAGE];
|
||||
}
|
||||
return next;
|
||||
}
|
||||
|
||||
function isQwenThinkingActive(body) {
|
||||
const thinking = body?.thinking;
|
||||
if (thinking === true || body?.enable_thinking === true) return true;
|
||||
return typeof thinking === "object" && thinking !== null && !Array.isArray(thinking) && thinking.type === "enabled";
|
||||
}
|
||||
|
||||
// Qwen rejects tool_choice="required" or object forms when thinking is active; neutralize to "auto".
|
||||
function sanitizeQwenThinkingToolChoice(body) {
|
||||
if (!isQwenThinkingActive(body)) return body;
|
||||
const tc = body.tool_choice;
|
||||
const incompatible = tc === "required" || (typeof tc === "object" && tc !== null);
|
||||
if (!incompatible) return body;
|
||||
return { ...body, tool_choice: "auto" };
|
||||
}
|
||||
|
||||
function buildQwenUpstreamHeaders(credentials, stream = true) {
|
||||
const token = credentials?.apiKey || credentials?.accessToken || "";
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${token}`,
|
||||
"User-Agent": QWEN_USER_AGENT,
|
||||
"X-DashScope-AuthType": "qwen-oauth",
|
||||
"X-DashScope-CacheControl": "enable",
|
||||
"X-DashScope-UserAgent": QWEN_USER_AGENT,
|
||||
"X-Stainless-Arch": QWEN_STAINLESS.arch,
|
||||
"X-Stainless-Lang": QWEN_STAINLESS.lang,
|
||||
"X-Stainless-Os": QWEN_STAINLESS.os,
|
||||
"X-Stainless-Package-Version": QWEN_STAINLESS.packageVersion,
|
||||
"X-Stainless-Retry-Count": QWEN_STAINLESS.retryCount,
|
||||
"X-Stainless-Runtime": QWEN_STAINLESS.runtime,
|
||||
"X-Stainless-Runtime-Version": QWEN_STAINLESS.runtimeVersion,
|
||||
Connection: "keep-alive",
|
||||
"Accept-Language": "*",
|
||||
"Sec-Fetch-Mode": "cors"
|
||||
};
|
||||
headers.Accept = stream ? "text/event-stream" : "application/json";
|
||||
return headers;
|
||||
}
|
||||
|
||||
export class QwenExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("qwen");
|
||||
}
|
||||
|
||||
// Qwen tokens are bound to a resource_url returned at OAuth time.
|
||||
// Using portal.qwen.ai when the token is issued for another shard returns 401/403.
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
const resourceUrl = credentials?.providerSpecificData?.resourceUrl;
|
||||
const host = resourceUrl ? resourceUrl.replace(/^https?:\/\//, "").replace(/\/$/, "") : "portal.qwen.ai";
|
||||
return `https://${host}/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
return buildQwenUpstreamHeaders(credentials, stream);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
let next = body && typeof body === "object" ? { ...body } : body;
|
||||
if (stream && next?.messages && !next.stream_options && !next.thinking && !next.enable_thinking && next.stream !== false) {
|
||||
next.stream_options = { include_usage: true };
|
||||
}
|
||||
next = sanitizeQwenThinkingToolChoice(next);
|
||||
return ensureQwenSystemMessage(next);
|
||||
}
|
||||
|
||||
// Override to capture resource_url from refresh response (required for buildUrl).
|
||||
async refreshCredentials(credentials, log) {
|
||||
if (!credentials?.refreshToken) return null;
|
||||
try {
|
||||
const response = await fetch(OAUTH_ENDPOINTS.qwen.token, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json" },
|
||||
body: new URLSearchParams({
|
||||
grant_type: "refresh_token",
|
||||
refresh_token: credentials.refreshToken,
|
||||
client_id: PROVIDERS.qwen.clientId
|
||||
})
|
||||
});
|
||||
if (!response.ok) return null;
|
||||
const tokens = await response.json();
|
||||
log?.info?.("TOKEN", "qwen refreshed");
|
||||
return {
|
||||
accessToken: tokens.access_token,
|
||||
refreshToken: tokens.refresh_token || credentials.refreshToken,
|
||||
expiresIn: tokens.expires_in,
|
||||
providerSpecificData: {
|
||||
...(credentials.providerSpecificData || {}),
|
||||
...(tokens.resource_url ? { resourceUrl: tokens.resource_url } : {})
|
||||
}
|
||||
};
|
||||
} catch (error) {
|
||||
log?.error?.("TOKEN", `qwen refresh error: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export default QwenExecutor;
|
||||
@@ -1,33 +1,67 @@
|
||||
import { detectFormat, getTargetFormat, resolveTransport } from "../services/provider.js";
|
||||
import {
|
||||
detectFormat,
|
||||
getTargetFormat,
|
||||
resolveTransport,
|
||||
} from "../services/provider.js";
|
||||
import { translateRequest } from "../translator/index.js";
|
||||
import { applyThinking, extractThinking, stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { normalizeClaudePassthrough, anchorClaudeCache } from "../translator/formats/claude.js";
|
||||
import { normalizeClaudePassthrough } from "../translator/formats/claude.js";
|
||||
import { createStreamController } from "../utils/streamHandler.js";
|
||||
import { refreshWithRetry } from "../services/tokenRefresh.js";
|
||||
import { createRequestLogger } from "../utils/requestLogger.js";
|
||||
import { getModelTargetFormat, getModelSupportedFormats, getModelStrip, getModelUpstreamId, getModelType, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
|
||||
import {
|
||||
getModelTargetFormat,
|
||||
getModelStrip,
|
||||
getModelUpstreamId,
|
||||
getModelType,
|
||||
PROVIDER_ID_TO_ALIAS,
|
||||
} from "../config/providerModels.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import {
|
||||
createErrorResult,
|
||||
parseUpstreamError,
|
||||
formatProviderError,
|
||||
} from "../utils/error.js";
|
||||
import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js";
|
||||
import { handleBypassRequest } from "../utils/bypassHandler.js";
|
||||
import { trackPendingRequest, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import {
|
||||
trackPendingRequest,
|
||||
appendRequestLog,
|
||||
saveRequestDetail,
|
||||
} from "@/lib/usageDb.js";
|
||||
import { getExecutor } from "../executors/index.js";
|
||||
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
||||
import { buildRequestDetail, extractRequestConfig } from "./chatCore/requestDetail.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
} from "./chatCore/requestDetail.js";
|
||||
import { handleForcedSSEToJson } from "./chatCore/sseToJsonHandler.js";
|
||||
import { handleNonStreamingResponse } from "./chatCore/nonStreamingHandler.js";
|
||||
import { handleStreamingResponse, buildOnStreamComplete } from "./chatCore/streamingHandler.js";
|
||||
import { detectClientTool, isNativePassthrough } from "../utils/clientDetector.js";
|
||||
import {
|
||||
handleStreamingResponse,
|
||||
buildOnStreamComplete,
|
||||
} from "./chatCore/streamingHandler.js";
|
||||
import { maybeRejectEarlyStreamError } from "../utils/streamErrorPeek.js";
|
||||
import {
|
||||
detectClientTool,
|
||||
isNativePassthrough,
|
||||
} from "../utils/clientDetector.js";
|
||||
import { dedupeTools } from "../utils/toolDeduper.js";
|
||||
import { injectCaveman } from "../rtk/caveman.js";
|
||||
import { injectPonytail } from "../rtk/ponytail.js";
|
||||
import { compressMessages, formatRtkLog } from "../rtk/index.js";
|
||||
import { compressWithHeadroom, formatHeadroomLog, formatHeadroomSizeLog, isHeadroomPhantomSavings } from "../rtk/headroom.js";
|
||||
import {
|
||||
compressWithHeadroom,
|
||||
formatHeadroomLog,
|
||||
formatHeadroomSizeLog,
|
||||
isHeadroomPhantomSavings,
|
||||
} from "../rtk/headroom.js";
|
||||
import { compressWithPxpipe } from "../rtk/pxpipe.js";
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
|
||||
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
import { extractThinking } from "../translator/concerns/thinkingUnified.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
/**
|
||||
@@ -37,61 +71,76 @@ import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
* @param {object} options.credentials - Provider credentials
|
||||
* @param {string} options.sourceFormatOverride - Override detected source format (e.g. "openai-responses")
|
||||
*/
|
||||
/**
|
||||
* Remove translator-internal continuity fields from the outbound upstream
|
||||
* body. The Responses→Chat request translator stashes reasoning
|
||||
* `encrypted_content` on assistant messages so a later openai→responses
|
||||
* round-trip can restore the store=false continuity blob; that stash must
|
||||
* never reach an upstream provider. Chat-native proxies reject the unknown
|
||||
* assistant-message field and answer every turn with a literal "400" body
|
||||
* (observed with multi-turn Codex sessions via OpenAI-compatible nodes).
|
||||
*/
|
||||
export function stripContinuityFields(body) {
|
||||
if (!body || !Array.isArray(body.messages)) return body;
|
||||
for (const msg of body.messages) {
|
||||
if (msg && typeof msg === "object") {
|
||||
delete msg.encrypted_content;
|
||||
delete msg.reasoning_encrypted_content;
|
||||
}
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking }) {
|
||||
export async function handleChatCore({
|
||||
body,
|
||||
modelInfo,
|
||||
credentials,
|
||||
log,
|
||||
onCredentialsRefreshed,
|
||||
onRequestSuccess,
|
||||
onDisconnect,
|
||||
clientRawRequest,
|
||||
connectionId,
|
||||
userAgent,
|
||||
apiKey,
|
||||
ccFilterNaming,
|
||||
rtkEnabled,
|
||||
headroomEnabled,
|
||||
headroomUrl,
|
||||
headroomCompressUserMessages,
|
||||
cavemanEnabled,
|
||||
cavemanLevel,
|
||||
ponytailEnabled,
|
||||
ponytailLevel,
|
||||
pxpipeEnabled,
|
||||
pxpipeMinChars,
|
||||
pxpipeTimeoutMs,
|
||||
pxpipeTransform,
|
||||
onPxpipeEvent,
|
||||
sourceFormatOverride,
|
||||
providerThinking,
|
||||
streamErrorPatterns,
|
||||
}) {
|
||||
const { provider, model } = modelInfo;
|
||||
const requestStartTime = Date.now();
|
||||
// Stable per-session color so all lines of one CLI conversation share a tag
|
||||
const sessionSeed = (() => {
|
||||
try {
|
||||
return resolveSessionId({ headers: clientRawRequest?.headers, body, connectionId, scope: provider });
|
||||
return resolveSessionId({
|
||||
headers: clientRawRequest?.headers,
|
||||
body,
|
||||
connectionId,
|
||||
scope: provider,
|
||||
});
|
||||
} catch {
|
||||
return connectionId || "";
|
||||
}
|
||||
})();
|
||||
const reqTag = log?.tagForSession ? log.tagForSession(sessionSeed) : (log?.nextTag ? log.nextTag() : "");
|
||||
const reqTag = log?.tagForSession
|
||||
? log.tagForSession(sessionSeed)
|
||||
: log?.nextTag
|
||||
? log.nextTag()
|
||||
: "";
|
||||
|
||||
const sourceFormat = sourceFormatOverride || detectFormat(body);
|
||||
|
||||
// Check for bypass patterns (warmup, skip, cc naming)
|
||||
const bypassResponse = handleBypassRequest(body, model, userAgent, ccFilterNaming);
|
||||
const bypassResponse = handleBypassRequest(
|
||||
body,
|
||||
model,
|
||||
userAgent,
|
||||
ccFilterNaming,
|
||||
);
|
||||
if (bypassResponse) return bypassResponse;
|
||||
|
||||
const alias = PROVIDER_ID_TO_ALIAS[provider] || provider;
|
||||
const modelTargetFormat = getModelTargetFormat(alias, model);
|
||||
// Multi-endpoint providers: pick transport matching sourceFormat → zero translation.
|
||||
// Per-model guard: only use the transport when the model declares support for that
|
||||
// sourceFormat — opencode-go models differ in endpoint support (kimi/glm only do
|
||||
// /chat/completions), so without this guard a claude-format request would wrongly
|
||||
// route kimi to /messages.
|
||||
const modelSupportedFormats = getModelSupportedFormats(alias, model);
|
||||
// Multi-endpoint providers: pick transport matching sourceFormat → zero translation
|
||||
const runtimeTransport = resolveTransport(provider, sourceFormat);
|
||||
// Per-model guard: when a model declares supportedFormats, only use the
|
||||
// sourceFormat-matched transport if that format is declared (opencode-go models
|
||||
// differ — kimi/glm only do /chat/completions). Undeclared models keep the
|
||||
// upstream default (use the transport), preserving behavior for glm/deepseek/...
|
||||
const useTransport = (!modelSupportedFormats || modelSupportedFormats.includes(sourceFormat)) ? runtimeTransport : null;
|
||||
const targetFormat = modelTargetFormat || useTransport?.format || getTargetFormat(provider, credentials);
|
||||
if (useTransport && credentials) credentials.runtimeTransport = useTransport;
|
||||
const targetFormat =
|
||||
modelTargetFormat || runtimeTransport?.format || getTargetFormat(provider);
|
||||
if (runtimeTransport && credentials)
|
||||
credentials.runtimeTransport = runtimeTransport;
|
||||
const stripList = getModelStrip(alias, model);
|
||||
const upstreamModel = getModelUpstreamId(alias, model);
|
||||
|
||||
@@ -109,14 +158,22 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
}
|
||||
}
|
||||
|
||||
const clientRequestedStreaming = body.stream === true || sourceFormat === FORMATS.ANTIGRAVITY || sourceFormat === FORMATS.GEMINI || sourceFormat === FORMATS.GEMINI_CLI;
|
||||
const clientRequestedStreaming =
|
||||
body.stream === true ||
|
||||
sourceFormat === FORMATS.ANTIGRAVITY ||
|
||||
sourceFormat === FORMATS.GEMINI ||
|
||||
sourceFormat === FORMATS.GEMINI_CLI;
|
||||
const providerRequiresStreaming = PROVIDERS[provider]?.forceStream === true;
|
||||
let stream = providerRequiresStreaming ? true : (body.stream !== false);
|
||||
let stream = providerRequiresStreaming ? true : body.stream !== false;
|
||||
|
||||
// Image generation models require non-streaming (Google v1internal:generateContent)
|
||||
const modelType = getModelType(alias, model);
|
||||
const isImageGenModel = modelType === "imageGen" || /image|imagen|image-generation/i.test(model);
|
||||
if (isImageGenModel && (provider === "antigravity" || provider === "gemini-cli")) {
|
||||
const isImageGenModel =
|
||||
modelType === "imageGen" || /image|imagen|image-generation/i.test(model);
|
||||
if (
|
||||
isImageGenModel &&
|
||||
(provider === "antigravity" || provider === "gemini-cli")
|
||||
) {
|
||||
stream = false;
|
||||
}
|
||||
|
||||
@@ -131,14 +188,31 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
const acceptHeader = clientRawRequest?.headers?.accept || "";
|
||||
const clientPrefersJson = acceptHeader.includes("application/json");
|
||||
const clientPrefersSSE = acceptHeader.includes("text/event-stream");
|
||||
if (clientPrefersJson && !clientPrefersSSE && body.stream !== true && !providerRequiresStreaming) {
|
||||
if (
|
||||
clientPrefersJson &&
|
||||
!clientPrefersSSE &&
|
||||
body.stream !== true &&
|
||||
!providerRequiresStreaming
|
||||
) {
|
||||
stream = false;
|
||||
}
|
||||
|
||||
const reqLogger = await createRequestLogger(sourceFormat, targetFormat, model);
|
||||
if (clientRawRequest) reqLogger.logClientRawRequest(clientRawRequest.endpoint, clientRawRequest.body, clientRawRequest.headers);
|
||||
const reqLogger = await createRequestLogger(
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
model,
|
||||
);
|
||||
if (clientRawRequest)
|
||||
reqLogger.logClientRawRequest(
|
||||
clientRawRequest.endpoint,
|
||||
clientRawRequest.body,
|
||||
clientRawRequest.headers,
|
||||
);
|
||||
reqLogger.logRawRequest(body);
|
||||
log?.debug?.("FORMAT", `${sourceFormat} → ${targetFormat} | stream=${stream}`);
|
||||
log?.debug?.(
|
||||
"FORMAT",
|
||||
`${sourceFormat} → ${targetFormat} | stream=${stream}`,
|
||||
);
|
||||
|
||||
// Native passthrough: CLI tool and provider are the same ecosystem
|
||||
// Skip all translation/normalization — only model and Bearer are swapped
|
||||
@@ -152,47 +226,61 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
if (!passthrough) {
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (stripUnsupportedModalities(body, sourceFormat, caps)) {
|
||||
log?.debug?.("MODALITY", `stripped unsupported media for ${provider}/${model}`);
|
||||
log?.debug?.(
|
||||
"MODALITY",
|
||||
`stripped unsupported media for ${provider}/${model}`,
|
||||
);
|
||||
}
|
||||
// Convert remote image URLs to base64 for targets that can't fetch URLs.
|
||||
try {
|
||||
const n = await prefetchRemoteImages(body, sourceFormat, targetFormat, { signal: undefined });
|
||||
if (n > 0) log?.debug?.("MODALITY", `prefetched ${n} remote image(s) for ${targetFormat}`);
|
||||
} catch (e) { log?.warn?.("MODALITY", `image prefetch failed: ${e.message}`); }
|
||||
const n = await prefetchRemoteImages(body, sourceFormat, targetFormat, {
|
||||
signal: undefined,
|
||||
});
|
||||
if (n > 0)
|
||||
log?.debug?.(
|
||||
"MODALITY",
|
||||
`prefetched ${n} remote image(s) for ${targetFormat}`,
|
||||
);
|
||||
} catch (e) {
|
||||
log?.warn?.("MODALITY", `image prefetch failed: ${e.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
let translatedBody;
|
||||
let toolNameMap;
|
||||
let customToolNames;
|
||||
if (passthrough) {
|
||||
log?.debug?.("PASSTHROUGH", `${clientTool} → ${provider} | native lossless`);
|
||||
log?.debug?.(
|
||||
"PASSTHROUGH",
|
||||
`${clientTool} → ${provider} | native lossless`,
|
||||
);
|
||||
translatedBody = { ...body, model: stripThinkingSuffix(upstreamModel) };
|
||||
if (provider === "codex") {
|
||||
const suffixThinking = {};
|
||||
applyThinking(sourceFormat, upstreamModel, suffixThinking, provider);
|
||||
if (suffixThinking.reasoning_effort) {
|
||||
const reasoning = translatedBody.reasoning;
|
||||
translatedBody.reasoning = {
|
||||
...(reasoning && typeof reasoning === "object" && !Array.isArray(reasoning) ? reasoning : {}),
|
||||
effort: suffixThinking.reasoning_effort,
|
||||
};
|
||||
delete translatedBody.reasoning_effort;
|
||||
}
|
||||
}
|
||||
// Normalize newer Cowork/CC beta shapes (adaptive thinking, mid-conversation system) the API rejects
|
||||
if (clientTool === "claude") normalizeClaudePassthrough(translatedBody, translatedBody.model);
|
||||
if (clientTool === "claude")
|
||||
normalizeClaudePassthrough(translatedBody, translatedBody.model);
|
||||
} else {
|
||||
translatedBody = translateRequest(sourceFormat, targetFormat, upstreamModel, body, stream, credentials, provider, reqLogger, stripList, connectionId, clientTool);
|
||||
translatedBody = translateRequest(
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
upstreamModel,
|
||||
body,
|
||||
stream,
|
||||
credentials,
|
||||
provider,
|
||||
reqLogger,
|
||||
stripList,
|
||||
connectionId,
|
||||
clientTool,
|
||||
);
|
||||
if (!translatedBody) {
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Failed to translate request for ${sourceFormat} → ${targetFormat}`);
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_REQUEST,
|
||||
`Failed to translate request for ${sourceFormat} → ${targetFormat}`,
|
||||
);
|
||||
}
|
||||
toolNameMap = translatedBody._toolNameMap;
|
||||
delete translatedBody._toolNameMap;
|
||||
customToolNames = translatedBody._customToolNames;
|
||||
delete translatedBody._customToolNames;
|
||||
translatedBody.model = stripThinkingSuffix(upstreamModel);
|
||||
stripContinuityFields(translatedBody);
|
||||
}
|
||||
|
||||
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
|
||||
@@ -200,7 +288,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools);
|
||||
if (stripped.length > 0) {
|
||||
translatedBody.tools = deduped;
|
||||
log?.debug?.("TOOLDEDUP", `stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`);
|
||||
log?.debug?.(
|
||||
"TOOLDEDUP",
|
||||
`stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -211,12 +302,26 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
// Request line: one correlated summary (fmt + thinking + counts + account)
|
||||
if (log?.line) {
|
||||
const clientModel = clientRawRequest?.body?.model || `${provider}/${model}`;
|
||||
const msgN = translatedBody.messages?.length || translatedBody.input?.length || translatedBody.contents?.length || body.messages?.length || body.input?.length || 0;
|
||||
const msgN =
|
||||
translatedBody.messages?.length ||
|
||||
translatedBody.input?.length ||
|
||||
translatedBody.contents?.length ||
|
||||
body.messages?.length ||
|
||||
body.input?.length ||
|
||||
0;
|
||||
const toolN = translatedBody.tools?.length || body.tools?.length || 0;
|
||||
const fmtStr = passthrough ? `FMT: ${sourceFormat} (passthrough)` : `FMT: ${sourceFormat}→${targetFormat}`;
|
||||
const showThinking = provider !== "grok-cli" || supportsGrokCliReasoningEffort(model);
|
||||
const think = showThinking ? log.fmtThink?.(extractThinking(translatedBody)) : null;
|
||||
const acc = credentials?.connectionName || credentials?.connectionId?.slice(0, 8) || "-";
|
||||
const fmtStr = passthrough
|
||||
? `FMT: ${sourceFormat} (passthrough)`
|
||||
: `FMT: ${sourceFormat}→${targetFormat}`;
|
||||
const showThinking =
|
||||
provider !== "grok-cli" || supportsGrokCliReasoningEffort(model);
|
||||
const think = showThinking
|
||||
? log.fmtThink?.(extractThinking(translatedBody))
|
||||
: null;
|
||||
const acc =
|
||||
credentials?.connectionName ||
|
||||
credentials?.connectionId?.slice(0, 8) ||
|
||||
"-";
|
||||
const parts = [
|
||||
`POST ${clientModel} → ${provider}/${model}`,
|
||||
fmtStr,
|
||||
@@ -231,29 +336,52 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
|
||||
// TTS models don't support tool messages/function calling
|
||||
if (getModelType(alias, model) === "tts" && translatedBody.messages) {
|
||||
translatedBody.messages = translatedBody.messages.filter(msg => msg.role !== "tool");
|
||||
translatedBody.messages = translatedBody.messages.filter(
|
||||
(msg) => msg.role !== "tool",
|
||||
);
|
||||
delete translatedBody.tools;
|
||||
}
|
||||
|
||||
// Per-request opt-out: client can bypass all token savers via header
|
||||
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
|
||||
const tokenSaverEnabled =
|
||||
clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
|
||||
|
||||
// RTK: compress tool_result content
|
||||
const rtkStats = compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled);
|
||||
const rtkStats = compressMessages(
|
||||
translatedBody,
|
||||
tokenSaverEnabled && rtkEnabled,
|
||||
);
|
||||
const rtkLine = formatRtkLog(rtkStats);
|
||||
if (rtkLine) console.log(rtkLine);
|
||||
|
||||
// Headroom: optional external proxy compression; fail open if proxy is absent.
|
||||
const headroomDiagnostics = {};
|
||||
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, diagnostics: headroomDiagnostics });
|
||||
const headroomStats = await compressWithHeadroom(translatedBody, {
|
||||
enabled: tokenSaverEnabled && headroomEnabled,
|
||||
url: headroomUrl,
|
||||
model: upstreamModel,
|
||||
format: finalFormat,
|
||||
compressUserMessages: headroomCompressUserMessages,
|
||||
diagnostics: headroomDiagnostics,
|
||||
});
|
||||
const headroomLine = formatHeadroomLog(headroomStats);
|
||||
const headroomSizeLine = formatHeadroomSizeLog(headroomDiagnostics);
|
||||
if (headroomLine) {
|
||||
log?.info?.("HEADROOM", `${headroomLine}${headroomSizeLine ? ` | ${headroomSizeLine}` : ""}`);
|
||||
log?.info?.(
|
||||
"HEADROOM",
|
||||
`${headroomLine}${headroomSizeLine ? ` | ${headroomSizeLine}` : ""}`,
|
||||
);
|
||||
if (isHeadroomPhantomSavings(headroomStats, headroomDiagnostics)) {
|
||||
log?.warn?.("HEADROOM", `reported token delta, but outbound JSON shrank <5%; provider may bill near-original payload | ${formatHeadroomSizeLog(headroomDiagnostics)}`);
|
||||
log?.warn?.(
|
||||
"HEADROOM",
|
||||
`reported token delta, but outbound JSON shrank <5%; provider may bill near-original payload | ${formatHeadroomSizeLog(headroomDiagnostics)}`,
|
||||
);
|
||||
}
|
||||
} else if (tokenSaverEnabled && headroomEnabled) log?.warn?.("HEADROOM", `skipped: ${headroomDiagnostics.reason || "compression unavailable"}${headroomDiagnostics.endpoint ? ` (${headroomDiagnostics.endpoint})` : ""}`);
|
||||
} else if (tokenSaverEnabled && headroomEnabled)
|
||||
log?.warn?.(
|
||||
"HEADROOM",
|
||||
`skipped: ${headroomDiagnostics.reason || "compression unavailable"}${headroomDiagnostics.endpoint ? ` (${headroomDiagnostics.endpoint})` : ""}`,
|
||||
);
|
||||
|
||||
// Token-saver flags accumulator for the single "⚙" log line below.
|
||||
const xf = [];
|
||||
@@ -274,27 +402,42 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
let pxpipeSummary = null;
|
||||
if (pxpipeEnabled) {
|
||||
const pxpipeResult = await compressWithPxpipe(translatedBody, {
|
||||
enabled: true, format: finalFormat, model: upstreamModel,
|
||||
minChars: pxpipeMinChars, timeoutMs: pxpipeTimeoutMs, transform: pxpipeTransform,
|
||||
enabled: true,
|
||||
format: finalFormat,
|
||||
model: upstreamModel,
|
||||
minChars: pxpipeMinChars,
|
||||
timeoutMs: pxpipeTimeoutMs,
|
||||
transform: pxpipeTransform,
|
||||
});
|
||||
pxpipeSummary = pxpipeResult.summary;
|
||||
if (pxpipeResult.body) translatedBody = pxpipeResult.body;
|
||||
if (pxpipeSummary?.applied) xf.push(`PXPIPE:${pxpipeSummary.imageCount}img`);
|
||||
try { onPxpipeEvent?.({ provider, model, ...pxpipeSummary }); } catch { /* stats must not break requests */ }
|
||||
if (pxpipeSummary?.applied)
|
||||
xf.push(`PXPIPE:${pxpipeSummary.imageCount}img`);
|
||||
try {
|
||||
onPxpipeEvent?.({ provider, model, ...pxpipeSummary });
|
||||
} catch {
|
||||
/* stats must not break requests */
|
||||
}
|
||||
}
|
||||
|
||||
if (xf.length && log?.line) log.line(reqTag, "⚙", xf.join(" · "));
|
||||
|
||||
// Pin cache breakpoints to the final body — every saver above can reshape
|
||||
// system/tools/messages, and a stale anchor costs a full prefix rewrite.
|
||||
if (passthrough && clientTool === "claude") anchorClaudeCache(translatedBody);
|
||||
|
||||
const executor = getExecutor(provider);
|
||||
trackPendingRequest(model, provider, connectionId, true);
|
||||
appendRequestLog({ model, provider, connectionId, status: "PENDING" }).catch(() => { });
|
||||
appendRequestLog({ model, provider, connectionId, status: "PENDING" }).catch(
|
||||
() => {},
|
||||
);
|
||||
|
||||
const msgCount = translatedBody.messages?.length || translatedBody.input?.length || translatedBody.contents?.length || translatedBody.request?.contents?.length || 0;
|
||||
log?.debug?.("REQUEST", `${provider.toUpperCase()} | ${model} | ${msgCount} msgs`);
|
||||
const msgCount =
|
||||
translatedBody.messages?.length ||
|
||||
translatedBody.input?.length ||
|
||||
translatedBody.contents?.length ||
|
||||
translatedBody.request?.contents?.length ||
|
||||
0;
|
||||
log?.debug?.(
|
||||
"REQUEST",
|
||||
`${provider.toUpperCase()} | ${model} | ${msgCount} msgs`,
|
||||
);
|
||||
|
||||
const streamController = createStreamController({
|
||||
onDisconnect: (reason) => {
|
||||
@@ -302,21 +445,35 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
if (onDisconnect) onDisconnect(reason);
|
||||
},
|
||||
onError: () => trackPendingRequest(model, provider, connectionId, false),
|
||||
log, provider, model, reqTag
|
||||
log,
|
||||
provider,
|
||||
model,
|
||||
reqTag,
|
||||
});
|
||||
|
||||
const proxyOptions = {
|
||||
connectionProxyEnabled: credentials?.providerSpecificData?.connectionProxyEnabled === true,
|
||||
connectionProxyUrl: credentials?.providerSpecificData?.connectionProxyUrl || "",
|
||||
connectionNoProxy: credentials?.providerSpecificData?.connectionNoProxy || "",
|
||||
connectionProxyEnabled:
|
||||
credentials?.providerSpecificData?.connectionProxyEnabled === true,
|
||||
connectionProxyUrl:
|
||||
credentials?.providerSpecificData?.connectionProxyUrl || "",
|
||||
connectionNoProxy:
|
||||
credentials?.providerSpecificData?.connectionNoProxy || "",
|
||||
vercelRelayUrl: credentials?.providerSpecificData?.vercelRelayUrl || "",
|
||||
};
|
||||
|
||||
if (proxyOptions.vercelRelayUrl) {
|
||||
const connectionName = credentials?.connectionName || credentials?.connectionId || "unknown";
|
||||
const poolId = credentials?.providerSpecificData?.connectionProxyPoolId || "none";
|
||||
log?.info?.("PROXY", `${provider.toUpperCase()} | ${model} | conn=${connectionName} | pool=${poolId} | vercel-relay=${proxyOptions.vercelRelayUrl}`);
|
||||
} else if (proxyOptions.connectionProxyEnabled && proxyOptions.connectionProxyUrl) {
|
||||
const connectionName =
|
||||
credentials?.connectionName || credentials?.connectionId || "unknown";
|
||||
const poolId =
|
||||
credentials?.providerSpecificData?.connectionProxyPoolId || "none";
|
||||
log?.info?.(
|
||||
"PROXY",
|
||||
`${provider.toUpperCase()} | ${model} | conn=${connectionName} | pool=${poolId} | vercel-relay=${proxyOptions.vercelRelayUrl}`,
|
||||
);
|
||||
} else if (
|
||||
proxyOptions.connectionProxyEnabled &&
|
||||
proxyOptions.connectionProxyUrl
|
||||
) {
|
||||
let maskedProxyUrl = proxyOptions.connectionProxyUrl;
|
||||
try {
|
||||
const parsed = new URL(proxyOptions.connectionProxyUrl);
|
||||
@@ -328,14 +485,23 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
// Keep raw if URL parsing fails
|
||||
}
|
||||
|
||||
const poolId = credentials?.providerSpecificData?.connectionProxyPoolId || "none";
|
||||
const connectionName = credentials?.connectionName || credentials?.connectionId || "unknown";
|
||||
log?.info?.("PROXY", `${provider.toUpperCase()} | ${model} | conn=${connectionName} | pool=${poolId} | url=${maskedProxyUrl}`);
|
||||
const poolId =
|
||||
credentials?.providerSpecificData?.connectionProxyPoolId || "none";
|
||||
const connectionName =
|
||||
credentials?.connectionName || credentials?.connectionId || "unknown";
|
||||
log?.info?.(
|
||||
"PROXY",
|
||||
`${provider.toUpperCase()} | ${model} | conn=${connectionName} | pool=${poolId} | url=${maskedProxyUrl}`,
|
||||
);
|
||||
}
|
||||
|
||||
if (proxyOptions.connectionProxyEnabled && proxyOptions.connectionNoProxy) {
|
||||
const connectionName = credentials?.connectionName || credentials?.connectionId || "unknown";
|
||||
log?.debug?.("PROXY", `${provider.toUpperCase()} | ${model} | conn=${connectionName} | no_proxy=${proxyOptions.connectionNoProxy}`);
|
||||
const connectionName =
|
||||
credentials?.connectionName || credentials?.connectionId || "unknown";
|
||||
log?.debug?.(
|
||||
"PROXY",
|
||||
`${provider.toUpperCase()} | ${model} | conn=${connectionName} | no_proxy=${proxyOptions.connectionNoProxy}`,
|
||||
);
|
||||
}
|
||||
|
||||
// Execute request
|
||||
@@ -344,7 +510,15 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
// exception: it is decoded by the executor into OpenAI-compatible output.
|
||||
let providerResponseFormat = targetFormat;
|
||||
try {
|
||||
const result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
|
||||
const result = await executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
providerResponse = result.response;
|
||||
providerUrl = result.url;
|
||||
providerHeaders = result.headers;
|
||||
@@ -353,111 +527,254 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody);
|
||||
} catch (error) {
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
appendRequestLog({ model, provider, connectionId, status: `FAILED ${error.name === "AbortError" ? 499 : HTTP_STATUS.BAD_GATEWAY}` }).catch(() => { });
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
appendRequestLog({
|
||||
model,
|
||||
provider,
|
||||
connectionId,
|
||||
status: `FAILED ${error.name === "AbortError" ? 499 : HTTP_STATUS.BAD_GATEWAY}`,
|
||||
}).catch(() => {});
|
||||
saveRequestDetail(
|
||||
buildRequestDetail({
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: translatedBody || null,
|
||||
response: { error: error.message || String(error), status: error.name === "AbortError" ? 499 : 502, thinking: null },
|
||||
response: {
|
||||
error: error.message || String(error),
|
||||
status: error.name === "AbortError" ? 499 : 502,
|
||||
thinking: null,
|
||||
},
|
||||
pxpipe: pxpipeSummary,
|
||||
status: "error"
|
||||
})).catch(() => { });
|
||||
status: "error",
|
||||
}),
|
||||
).catch(() => {});
|
||||
|
||||
if (error.name === "AbortError") {
|
||||
streamController.handleError(error);
|
||||
return createErrorResult(499, "Request aborted");
|
||||
}
|
||||
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
|
||||
const errMsg = formatProviderError(
|
||||
error,
|
||||
provider,
|
||||
model,
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
);
|
||||
if (log?.errorLine) {
|
||||
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms\n ${errMsg}${error.stack ? `\n ${error.stack}` : ""}`);
|
||||
log.errorLine(
|
||||
reqTag,
|
||||
"✗",
|
||||
`ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms\n ${errMsg}${error.stack ? `\n ${error.stack}` : ""}`,
|
||||
);
|
||||
}
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, errMsg);
|
||||
}
|
||||
|
||||
// Config-driven in-stream error detection: peek the first bytes of a 200
|
||||
// stream; if a configured pattern matches, fail fast with 502 so account
|
||||
// /combo fallback can still run before any byte reaches the client.
|
||||
const configuredPatterns = streamErrorPatterns?.[provider];
|
||||
if (
|
||||
providerResponse?.ok &&
|
||||
Array.isArray(configuredPatterns) &&
|
||||
configuredPatterns.length > 0
|
||||
) {
|
||||
providerResponse = await maybeRejectEarlyStreamError(
|
||||
providerResponse,
|
||||
configuredPatterns,
|
||||
{ signal: streamController.signal },
|
||||
);
|
||||
}
|
||||
|
||||
// Handle 401/403 - try token refresh (skip for noAuth providers)
|
||||
if (!executor.noAuth && (providerResponse.status === HTTP_STATUS.UNAUTHORIZED || providerResponse.status === HTTP_STATUS.FORBIDDEN)) {
|
||||
if (
|
||||
!executor.noAuth &&
|
||||
(providerResponse.status === HTTP_STATUS.UNAUTHORIZED ||
|
||||
providerResponse.status === HTTP_STATUS.FORBIDDEN)
|
||||
) {
|
||||
try {
|
||||
// Mutate credentials after each successful refresh: rotating refresh_token
|
||||
// providers (xAI/grok-cli) issue a new RT on every refresh; without this,
|
||||
// refreshWithRetry's 2nd/3rd attempt reuses the already-consumed RT →
|
||||
// invalid_grant → auth_failed retryable=false.
|
||||
const newCredentials = await refreshWithRetry(async () => {
|
||||
const newCredentials = await refreshWithRetry(
|
||||
async () => {
|
||||
const result = await executor.refreshCredentials(credentials, log);
|
||||
if (result?.refreshToken && result.refreshToken !== credentials.refreshToken) {
|
||||
if (result.accessToken) credentials.accessToken = result.accessToken;
|
||||
if (
|
||||
result?.refreshToken &&
|
||||
result.refreshToken !== credentials.refreshToken
|
||||
) {
|
||||
if (result.accessToken)
|
||||
credentials.accessToken = result.accessToken;
|
||||
credentials.refreshToken = result.refreshToken;
|
||||
}
|
||||
return result;
|
||||
}, 3, log);
|
||||
},
|
||||
3,
|
||||
log,
|
||||
);
|
||||
if (newCredentials?.accessToken || newCredentials?.copilotToken) {
|
||||
if (log?.line) log.line(reqTag, "🔑", `TOKEN REFRESHED · ${provider}/${model}`);
|
||||
if (log?.line)
|
||||
log.line(reqTag, "🔑", `TOKEN REFRESHED · ${provider}/${model}`);
|
||||
Object.assign(credentials, newCredentials);
|
||||
if (onCredentialsRefreshed) {
|
||||
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
|
||||
try {
|
||||
await onCredentialsRefreshed(newCredentials);
|
||||
} catch (e) {
|
||||
log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`);
|
||||
}
|
||||
}
|
||||
try {
|
||||
const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
|
||||
const retryResult = await executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
if (retryResult.response.ok) {
|
||||
providerResponse = retryResult.response;
|
||||
providerUrl = retryResult.url;
|
||||
providerResponseFormat = retryResult.responseFormat || targetFormat;
|
||||
}
|
||||
} catch { log?.warn?.("TOKEN", `${provider.toUpperCase()} | retry after refresh failed`); }
|
||||
} catch {
|
||||
log?.warn?.(
|
||||
"TOKEN",
|
||||
`${provider.toUpperCase()} | retry after refresh failed`,
|
||||
);
|
||||
}
|
||||
} else {
|
||||
log?.warn?.("TOKEN", `${provider.toUpperCase()} | refresh failed`);
|
||||
}
|
||||
} catch (e) {
|
||||
log?.warn?.("TOKEN", `${provider.toUpperCase()} | refresh threw: ${e.message}`);
|
||||
log?.warn?.(
|
||||
"TOKEN",
|
||||
`${provider.toUpperCase()} | refresh threw: ${e.message}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Provider returned error
|
||||
if (!providerResponse.ok) {
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
const { statusCode, message, resetsAtMs } = await parseUpstreamError(providerResponse, executor);
|
||||
appendRequestLog({ model, provider, connectionId, status: `FAILED ${statusCode}` }).catch(() => { });
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
const { statusCode, message, resetsAtMs } = await parseUpstreamError(
|
||||
providerResponse,
|
||||
executor,
|
||||
);
|
||||
appendRequestLog({
|
||||
model,
|
||||
provider,
|
||||
connectionId,
|
||||
status: `FAILED ${statusCode}`,
|
||||
}).catch(() => {});
|
||||
saveRequestDetail(
|
||||
buildRequestDetail({
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
response: { error: message, status: statusCode, thinking: null },
|
||||
pxpipe: pxpipeSummary,
|
||||
status: "error"
|
||||
})).catch(() => { });
|
||||
status: "error",
|
||||
}),
|
||||
).catch(() => {});
|
||||
|
||||
const errMsg = formatProviderError(new Error(message), provider, model, statusCode);
|
||||
const errMsg = formatProviderError(
|
||||
new Error(message),
|
||||
provider,
|
||||
model,
|
||||
statusCode,
|
||||
);
|
||||
if (log?.errorLine) {
|
||||
const urlStr = providerUrl ? `\n URL: ${providerUrl}` : "";
|
||||
log.errorLine(reqTag, "✗", `ERROR ${statusCode} · ${provider}/${model} · ${Date.now() - requestStartTime}ms${urlStr}\n ${errMsg}`);
|
||||
log.errorLine(
|
||||
reqTag,
|
||||
"✗",
|
||||
`ERROR ${statusCode} · ${provider}/${model} · ${Date.now() - requestStartTime}ms${urlStr}\n ${errMsg}`,
|
||||
);
|
||||
}
|
||||
reqLogger.logError(new Error(message), finalBody || translatedBody);
|
||||
return createErrorResult(statusCode, errMsg, resetsAtMs);
|
||||
}
|
||||
|
||||
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log };
|
||||
const appendLog = (extra) => appendRequestLog({ model, provider, connectionId, ...extra }).catch(() => { });
|
||||
const trackDone = () => trackPendingRequest(model, provider, connectionId, false);
|
||||
const sharedCtx = {
|
||||
provider,
|
||||
model,
|
||||
body,
|
||||
stream,
|
||||
translatedBody,
|
||||
finalBody,
|
||||
requestStartTime,
|
||||
connectionId,
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
pxpipe: pxpipeSummary,
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
};
|
||||
const appendLog = (extra) =>
|
||||
appendRequestLog({ model, provider, connectionId, ...extra }).catch(
|
||||
() => {},
|
||||
);
|
||||
const trackDone = () =>
|
||||
trackPendingRequest(model, provider, connectionId, false);
|
||||
|
||||
// Provider forced streaming but client wants JSON
|
||||
if (!clientRequestedStreaming && providerRequiresStreaming) {
|
||||
const result = await handleForcedSSEToJson({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, customToolNames, trackDone, appendLog });
|
||||
if (result) { streamController.handleComplete(); return result; }
|
||||
const result = await handleForcedSSEToJson({
|
||||
...sharedCtx,
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
trackDone,
|
||||
appendLog,
|
||||
});
|
||||
if (result) {
|
||||
streamController.handleComplete();
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
// True non-streaming response
|
||||
if (!stream) {
|
||||
const result = await handleNonStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, reqLogger, toolNameMap, customToolNames, trackDone, appendLog });
|
||||
const result = await handleNonStreamingResponse({
|
||||
...sharedCtx,
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
targetFormat: providerResponseFormat,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
trackDone,
|
||||
appendLog,
|
||||
});
|
||||
streamController.handleComplete();
|
||||
return result;
|
||||
}
|
||||
|
||||
// Streaming response
|
||||
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId });
|
||||
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({
|
||||
...sharedCtx,
|
||||
});
|
||||
return handleStreamingResponse({
|
||||
...sharedCtx,
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
targetFormat: providerResponseFormat,
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
streamController,
|
||||
onStreamComplete,
|
||||
streamDetailId,
|
||||
});
|
||||
}
|
||||
|
||||
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {
|
||||
|
||||
@@ -9,7 +9,6 @@ import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
function parseToolArguments(value) {
|
||||
if (!value) return {};
|
||||
@@ -61,93 +60,11 @@ function openAICompletionToClaudeMessage(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions non-streaming response body into the
|
||||
* OpenAI Responses API shape. Used when a Responses-format client (e.g. Codex)
|
||||
* is routed to a Chat Completions upstream and `stream:false` — the streaming
|
||||
* path already emits Responses events, but the JSON path returned a raw
|
||||
* `chat.completion` body, so tool_calls were invisible to Responses clients.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function openAICompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
// Reasoning → a reasoning item (summary text), mirroring the streaming path.
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
// Assistant text → a message item with output_text content.
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
// tool_calls → function_call/custom_tool_call items (Responses-native tool shape).
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
const status = choice.finish_reason === "tool_calls" ? "completed" : (choice.finish_reason === "stop" ? "completed" : (choice.finish_reason || "completed"));
|
||||
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status,
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate non-streaming response body from provider format → OpenAI format.
|
||||
*/
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames = null) {
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
|
||||
if (targetFormat === sourceFormat) return responseBody;
|
||||
// Provider responded in OpenAI Chat Completions shape but the client speaks
|
||||
// Responses API — convert so tool_calls/text surface as Responses `output`.
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return openAICompletionToResponses(responseBody, customToolNames);
|
||||
}
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
||||
return openAICompletionToClaudeMessage(responseBody);
|
||||
}
|
||||
@@ -281,7 +198,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
||||
/**
|
||||
* Handle non-streaming response from provider.
|
||||
*/
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
trackDone();
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
let responseBody;
|
||||
@@ -322,12 +239,9 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
: responseBody;
|
||||
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
||||
// Responses-format translation produces a `object:"response"` body with no
|
||||
// `choices`; skip the Chat-Completions-specific post-processing below for it.
|
||||
const isResponsesResponse = sourceFormat === FORMATS.OPENAI_RESPONSES && translatedResponse?.object === "response";
|
||||
|
||||
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
||||
if (translatedResponse?.choices?.[0]) {
|
||||
@@ -340,13 +254,13 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
|
||||
// Ensure OpenAI-required fields
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
}
|
||||
|
||||
// Strip Azure-specific fields
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
if (!isClaudeMessageResponse) {
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
@@ -360,7 +274,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse && translatedResponse?.choices) {
|
||||
if (!isClaudeMessageResponse && translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
|
||||
@@ -44,14 +44,13 @@ export function extractUsageFromResponse(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
// Gemini format. Antigravity / gemini-cli wrap the payload in { response: {...} }.
|
||||
const usageMetadata = responseBody.usageMetadata || responseBody.response?.usageMetadata;
|
||||
if (usageMetadata) {
|
||||
// Gemini format
|
||||
if (responseBody.usageMetadata) {
|
||||
return {
|
||||
prompt_tokens: usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: usageMetadata.thoughtsTokenCount || 0
|
||||
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -1,13 +1,19 @@
|
||||
import { convertResponsesStreamToJson } from "../../transformer/streamToJsonConverter.js";
|
||||
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
saveUsageStats,
|
||||
formatDoneLine,
|
||||
} from "./requestDetail.js";
|
||||
|
||||
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
|
||||
const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
const isResponsesProvider = (p) =>
|
||||
PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
import { saveRequestDetail, appendRequestLog } from "@/lib/usageDb.js";
|
||||
|
||||
function textFromResponsesMessageItem(item) {
|
||||
@@ -35,76 +41,6 @@ function pickAssistantMessageForChatCompletion(output) {
|
||||
return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions JSON body into the Responses API shape.
|
||||
* Inlined here (not imported from nonStreamingHandler.js) to avoid a circular
|
||||
* import. Mirrors openAICompletionToResponses in nonStreamingHandler.js.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function chatCompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse OpenAI-style SSE text into a single chat completion JSON.
|
||||
* Used when provider forces streaming but client wants non-streaming.
|
||||
@@ -122,7 +58,9 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
const chunk = JSON.parse(payload);
|
||||
if (chunk?.error) streamError = chunk.error;
|
||||
else chunks.push(chunk);
|
||||
} catch { /* ignore malformed lines */ }
|
||||
} catch {
|
||||
/* ignore malformed lines */
|
||||
}
|
||||
}
|
||||
|
||||
if (streamError) return { error: streamError };
|
||||
@@ -138,8 +76,13 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
for (const chunk of chunks) {
|
||||
const choice = chunk?.choices?.[0];
|
||||
const delta = choice?.delta || {};
|
||||
if (typeof delta.content === "string" && delta.content.length > 0) contentParts.push(delta.content);
|
||||
if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0) reasoningParts.push(delta.reasoning_content);
|
||||
if (typeof delta.content === "string" && delta.content.length > 0)
|
||||
contentParts.push(delta.content);
|
||||
if (
|
||||
typeof delta.reasoning_content === "string" &&
|
||||
delta.reasoning_content.length > 0
|
||||
)
|
||||
reasoningParts.push(delta.reasoning_content);
|
||||
if (choice?.finish_reason) finishReason = choice.finish_reason;
|
||||
if (chunk?.usage && typeof chunk.usage === "object") usage = chunk.usage;
|
||||
|
||||
@@ -148,20 +91,31 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
for (const tc of delta.tool_calls) {
|
||||
const idx = tc.index ?? 0;
|
||||
if (!toolCallMap.has(idx)) {
|
||||
toolCallMap.set(idx, { id: tc.id || "", type: "function", function: { name: "", arguments: "" } });
|
||||
toolCallMap.set(idx, {
|
||||
id: tc.id || "",
|
||||
type: "function",
|
||||
function: { name: "", arguments: "" },
|
||||
});
|
||||
}
|
||||
const existing = toolCallMap.get(idx);
|
||||
if (tc.id) existing.id = tc.id;
|
||||
if (tc.function?.name) existing.function.name += tc.function.name;
|
||||
if (tc.function?.arguments) existing.function.arguments += tc.function.arguments;
|
||||
if (tc.function?.arguments)
|
||||
existing.function.arguments += tc.function.arguments;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const message = { role: "assistant", content: contentParts.join("") || (toolCallMap.size > 0 ? null : "") };
|
||||
if (reasoningParts.length > 0) message.reasoning_content = reasoningParts.join("");
|
||||
const message = {
|
||||
role: "assistant",
|
||||
content: contentParts.join("") || (toolCallMap.size > 0 ? null : ""),
|
||||
};
|
||||
if (reasoningParts.length > 0)
|
||||
message.reasoning_content = reasoningParts.join("");
|
||||
if (toolCallMap.size > 0) {
|
||||
message.tool_calls = [...toolCallMap.entries()].sort((a, b) => a[0] - b[0]).map(([, tc]) => tc);
|
||||
message.tool_calls = [...toolCallMap.entries()]
|
||||
.sort((a, b) => a[0] - b[0])
|
||||
.map(([, tc]) => tc);
|
||||
}
|
||||
|
||||
const result = {
|
||||
@@ -169,7 +123,7 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
object: "chat.completion",
|
||||
created: first.created || Math.floor(Date.now() / 1000),
|
||||
model: first.model || fallbackModel || "unknown",
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }]
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
};
|
||||
if (usage) result.usage = usage;
|
||||
return result;
|
||||
@@ -179,112 +133,201 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
* Handle case: provider forced streaming but client wants JSON.
|
||||
* Supports both Codex/Responses API SSE and standard Chat Completions SSE.
|
||||
*/
|
||||
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, targetFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, customToolNames, trackDone, appendLog, reqTag, log }) {
|
||||
export async function handleForcedSSEToJson({
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
provider,
|
||||
model,
|
||||
body,
|
||||
stream,
|
||||
translatedBody,
|
||||
finalBody,
|
||||
requestStartTime,
|
||||
connectionId,
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
trackDone,
|
||||
appendLog,
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
}) {
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
const isSSE = contentType.includes("text/event-stream") || (contentType === "" && isResponsesProvider(provider));
|
||||
const isSSE =
|
||||
contentType.includes("text/event-stream") ||
|
||||
(contentType === "" && isResponsesProvider(provider));
|
||||
if (!isSSE) return null; // not handled here
|
||||
|
||||
trackDone();
|
||||
|
||||
const ctx = {
|
||||
provider, model, connectionId,
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
};
|
||||
|
||||
// Codex/Responses API SSE path
|
||||
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
|
||||
// not the client format: a Responses-API client behind a chat-native forced-streaming
|
||||
// provider still receives chat SSE chunks, which must go through the standard path.
|
||||
const isCodexResponsesApi = isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const isCodexResponsesApi =
|
||||
isResponsesProvider(provider) || sourceFormat === FORMATS.OPENAI_RESPONSES;
|
||||
if (isCodexResponsesApi) {
|
||||
try {
|
||||
const jsonResponse = await convertResponsesStreamToJson(providerResponse.body);
|
||||
const jsonResponse = await convertResponsesStreamToJson(
|
||||
providerResponse.body,
|
||||
);
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
|
||||
const usage = jsonResponse.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
saveUsageStats({
|
||||
provider,
|
||||
model,
|
||||
tokens: usage,
|
||||
connectionId,
|
||||
apiKey,
|
||||
endpoint: clientRawRequest?.endpoint,
|
||||
silent: true,
|
||||
});
|
||||
if (log?.line)
|
||||
log.line(
|
||||
reqTag,
|
||||
"📊",
|
||||
formatDoneLine({
|
||||
usage,
|
||||
latency: { total: Date.now() - requestStartTime },
|
||||
}),
|
||||
);
|
||||
|
||||
// Same cache-inclusive total for the recorded detail, so the DB and the
|
||||
// client-facing usage can never disagree.
|
||||
const inTokensForLog = (usage.input_tokens || 0)
|
||||
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0)
|
||||
+ (usage.cache_creation_input_tokens || 0);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(jsonResponse.output);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(
|
||||
jsonResponse.output,
|
||||
);
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: { prompt_tokens: inTokensForLog, completion_tokens: usage.output_tokens || 0 },
|
||||
response: { content: textContent, thinking: null, finish_reason: jsonResponse.status || "unknown" },
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {});
|
||||
tokens: {
|
||||
prompt_tokens: usage.input_tokens || 0,
|
||||
completion_tokens: usage.output_tokens || 0,
|
||||
},
|
||||
response: {
|
||||
content: textContent,
|
||||
thinking: null,
|
||||
finish_reason: jsonResponse.status || "unknown",
|
||||
},
|
||||
status: "success",
|
||||
},
|
||||
{ endpoint: clientRawRequest?.endpoint || null },
|
||||
),
|
||||
).catch(() => {});
|
||||
|
||||
// Client is Responses API → return as-is
|
||||
if (sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return { success: true, response: new Response(JSON.stringify(jsonResponse), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(jsonResponse), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
// Build client-format response.
|
||||
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
|
||||
// only input+output under-reports prompt_tokens — measured: 2012 reported
|
||||
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache
|
||||
// counters in, and keep them visible in prompt_tokens_details so a client can
|
||||
// tell a cache hit from a small prompt.
|
||||
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens || 0;
|
||||
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
|
||||
// Build client-format response
|
||||
const inTokens = usage.input_tokens || 0;
|
||||
const outTokens = usage.output_tokens || 0;
|
||||
const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
|
||||
? { prompt_tokens_details: {
|
||||
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
|
||||
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
|
||||
: {};
|
||||
let finalResp;
|
||||
|
||||
// Extract tool calls from Responses API output (function_call items)
|
||||
const funcCallItems = (jsonResponse.output || []).filter(item => item.type === "function_call");
|
||||
const funcCallItems = (jsonResponse.output || []).filter(
|
||||
(item) => item.type === "function_call",
|
||||
);
|
||||
const toolCalls = funcCallItems.map((item, idx) => ({
|
||||
id: item.call_id || `call_${item.name}_${Date.now()}_${idx}`,
|
||||
type: "function",
|
||||
function: {
|
||||
name: item.name,
|
||||
arguments: typeof item.arguments === "string" ? item.arguments : JSON.stringify(item.arguments || {})
|
||||
}
|
||||
arguments:
|
||||
typeof item.arguments === "string"
|
||||
? item.arguments
|
||||
: JSON.stringify(item.arguments || {}),
|
||||
},
|
||||
}));
|
||||
const hasToolCalls = toolCalls.length > 0;
|
||||
|
||||
if (sourceFormat === FORMATS.ANTIGRAVITY || sourceFormat === FORMATS.GEMINI || sourceFormat === FORMATS.GEMINI_CLI) {
|
||||
if (
|
||||
sourceFormat === FORMATS.ANTIGRAVITY ||
|
||||
sourceFormat === FORMATS.GEMINI ||
|
||||
sourceFormat === FORMATS.GEMINI_CLI
|
||||
) {
|
||||
finalResp = {
|
||||
response: {
|
||||
candidates: [{ content: { role: "model", parts: [{ text: textContent || "" }] }, finishReason: "STOP", index: 0 }],
|
||||
usageMetadata: { promptTokenCount: inTokens, candidatesTokenCount: outTokens, totalTokenCount: inTokens + outTokens },
|
||||
candidates: [
|
||||
{
|
||||
content: {
|
||||
role: "model",
|
||||
parts: [{ text: textContent || "" }],
|
||||
},
|
||||
finishReason: "STOP",
|
||||
index: 0,
|
||||
},
|
||||
],
|
||||
usageMetadata: {
|
||||
promptTokenCount: inTokens,
|
||||
candidatesTokenCount: outTokens,
|
||||
totalTokenCount: inTokens + outTokens,
|
||||
},
|
||||
modelVersion: model,
|
||||
responseId: jsonResponse.id || `resp_${Date.now()}`
|
||||
}
|
||||
responseId: jsonResponse.id || `resp_${Date.now()}`,
|
||||
},
|
||||
};
|
||||
} else {
|
||||
const message = { role: "assistant", content: textContent || (hasToolCalls ? null : "") };
|
||||
const message = {
|
||||
role: "assistant",
|
||||
content: textContent || (hasToolCalls ? null : ""),
|
||||
};
|
||||
if (hasToolCalls) message.tool_calls = toolCalls;
|
||||
const responseDone = jsonResponse.status === "completed" || jsonResponse.status === "done";
|
||||
const finishReason = hasToolCalls ? "tool_calls" : (responseDone ? "stop" : (jsonResponse.status || "stop"));
|
||||
const responseDone =
|
||||
jsonResponse.status === "completed" || jsonResponse.status === "done";
|
||||
const finishReason = hasToolCalls
|
||||
? "tool_calls"
|
||||
: responseDone
|
||||
? "stop"
|
||||
: jsonResponse.status || "stop";
|
||||
finalResp = {
|
||||
id: jsonResponse.id || `chatcmpl-${Date.now()}`,
|
||||
object: "chat.completion",
|
||||
created: jsonResponse.created_at || Math.floor(Date.now() / 1000),
|
||||
model: jsonResponse.model || model,
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
usage: { prompt_tokens: inTokens, completion_tokens: outTokens, total_tokens: inTokens + outTokens, ...cacheDetails }
|
||||
usage: {
|
||||
prompt_tokens: inTokens,
|
||||
completion_tokens: outTokens,
|
||||
total_tokens: inTokens + outTokens,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
return { success: true, response: new Response(JSON.stringify(finalResp), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(finalResp), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Responses API SSE→JSON failed:", err);
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON");
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
"Failed to convert streaming response to JSON",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -292,11 +335,29 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, ta
|
||||
try {
|
||||
const sseText = await providerResponse.text();
|
||||
const parsed = parseSSEToOpenAIResponse(sseText, model);
|
||||
if (!parsed) return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Invalid SSE response for non-streaming request");
|
||||
if (!parsed)
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
"Invalid SSE response for non-streaming request",
|
||||
);
|
||||
if (parsed.error) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
parsed.error.message || "Upstream SSE stream failed"
|
||||
parsed.error.message || "Upstream SSE stream failed",
|
||||
);
|
||||
}
|
||||
|
||||
// Config-driven in-stream error detection: the request "succeeded" at the
|
||||
// HTTP level, but the content signals an upstream failure — treat it as an
|
||||
// error so account/combo fallback and FAILED logging kick in.
|
||||
const matchedPattern = matchStreamErrorPatterns(
|
||||
streamErrorPatterns?.[provider],
|
||||
parsed.choices?.[0]?.message?.content || "",
|
||||
);
|
||||
if (matchedPattern) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
`Stream error pattern matched: ${matchedPattern}`,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -304,30 +365,42 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, ta
|
||||
|
||||
const usage = parsed.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
saveUsageStats({
|
||||
provider,
|
||||
model,
|
||||
tokens: usage,
|
||||
connectionId,
|
||||
apiKey,
|
||||
endpoint: clientRawRequest?.endpoint,
|
||||
silent: true,
|
||||
});
|
||||
if (log?.line)
|
||||
log.line(
|
||||
reqTag,
|
||||
"📊",
|
||||
formatDoneLine({
|
||||
usage,
|
||||
latency: { total: Date.now() - requestStartTime },
|
||||
}),
|
||||
);
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: usage,
|
||||
response: {
|
||||
content: parsed.choices?.[0]?.message?.content || null,
|
||||
thinking: parsed.choices?.[0]?.message?.reasoning_content || null,
|
||||
finish_reason: parsed.choices?.[0]?.finish_reason || "unknown"
|
||||
finish_reason: parsed.choices?.[0]?.finish_reason || "unknown",
|
||||
},
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {});
|
||||
|
||||
// Re-attach usage explicitly. This handler already HAS the correct usage — it is
|
||||
// the same object written to the usage DB, and for a cached Claude request that DB
|
||||
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
|
||||
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
|
||||
// on both sides). Whatever drops it between assembly and serialisation, the client
|
||||
// must not be left unable to account for its own token spend: a caller cannot tell
|
||||
// a 90%-cached request from a cheap one without this.
|
||||
if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
|
||||
status: "success",
|
||||
},
|
||||
{ endpoint: clientRawRequest?.endpoint || null },
|
||||
),
|
||||
).catch(() => {});
|
||||
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
@@ -341,19 +414,20 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, ta
|
||||
}
|
||||
}
|
||||
|
||||
// A Responses-format client (e.g. Codex) forced this provider to stream,
|
||||
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
|
||||
// body; convert it to the Responses `output` shape so tool_calls are not
|
||||
// lost on the non-streaming return path. Inlined (not imported from
|
||||
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
|
||||
// already imports parseSSEToOpenAIResponse from this module.
|
||||
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
|
||||
? chatCompletionToResponses(parsed, customToolNames)
|
||||
: parsed;
|
||||
|
||||
return { success: true, response: new Response(JSON.stringify(finalBody), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(parsed), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Chat Completions SSE→JSON failed:", err);
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON");
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
"Failed to convert streaming response to JSON",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,11 +1,20 @@
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { needsTranslation } from "../../translator/index.js";
|
||||
import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js";
|
||||
import {
|
||||
createSSETransformStreamWithLogger,
|
||||
createPassthroughStreamWithLogger,
|
||||
} from "../../utils/stream.js";
|
||||
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
saveUsageStats,
|
||||
formatDoneLine,
|
||||
} from "./requestDetail.js";
|
||||
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
|
||||
@@ -22,33 +31,108 @@ const CODEX_SOURCE_TO_TARGET = {
|
||||
/**
|
||||
* Determine which SSE transform stream to use based on provider/format.
|
||||
*/
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) {
|
||||
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
|
||||
function buildTransformStream({
|
||||
provider,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
}) {
|
||||
const isDroidCLI =
|
||||
userAgent?.toLowerCase().includes("droid") ||
|
||||
userAgent?.toLowerCase().includes("codex-cli");
|
||||
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
|
||||
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
const needsCodexTranslation = isResponsesProvider && targetFormat === FORMATS.OPENAI_RESPONSES && !isDroidCLI;
|
||||
const isResponsesProvider =
|
||||
PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
const needsCodexTranslation =
|
||||
isResponsesProvider &&
|
||||
targetFormat === FORMATS.OPENAI_RESPONSES &&
|
||||
!isDroidCLI;
|
||||
|
||||
if (needsCodexTranslation) {
|
||||
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
return createSSETransformStreamWithLogger(
|
||||
FORMATS.OPENAI_RESPONSES,
|
||||
codexTarget,
|
||||
provider,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
);
|
||||
}
|
||||
|
||||
if (needsTranslation(targetFormat, sourceFormat)) {
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
return createSSETransformStreamWithLogger(
|
||||
targetFormat,
|
||||
sourceFormat,
|
||||
provider,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
);
|
||||
}
|
||||
|
||||
return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey);
|
||||
return createPassthroughStreamWithLogger(
|
||||
provider,
|
||||
reqLogger,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) {
|
||||
export async function handleStreamingResponse({
|
||||
providerResponse,
|
||||
provider,
|
||||
model,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
userAgent,
|
||||
body,
|
||||
stream,
|
||||
translatedBody,
|
||||
finalBody,
|
||||
requestStartTime,
|
||||
connectionId,
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
streamController,
|
||||
onStreamComplete,
|
||||
streamDetailId,
|
||||
pxpipe,
|
||||
reqTag,
|
||||
log,
|
||||
}) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
.catch(err => {
|
||||
console.error("[ChatCore] onRequestSuccess failed:", err?.message || err);
|
||||
.catch((err) => {
|
||||
console.error(
|
||||
"[ChatCore] onRequestSuccess failed:",
|
||||
err?.message || err,
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -59,84 +143,192 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
// return a clean JSON error instead. The message is stripped of HTML tags
|
||||
// and clamped so untrusted upstream text never reaches the client verbatim
|
||||
// (the UI may render error.message as HTML).
|
||||
const upstreamContentType = (providerResponse.headers.get('content-type') || '').toLowerCase();
|
||||
if (upstreamContentType && !upstreamContentType.includes('text/event-stream') && !upstreamContentType.includes('application/json')) {
|
||||
const bodyText = await providerResponse.text().catch(() => '');
|
||||
const upstreamContentType = (
|
||||
providerResponse.headers.get("content-type") || ""
|
||||
).toLowerCase();
|
||||
if (
|
||||
upstreamContentType &&
|
||||
!upstreamContentType.includes("text/event-stream") &&
|
||||
!upstreamContentType.includes("application/json")
|
||||
) {
|
||||
const bodyText = await providerResponse.text().catch(() => "");
|
||||
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
|
||||
const sanitizedTitle = (titleMatch?.[1] || '').replace(/<[^>]*>/g, '').replace(/[\r\n]+/g, ' ').trim().slice(0, 160);
|
||||
const shortMsg = sanitizedTitle
|
||||
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const sanitizedTitle = (titleMatch?.[1] || "")
|
||||
.replace(/<[^>]*>/g, "")
|
||||
.replace(/[\r\n]+/g, " ")
|
||||
.trim()
|
||||
.slice(0, 160);
|
||||
const shortMsg =
|
||||
sanitizedTitle ||
|
||||
(bodyText.length < 200
|
||||
? bodyText
|
||||
.replace(/<[^>]*>/g, "")
|
||||
.trim()
|
||||
.slice(0, 160)
|
||||
: `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
if (log?.errorLine) log.errorLine(reqTag, "✗", `BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`);
|
||||
else console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
if (log?.errorLine)
|
||||
log.errorLine(
|
||||
reqTag,
|
||||
"✗",
|
||||
`BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`,
|
||||
);
|
||||
else
|
||||
console.warn(
|
||||
`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`,
|
||||
);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
response: new Response(JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }), {
|
||||
response: new Response(
|
||||
JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }),
|
||||
{
|
||||
status,
|
||||
headers: { 'Content-Type': 'application/json', 'Access-Control-Allow-Origin': '*' },
|
||||
}),
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
},
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey });
|
||||
const transformStream = buildTransformStream({
|
||||
provider,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
});
|
||||
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null;
|
||||
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
|
||||
const isResponsesPassthrough =
|
||||
sourceFormat === FORMATS.OPENAI_RESPONSES &&
|
||||
targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough
|
||||
? buildAbortedResponsesTerminalBytes
|
||||
: null;
|
||||
const stallTimeoutMs =
|
||||
PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(
|
||||
providerResponse,
|
||||
transformStream,
|
||||
streamController,
|
||||
onAbortTerminal,
|
||||
stallTimeoutMs,
|
||||
);
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: "[Streaming - raw response not captured]",
|
||||
response: { content: "[Streaming in progress...]", thinking: null, type: "streaming" },
|
||||
response: {
|
||||
content: "[Streaming in progress...]",
|
||||
thinking: null,
|
||||
type: "streaming",
|
||||
},
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save streaming request:", err.message);
|
||||
status: "success",
|
||||
},
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to save streaming request:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(transformedBody, { headers: SSE_HEADERS })
|
||||
response: new Response(transformedBody, { headers: SSE_HEADERS }),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Build onStreamComplete callback for streaming usage tracking.
|
||||
*/
|
||||
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest, pxpipe, reqTag, log }) {
|
||||
export function buildOnStreamComplete({
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
apiKey,
|
||||
requestStartTime,
|
||||
body,
|
||||
stream,
|
||||
finalBody,
|
||||
translatedBody,
|
||||
clientRawRequest,
|
||||
pxpipe,
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
}) {
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
|
||||
const onStreamComplete = (contentObj, usage, ttftAt) => {
|
||||
const latency = {
|
||||
ttft: ttftAt ? ttftAt - requestStartTime : Date.now() - requestStartTime,
|
||||
total: Date.now() - requestStartTime
|
||||
total: Date.now() - requestStartTime,
|
||||
};
|
||||
const safeContent = contentObj?.content || "[Empty streaming response]";
|
||||
const safeThinking = contentObj?.thinking || null;
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
latency,
|
||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: safeContent,
|
||||
response: { content: safeContent, thinking: safeThinking, type: "streaming" },
|
||||
response: {
|
||||
content: safeContent,
|
||||
thinking: safeThinking,
|
||||
type: "streaming",
|
||||
},
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to update streaming content:", err.message);
|
||||
status: streamStatusForContent(
|
||||
streamErrorPatterns?.[provider],
|
||||
safeContent,
|
||||
),
|
||||
},
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to update streaming content:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
|
||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, label: "STREAM USAGE", silent: true });
|
||||
saveUsageStats({
|
||||
provider,
|
||||
model,
|
||||
tokens: usage,
|
||||
connectionId,
|
||||
apiKey,
|
||||
endpoint: clientRawRequest?.endpoint,
|
||||
label: "STREAM USAGE",
|
||||
silent: true,
|
||||
});
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency }));
|
||||
};
|
||||
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
import createOpenAIEmbeddingAdapter from "./openai.js";
|
||||
import gemini from "./gemini.js";
|
||||
import openaiCompatNode from "./openaiCompatNode.js";
|
||||
import selfhostedEmbedding from "./selfhostedEmbedding.js";
|
||||
|
||||
const OPENAI_COMPAT_PROVIDERS = [
|
||||
"openai", "openrouter", "mistral", "voyage-ai", "fireworks",
|
||||
@@ -14,12 +13,6 @@ const ADAPTERS = {
|
||||
...Object.fromEntries(OPENAI_COMPAT_PROVIDERS.map((id) => [id, createOpenAIEmbeddingAdapter(id)])),
|
||||
gemini,
|
||||
google_ai_studio: gemini,
|
||||
// Self-hosted reads creds.providerSpecificData.baseUrl (one provider, many
|
||||
// servers) — but via its OWN adapter, not openaiCompatNode: that one falls back
|
||||
// to api.openai.com when no baseUrl is set, which under a provider called
|
||||
// "Self-hosted Embedding" means silently shipping the input and API key to
|
||||
// OpenAI. selfhostedEmbedding refuses instead.
|
||||
"selfhosted-embedding": selfhostedEmbedding,
|
||||
};
|
||||
|
||||
export function getEmbeddingAdapter(provider) {
|
||||
|
||||
@@ -1,46 +0,0 @@
|
||||
// Self-hosted embeddings — like openaiCompatNode, but the baseUrl is REQUIRED.
|
||||
//
|
||||
// openaiCompatNode falls back to https://api.openai.com/v1 when a connection
|
||||
// carries no providerSpecificData.baseUrl. For a custom NODE that default is
|
||||
// defensible: the node was created by pointing at some OpenAI-compatible URL, and
|
||||
// OpenAI is the archetype. For a provider whose entire purpose is "my own
|
||||
// server", it is actively harmful — a connection saved without a baseUrl sends
|
||||
// the INPUT TEXT and the API KEY to OpenAI, silently, under a provider named
|
||||
// "Self-hosted Embedding".
|
||||
//
|
||||
// Observed exactly that with a placeholder connection (2026-08-04):
|
||||
//
|
||||
// [selfhosted-embedding/embedding] [401]: Incorrect API key provided: abc.
|
||||
// You can find your API key at https://platform.openai.com/account/api-keys.
|
||||
//
|
||||
// The key "abc" was typed as a throwaway for a LOCAL server and left the network.
|
||||
// A self-hosted provider must never have a cloud fallback, so this one refuses
|
||||
// instead: no baseUrl means a configuration error, reported as such.
|
||||
import createOpenAIEmbeddingAdapter from "./openai.js";
|
||||
|
||||
const baseAdapter = createOpenAIEmbeddingAdapter("openai");
|
||||
|
||||
export class MissingBaseUrlError extends Error {
|
||||
constructor() {
|
||||
super(
|
||||
"Self-hosted Embedding needs an endpoint: set this connection's baseUrl to " +
|
||||
"the OpenAI base URL of your server, e.g. http://host:8080/v1 (note the /v1 — " +
|
||||
"\"/embeddings\" is appended to it). Refusing to fall back to api.openai.com, " +
|
||||
"which would send your input and API key to OpenAI."
|
||||
);
|
||||
this.name = "MissingBaseUrlError";
|
||||
this.isConfigError = true;
|
||||
}
|
||||
}
|
||||
|
||||
export default {
|
||||
...baseAdapter,
|
||||
buildUrl: (_model, creds) => {
|
||||
const rawBaseUrl = creds?.providerSpecificData?.baseUrl;
|
||||
if (!rawBaseUrl || !String(rawBaseUrl).trim()) throw new MissingBaseUrlError();
|
||||
// Accept either the OpenAI base or a full embeddings URL, so a value pasted
|
||||
// from a curl example works as well as one typed from the help text.
|
||||
const baseUrl = String(rawBaseUrl).trim().replace(/\/$/, "").replace(/\/embeddings$/, "");
|
||||
return `${baseUrl}/embeddings`;
|
||||
},
|
||||
};
|
||||
@@ -1,5 +1,5 @@
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { HTTP_STATUS, FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { getExecutor } from "../executors/index.js";
|
||||
import { refreshWithRetry } from "../services/tokenRefresh.js";
|
||||
import { getEmbeddingAdapter } from "./embeddingProviders/index.js";
|
||||
@@ -38,24 +38,13 @@ export async function handleEmbeddingsCore({
|
||||
}
|
||||
|
||||
const ctx = { input };
|
||||
// buildUrl/buildHeaders/buildBody were called bare. An adapter that rejects a
|
||||
// misconfigured connection — selfhosted-embedding throws when no baseUrl is set
|
||||
// rather than silently falling back to api.openai.com — would have escaped this
|
||||
// function uncaught, surfacing as a 500 or a request that never settles. A
|
||||
// configuration mistake is a 400 with the reason in it.
|
||||
let url, headers, requestBody;
|
||||
try {
|
||||
url = adapter.buildUrl(model, credentials, ctx);
|
||||
headers = adapter.buildHeaders(credentials, ctx);
|
||||
requestBody = adapter.buildBody(model, {
|
||||
const url = adapter.buildUrl(model, credentials, ctx);
|
||||
const headers = adapter.buildHeaders(credentials, ctx);
|
||||
const requestBody = adapter.buildBody(model, {
|
||||
input,
|
||||
encoding_format: body.encoding_format || "float",
|
||||
dimensions: body.dimensions,
|
||||
});
|
||||
} catch (error) {
|
||||
log?.debug?.("EMBEDDINGS", `Request build failed: ${error.message}`);
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}/${model}] ${error.message}`);
|
||||
}
|
||||
|
||||
log?.debug?.("EMBEDDINGS", `${provider.toUpperCase()} | ${model} | input_type=${Array.isArray(input) ? `array[${input.length}]` : "string"}`);
|
||||
|
||||
@@ -65,9 +54,6 @@ export async function handleEmbeddingsCore({
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify(requestBody),
|
||||
...(typeof AbortSignal?.timeout === "function"
|
||||
? { signal: AbortSignal.timeout(FETCH_CONNECT_TIMEOUT_MS) }
|
||||
: {}),
|
||||
});
|
||||
} catch (error) {
|
||||
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
|
||||
|
||||
@@ -96,7 +96,7 @@ export async function handleImageGenerationCore({
|
||||
let requestBody;
|
||||
|
||||
try {
|
||||
url = adapter.buildUrl(model, credentials);
|
||||
url = adapter.buildUrl(model, credentials, body);
|
||||
requestBody = await adapter.buildBody(model, body);
|
||||
headers = adapter.buildHeaders(credentials, requestBody, model, body);
|
||||
} catch (error) {
|
||||
@@ -140,7 +140,7 @@ export async function handleImageGenerationCore({
|
||||
try {
|
||||
const retryBody = await adapter.buildBody(model, body);
|
||||
const retryHeaders = adapter.buildHeaders(credentials, retryBody, model, body);
|
||||
const retryUrl = adapter.buildUrl(model, credentials);
|
||||
const retryUrl = adapter.buildUrl(model, credentials, body);
|
||||
providerResponse = await fetch(retryUrl, {
|
||||
method: "POST",
|
||||
headers: retryHeaders,
|
||||
|
||||
@@ -12,6 +12,7 @@ import blackForestLabs from "./blackForestLabs.js";
|
||||
import runwayml from "./runwayml.js";
|
||||
import cloudflareAi from "./cloudflareAi.js";
|
||||
import antigravity from "./antigravity.js";
|
||||
import xai from "./xai.js";
|
||||
|
||||
const ADAPTERS = {
|
||||
openai: createOpenAIAdapter("openai"),
|
||||
@@ -19,7 +20,7 @@ const ADAPTERS = {
|
||||
openrouter: createOpenAIAdapter("openrouter"),
|
||||
recraft: createOpenAIAdapter("recraft"),
|
||||
"vercel-ai-gateway": createOpenAIAdapter("vercel-ai-gateway"),
|
||||
xai: createOpenAIAdapter("xai"),
|
||||
xai,
|
||||
gemini,
|
||||
codex,
|
||||
sdwebui,
|
||||
|
||||
137
open-sse/handlers/imageProviders/xai.js
Normal file
137
open-sse/handlers/imageProviders/xai.js
Normal file
@@ -0,0 +1,137 @@
|
||||
// xAI Grok Imagine — text-to-image + single/multi image editing
|
||||
// Docs:
|
||||
// https://docs.x.ai/developers/model-capabilities/images/generation
|
||||
// https://docs.x.ai/developers/model-capabilities/images/editing
|
||||
// https://docs.x.ai/developers/model-capabilities/images/multi-image-editing
|
||||
import { sizeToAspectRatio } from "./_base.js";
|
||||
import { PROVIDER_MEDIA } from "../../providers/index.js";
|
||||
|
||||
const IMG_CFG = PROVIDER_MEDIA["xai"]?.imageConfig || {};
|
||||
const GENERATIONS_URL = IMG_CFG.baseUrl || "https://api.x.ai/v1/images/generations";
|
||||
const EDITS_URL = IMG_CFG.editsUrl || "https://api.x.ai/v1/images/edits";
|
||||
|
||||
const ASPECT_RATIOS = new Set([
|
||||
"auto",
|
||||
"1:1",
|
||||
"16:9",
|
||||
"9:16",
|
||||
"4:3",
|
||||
"3:2",
|
||||
"2:3",
|
||||
"9:19.5",
|
||||
"20:9",
|
||||
]);
|
||||
|
||||
function hasEditInput(body) {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (body.image) return true;
|
||||
return Array.isArray(body.images) && body.images.some(Boolean);
|
||||
}
|
||||
|
||||
/** Normalize client image input → xAI image ref object */
|
||||
function toXaiImageRef(input) {
|
||||
if (!input) return null;
|
||||
|
||||
if (typeof input === "object") {
|
||||
// Already xAI-shaped or partial
|
||||
if (input.file_id) {
|
||||
return {
|
||||
type: input.type || "image_url",
|
||||
file_id: input.file_id,
|
||||
...(input.url ? { url: input.url } : {}),
|
||||
};
|
||||
}
|
||||
if (input.url) {
|
||||
return { type: input.type || "image_url", url: input.url };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
if (typeof input !== "string") return null;
|
||||
const trimmed = input.trim();
|
||||
if (!trimmed) return null;
|
||||
|
||||
// Public URL or data URI
|
||||
if (/^https?:\/\//i.test(trimmed) || /^data:image\//i.test(trimmed)) {
|
||||
return { type: "image_url", url: trimmed };
|
||||
}
|
||||
|
||||
// Raw base64 → data URI
|
||||
return { type: "image_url", url: `data:image/png;base64,${trimmed}` };
|
||||
}
|
||||
|
||||
function collectImageRefs(body) {
|
||||
const refs = [];
|
||||
if (Array.isArray(body.images)) {
|
||||
for (const item of body.images) {
|
||||
const ref = toXaiImageRef(item);
|
||||
if (ref) refs.push(ref);
|
||||
}
|
||||
}
|
||||
if (body.image) {
|
||||
const ref = toXaiImageRef(body.image);
|
||||
if (ref) refs.push(ref);
|
||||
}
|
||||
// xAI multi-edit supports up to 3 source images
|
||||
return refs.slice(0, 3);
|
||||
}
|
||||
|
||||
function resolveAspectRatio(body) {
|
||||
if (typeof body.aspect_ratio === "string" && body.aspect_ratio.trim()) {
|
||||
const ratio = body.aspect_ratio.trim();
|
||||
if (ASPECT_RATIOS.has(ratio)) return ratio;
|
||||
// Pass through unknown ratio strings (upstream will validate)
|
||||
return ratio;
|
||||
}
|
||||
// OpenAI-style size → aspect ratio (skip auto)
|
||||
if (body.size && body.size !== "auto") {
|
||||
return sizeToAspectRatio(body.size);
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveResolution(body) {
|
||||
if (typeof body.resolution !== "string") return undefined;
|
||||
const value = body.resolution.trim().toLowerCase();
|
||||
if (!value || value === "auto") return undefined;
|
||||
return value; // "1k" | "2k"
|
||||
}
|
||||
|
||||
export default {
|
||||
buildUrl: (_model, _credentials, body) => (hasEditInput(body) ? EDITS_URL : GENERATIONS_URL),
|
||||
|
||||
buildHeaders: (creds) => {
|
||||
const headers = { "Content-Type": "application/json", ...(IMG_CFG.headers || {}) };
|
||||
const key = creds?.apiKey || creds?.accessToken;
|
||||
if (key) headers["Authorization"] = `Bearer ${key}`;
|
||||
return headers;
|
||||
},
|
||||
|
||||
buildBody: (model, body) => {
|
||||
const req = {
|
||||
model,
|
||||
prompt: body.prompt,
|
||||
};
|
||||
|
||||
if (body.n != null) req.n = body.n;
|
||||
if (body.response_format) req.response_format = body.response_format;
|
||||
|
||||
const aspectRatio = resolveAspectRatio(body);
|
||||
if (aspectRatio) req.aspect_ratio = aspectRatio;
|
||||
|
||||
const resolution = resolveResolution(body);
|
||||
if (resolution) req.resolution = resolution;
|
||||
|
||||
const refs = collectImageRefs(body);
|
||||
if (refs.length === 1) {
|
||||
req.image = refs[0];
|
||||
} else if (refs.length > 1) {
|
||||
req.images = refs;
|
||||
}
|
||||
|
||||
return req;
|
||||
},
|
||||
|
||||
// xAI already returns OpenAI-compatible { created, data: [{ url | b64_json }] }
|
||||
normalize: (responseBody) => responseBody,
|
||||
};
|
||||
@@ -29,8 +29,6 @@
|
||||
* @property {Record<string,unknown>} [providerSpecificData]
|
||||
*/
|
||||
|
||||
import { assertPublicUrl } from "../../../src/shared/utils/ssrfGuard.js";
|
||||
|
||||
// ── Helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
@@ -65,31 +63,12 @@ export function getProviderSetting(params, key) {
|
||||
|
||||
/**
|
||||
* Resolve base URL with optional override from providerOptions.baseUrl.
|
||||
*
|
||||
* The override is client-controlled and therefore SSRF-hardened: only public
|
||||
* http(s) URLs are accepted (internal/private/loopback/metadata addresses are
|
||||
* rejected via assertPublicUrl). The provider's own configured baseUrl is
|
||||
* trusted as-is (admin-controlled).
|
||||
*
|
||||
* @param {SearchProviderConfig} config
|
||||
* @param {SearchRequestParams} params
|
||||
* @returns {string}
|
||||
*/
|
||||
export function resolveBaseUrl(config, params) {
|
||||
const override = getProviderSetting(params, "baseUrl");
|
||||
if (override) {
|
||||
// SSRF guard: client-supplied base URLs must be public http(s) only.
|
||||
let parsed;
|
||||
try {
|
||||
parsed = new URL(override);
|
||||
} catch {
|
||||
throw new Error(`Invalid baseUrl: ${override}`);
|
||||
}
|
||||
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") {
|
||||
throw new Error(`Invalid baseUrl protocol: ${parsed.protocol}`);
|
||||
}
|
||||
assertPublicUrl(override);
|
||||
}
|
||||
return (override || config.baseUrl).replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
|
||||
@@ -170,17 +170,9 @@ export async function handleSttCore({ provider, model, formData, credentials, st
|
||||
const file = formData.get("file");
|
||||
if (!file) return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: file");
|
||||
|
||||
let cfg = sttConfig;
|
||||
const cfg = sttConfig;
|
||||
if (!cfg) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support STT`);
|
||||
|
||||
// Per-connection endpoint override. Registry entries carry a fixed baseUrl,
|
||||
// which is right for a named cloud service but useless for a self-hosted one
|
||||
// whose address only the operator knows. Opt-in: absent unless the connection
|
||||
// sets it, so cloud providers are untouched. Mirrors the custom embedding
|
||||
// providers, which already resolve baseUrl the same way.
|
||||
const overrideUrl = credentials?.providerSpecificData?.baseUrl;
|
||||
if (overrideUrl) cfg = { ...cfg, baseUrl: String(overrideUrl).replace(/\/+$/, "") };
|
||||
|
||||
const token = cfg.authType === "none" ? null : (credentials?.apiKey || credentials?.accessToken);
|
||||
if (cfg.authType !== "none" && !token) {
|
||||
return createErrorResult(HTTP_STATUS.UNAUTHORIZED, `No credentials for STT provider: ${provider}`);
|
||||
|
||||
@@ -48,16 +48,16 @@ function createTtsResponse(base64Audio, format, responseFormat) {
|
||||
*
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language, style }) {
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language }) {
|
||||
if (!input?.trim()) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: input");
|
||||
}
|
||||
|
||||
try {
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini, xiaomi-mimo)
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini)
|
||||
const adapter = getTtsAdapter(provider);
|
||||
if (adapter) {
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language, style });
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language });
|
||||
// Adapter may return a full {success, response} (legacy) or {base64, format}
|
||||
if (result.success !== undefined) return result;
|
||||
return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
|
||||
@@ -51,25 +51,6 @@ async function huggingface({ baseUrl, apiKey, text, modelId }) {
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Fish Audio: model travels in an HTTP header, the voice is a reference_id, returns binary
|
||||
async function fishAudio({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
"model": modelId || "s2.1-pro-free",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
format: "mp3",
|
||||
...(voiceId ? { reference_id: voiceId } : {}),
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Inworld: Basic auth, JSON { audioContent }
|
||||
async function inworld({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
@@ -185,5 +166,4 @@ export const FORMAT_HANDLERS = {
|
||||
tortoise,
|
||||
openai: openaiCompat,
|
||||
"minimax-tts": minimaxTts,
|
||||
"fish-audio": fishAudio,
|
||||
};
|
||||
|
||||
@@ -6,8 +6,6 @@ import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
||||
import openai from "./openai.js";
|
||||
import openrouter from "./openrouter.js";
|
||||
import gemini, { fetchGeminiVoices } from "./gemini.js";
|
||||
import xiaomiMimo from "./xiaomi-mimo.js";
|
||||
import selfhostedTts from "./selfhostedTts.js";
|
||||
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
@@ -20,8 +18,6 @@ const SPECIAL_ADAPTERS = {
|
||||
openai,
|
||||
openrouter,
|
||||
gemini,
|
||||
"xiaomi-mimo": xiaomiMimo,
|
||||
"selfhosted-tts": selfhostedTts,
|
||||
};
|
||||
|
||||
export function getTtsAdapter(provider) {
|
||||
|
||||
@@ -1,69 +0,0 @@
|
||||
// Self-hosted OpenAI-compatible TTS — POST {baseUrl}/v1/audio/speech.
|
||||
//
|
||||
// A SPECIAL_ADAPTER rather than a genericFormats handler on purpose: the generic
|
||||
// dispatcher resolves baseUrl from the static registry entry
|
||||
// (`synthesizeViaConfig` reads `cfg.baseUrl`) and never looks at the connection,
|
||||
// which is exactly the limitation this provider exists to lift.
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
const DEFAULT_BASE_URL = "http://localhost:8880";
|
||||
const DEFAULT_MODEL = "kokoro";
|
||||
const DEFAULT_VOICE = "af_heart";
|
||||
|
||||
export default {
|
||||
async synthesize(text, model, credentials, responseFormat = "mp3") {
|
||||
// Accept either providerSpecificData.baseUrl (how the custom embedding and
|
||||
// STT providers carry it) or a bare credentials.baseUrl (how the OpenAI TTS
|
||||
// adapter does), so a connection configured either way works.
|
||||
const raw = credentials?.providerSpecificData?.baseUrl || credentials?.baseUrl || DEFAULT_BASE_URL;
|
||||
// Tolerate a baseUrl given as the full endpoint or with a trailing /v1 —
|
||||
// both are natural things to paste, and silently double-appending the path
|
||||
// would 404 with nothing pointing at the cause.
|
||||
const base = String(raw)
|
||||
.replace(/\/+$/, "")
|
||||
.replace(/\/v1\/audio\/speech$/, "")
|
||||
.replace(/\/v1$/, "");
|
||||
|
||||
// The provider prefix is already stripped by getModelInfo, so `model` here is
|
||||
// "kokoro" or "kokoro/af_heart" — NOT "selfhosted-tts/...".
|
||||
//
|
||||
// A bare value is the MODEL, not the voice. The OpenAI adapter reads a bare
|
||||
// value as a voice, which is right for a service whose model is fixed
|
||||
// ("tts-1") and whose voice varies — but wrong here, where the model is the
|
||||
// variable part. Treating it as a voice sent voice="kokoro" upstream and
|
||||
// Kokoro answered 400, so `selfhosted-tts/kokoro` — the obvious way to
|
||||
// address this provider — was the one form that did not work (verified
|
||||
// against a live Kokoro through 9router, 2026-08-03).
|
||||
let ttsModel = DEFAULT_MODEL;
|
||||
let voice = DEFAULT_VOICE;
|
||||
if (model) {
|
||||
const parts = String(model).split("/").filter(Boolean);
|
||||
if (parts.length >= 2) {
|
||||
ttsModel = parts[0];
|
||||
voice = parts.slice(1).join("/");
|
||||
} else if (parts.length === 1) {
|
||||
ttsModel = parts[0];
|
||||
}
|
||||
}
|
||||
|
||||
const res = await fetch(`${base}/v1/audio/speech`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(credentials?.apiKey ? { Authorization: `Bearer ${credentials.apiKey}` } : {}),
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: ttsModel,
|
||||
voice,
|
||||
input: text,
|
||||
response_format: responseFormat,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.error?.message || `Self-hosted TTS failed: ${res.status}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: responseFormat };
|
||||
},
|
||||
};
|
||||
@@ -1,65 +0,0 @@
|
||||
// Xiaomi MiMo TTS — via OpenAI-compatible chat completions (non-streaming).
|
||||
// Docs: https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5
|
||||
// Message contract: target text in `role: assistant` content, style/voice
|
||||
// instructions in `role: user` content. Voice is selected via the top-level
|
||||
// `audio.voice` field (NOT embedded in the model name).
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
const DEFAULT_MODEL = "mimo-v2.5-tts";
|
||||
const DEFAULT_VOICE = "mimo_default";
|
||||
|
||||
export default {
|
||||
synthesize(text, model, credentials, responseFormat, { style, language } = {}) {
|
||||
if (!credentials?.apiKey) throw new Error("xiaomi-mimo API key required");
|
||||
return synthesizeMiMo(text, model, credentials.apiKey, style, language);
|
||||
},
|
||||
};
|
||||
|
||||
export async function synthesizeMiMo(text, model, apiKey, style, language) {
|
||||
const { modelId, voiceId } = parseModelVoice(model, DEFAULT_MODEL, DEFAULT_VOICE, [DEFAULT_MODEL]);
|
||||
|
||||
// Language and style are soft instructions → prepend as a role:user message.
|
||||
// MiMo auto-detects the spoken language of the text; the hint only nudges it
|
||||
// (e.g. "Speak in English.") and is independent of the chosen voice.
|
||||
const instructions = [];
|
||||
if (language) instructions.push(`Speak in ${language}.`);
|
||||
if (style) instructions.push(style);
|
||||
|
||||
const messages = [{ role: "assistant", content: text }];
|
||||
if (instructions.length) messages.unshift({ role: "user", content: instructions.join(" ") });
|
||||
|
||||
const res = await fetch("https://api.xiaomimimo.com/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
stream: false,
|
||||
messages,
|
||||
audio: {
|
||||
format: "wav",
|
||||
voice: voiceId || DEFAULT_VOICE,
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
const rawText = await res.text();
|
||||
let data = {};
|
||||
if (rawText) {
|
||||
try { data = JSON.parse(rawText); } catch { data = {}; }
|
||||
}
|
||||
|
||||
if (!res.ok) {
|
||||
throw new Error(data?.error?.message || rawText || `MiMo TTS error (${res.status})`);
|
||||
}
|
||||
|
||||
const audio = data?.choices?.[0]?.message?.audio?.data;
|
||||
if (!audio) throw new Error(data?.error?.message || "MiMo TTS returned no audio");
|
||||
|
||||
return {
|
||||
base64: audio,
|
||||
format: data?.choices?.[0]?.message?.audio?.format || "wav",
|
||||
};
|
||||
}
|
||||
@@ -47,6 +47,7 @@ export {
|
||||
refreshAccessToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshIflowToken,
|
||||
refreshGitHubToken,
|
||||
|
||||
@@ -205,7 +205,6 @@ export const PATTERN_CAPABILITIES = [
|
||||
|
||||
// ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─
|
||||
{ pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } },
|
||||
{ pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } },
|
||||
{ pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-2.5*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-budget", thinkingRange: { min: 0, max: 24576 }, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
@@ -280,7 +279,7 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*minimax*", caps: { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 131072 } },
|
||||
|
||||
// ── Xiaomi MiMo (vision, 1M / 262K ctx) ──────────────────────────
|
||||
{ pattern: "*mimo*v2.5*", caps: { vision: true, audioInput: true, videoInput: true, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*v2.5*", caps: { vision: true, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*omni*", caps: { vision: true, audioInput: true, contextWindow: 262144, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*", caps: { vision: true, contextWindow: 262144, maxOutput: 131072 } },
|
||||
|
||||
|
||||
@@ -38,11 +38,3 @@ export function modelStrip(model) {
|
||||
export function modelTargetFormat(model) {
|
||||
return model?.targetFormat || MODEL_DEFAULTS.targetFormat;
|
||||
}
|
||||
|
||||
// Per-model declared upstream formats (e.g. ["openai", "claude"]). Guards the
|
||||
// sourceFormat-matched transport for multi-endpoint providers whose models differ
|
||||
// in endpoint support (opencode-go: kimi/glm only do /chat/completions, minimax/qwen
|
||||
// also do /messages, deepseek also does /responses).
|
||||
export function modelSupportedFormats(model) {
|
||||
return model?.supportedFormats || null;
|
||||
}
|
||||
|
||||
@@ -57,10 +57,6 @@ export const MODEL_PRICING = {
|
||||
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
|
||||
// === Gemini ===
|
||||
"gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
@@ -145,122 +141,6 @@ export const PROVIDER_PRICING = {
|
||||
gh: {
|
||||
"gpt-5.3-codex": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
},
|
||||
// TokenRouter — exact rates from https://api.tokenrouter.com/api/pricing ($1/1M tokens).
|
||||
// Ratio→USD: input = model_ratio×2, output = model_ratio×completion_ratio×2.
|
||||
// These override the canonical MODEL_PRICING/PATTERN_PRICING, whose rates often
|
||||
// differ from TokenRouter's reseller pricing.
|
||||
tokenrouter: {
|
||||
"MiniMax-M3": { input: 0.3, output: 1.2, cached: 0.06, reasoning: 1.2 },
|
||||
"anthropic/claude-fable-5": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-haiku-4.5": { input: 1.0, output: 5.0, cached: 0.1, cache_creation: 1.25, reasoning: 5.0 },
|
||||
"anthropic/claude-opus-4.5": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.6": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.7": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.7-fast": { input: 30, output: 150, cached: 3.0, reasoning: 150 },
|
||||
"anthropic/claude-opus-4.8": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.8-fast": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-opus-5": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-5-fast": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-sonnet-4": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-4.5": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-4.6": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-5": { input: 2, output: 10, cached: 0.2, reasoning: 10 },
|
||||
"claude-opus-4-8-m-aws": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"deepseek/deepseek-v3.2": { input: 0.26, output: 0.38, cached: 0.13, reasoning: 0.38 },
|
||||
"deepseek/deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28 },
|
||||
"deepseek/deepseek-v4-flash-0731": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28 },
|
||||
"deepseek/deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87 },
|
||||
"ex/gpt-5.4": { input: 2.5, output: 15.0, cached: 0.25, reasoning: 15.0 },
|
||||
"google/gemini-2.5-flash-image": { input: 0.3, output: 2.5, reasoning: 2.5 },
|
||||
"google/gemini-3-flash-preview": { input: 0.5, output: 3.0, cached: 0.05, cache_creation: 0.08333, reasoning: 3.0 },
|
||||
"google/gemini-3-pro-image-preview": { input: 2, output: 12, reasoning: 12 },
|
||||
"google/gemini-3.1-flash-image-preview": { input: 0.5, output: 3.0, reasoning: 3.0 },
|
||||
"google/gemini-3.1-flash-lite-image": { input: 0.25, output: 1.5, reasoning: 1.5 },
|
||||
"google/gemini-3.1-pro-preview": { input: 2, output: 12, cached: 0.2, cache_creation: 0.375, reasoning: 12 },
|
||||
"google/gemini-3.5-flash": { input: 1.5, output: 9.0, cached: 0.15, cache_creation: 0.08333, reasoning: 9.0 },
|
||||
"google/gemini-3.5-flash-lite": { input: 0.3, output: 2.5, cached: 0.03, cache_creation: 0.08333, reasoning: 2.5 },
|
||||
"google/gemini-3.6-flash": { input: 1.5, output: 7.5, cached: 0.15, cache_creation: 0.08333, reasoning: 7.5 },
|
||||
"google/gemini-embedding-2": { input: 1.0, output: 6.0, cached: 0.1, reasoning: 6.0 },
|
||||
"google/gemma-4-26b-a4b-it": { input: 0.06, output: 0.33, reasoning: 0.33 },
|
||||
"kling-3.0-turbo": { input: 2.1, output: 2.1, reasoning: 2.1 },
|
||||
"microsoft/mai-image-2.5": { input: 5.0, output: 47.0, reasoning: 47.0 },
|
||||
"minimax/minimax-m2-her": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.1": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.1-highspeed": { input: 0.6, output: 2.4, cached: 0.06, reasoning: 2.4 },
|
||||
"minimax/minimax-m2.5": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.7": { input: 0.3, output: 1.2, cached: 0.06, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.7-highspeed": { input: 0.6, output: 2.4, cached: 0.06, reasoning: 2.4 },
|
||||
"miromind/mirothinker-1-7-deepresearch": { input: 4, output: 25.0, reasoning: 25.0 },
|
||||
"miromind/mirothinker-1-7-deepresearch-mini": { input: 1.25, output: 10.0, reasoning: 10.0 },
|
||||
"mistralai/devstral-2512": { input: 0.4, output: 2.0, cached: 0.04, reasoning: 2.0 },
|
||||
"mistralai/mistral-medium-3-5": { input: 1.5, output: 7.5, reasoning: 7.5 },
|
||||
"mistralai/mistral-small-2603": { input: 0.15, output: 0.6, cached: 0.015, reasoning: 0.6 },
|
||||
"mistralai/voxtral-small-24b-2507": { input: 0.1, output: 0.3, cached: 0.01, reasoning: 0.3 },
|
||||
"moonshotai/kimi-k2.5": { input: 0.6, output: 3.0, cached: 0.1, reasoning: 3.0 },
|
||||
"moonshotai/kimi-k2.6": { input: 0.95, output: 4.0, cached: 0.16, reasoning: 4.0 },
|
||||
"moonshotai/kimi-k2.7-code": { input: 0.9286, output: 3.8571, cached: 0.1857, reasoning: 3.8571 },
|
||||
"moonshotai/kimi-k3": { input: 3.0, output: 15.0, cached: 0.3, reasoning: 15.0 },
|
||||
"nvidia/nemotron-3-super-120b-a12b": { input: 0.3, output: 0.9, cached: 0.1, reasoning: 0.9 },
|
||||
"openai/gpt-4o-mini": { input: 0.15, output: 0.6, cached: 0.075, reasoning: 0.6 },
|
||||
"openai/gpt-5": { input: 1.25, output: 10.0, cached: 0.125, reasoning: 10.0 },
|
||||
"openai/gpt-5-image": { input: 10, output: 40, cached: 2.5, reasoning: 40 },
|
||||
"openai/gpt-5-image-mini": { input: 2.5, output: 8.0, cached: 0.25, reasoning: 8.0 },
|
||||
"openai/gpt-5-mini": { input: 0.25, output: 2.0, cached: 0.025, reasoning: 2.0 },
|
||||
"openai/gpt-5.2": { input: 1.75, output: 14.0, cached: 0.175, reasoning: 14.0 },
|
||||
"openai/gpt-5.3-codex": { input: 1.75, output: 14.0, cached: 0.175, reasoning: 14.0 },
|
||||
"openai/gpt-5.4": { input: 2.5, output: 15.0, cached: 0.25, reasoning: 15.0 },
|
||||
"openai/gpt-5.4-image-2": { input: 8, output: 30.0, cached: 2.0, reasoning: 30.0 },
|
||||
"openai/gpt-5.4-mini": { input: 0.75, output: 4.5, cached: 0.075, reasoning: 4.5 },
|
||||
"openai/gpt-5.4-nano": { input: 0.2, output: 1.25, cached: 0.02, reasoning: 1.25 },
|
||||
"openai/gpt-5.4-pro": { input: 30, output: 180, reasoning: 180 },
|
||||
"openai/gpt-5.5": { input: 5.0, output: 30.0, cached: 0.5, reasoning: 30.0 },
|
||||
"openai/gpt-5.5-pro": { input: 30, output: 180, reasoning: 180 },
|
||||
"openai/gpt-5.6-luna": { input: 0.2, output: 1.2, cached: 0.02, cache_creation: 0.25, reasoning: 1.2 },
|
||||
"openai/gpt-5.6-sol": { input: 5.0, output: 30.0, cached: 0.5, cache_creation: 6.25, reasoning: 30.0 },
|
||||
"openai/gpt-5.6-terra": { input: 2, output: 12, cached: 0.2, cache_creation: 2.5, reasoning: 12 },
|
||||
"openai/gpt-audio": { input: 2.5, output: 10.0, reasoning: 10.0 },
|
||||
"openai/gpt-audio-mini": { input: 0.6, output: 2.4, reasoning: 2.4 },
|
||||
"openai/gpt-oss-120b": { input: 0.039, output: 0.18, reasoning: 0.18 },
|
||||
"qwen/qwen3-coder-next": { input: 0.12, output: 0.75, cached: 0.06, reasoning: 0.75 },
|
||||
"qwen/qwen3.5-122b-a10b": { input: 0.26, output: 2.08, reasoning: 2.08 },
|
||||
"qwen/qwen3.5-35b-a3b": { input: 0.1625, output: 1.3, reasoning: 1.3 },
|
||||
"qwen/qwen3.5-397b-a17b": { input: 0.39, output: 2.34, reasoning: 2.34 },
|
||||
"qwen/qwen3.5-9b": { input: 0.1, output: 0.15, reasoning: 0.15 },
|
||||
"qwen/qwen3.5-flash": { input: 0.1048, output: 0.4194, reasoning: 0.4194 },
|
||||
"qwen/qwen3.5-plus-02-15": { input: 0.26, output: 1.56, reasoning: 1.56 },
|
||||
"qwen/qwen3.6-plus": { input: 0.54, output: 3.21, reasoning: 3.21 },
|
||||
"qwen/qwen3.7-max": { input: 1.25, output: 3.75, cached: 0.25, reasoning: 3.75 },
|
||||
"qwen/qwen3.7-plus": { input: 0.4, output: 1.6, cached: 0.08, reasoning: 1.6 },
|
||||
"qwen/qwen3.8-max": { input: 2, output: 6, cached: 0.25, cache_creation: 2.5, reasoning: 6 },
|
||||
"qwen3.5-omni-plus": { input: 1.0, output: 5.7143, reasoning: 5.7143 },
|
||||
"qwen3.6-flash": { input: 0.171, output: 1.029, cached: 0.017, cache_creation: 0.214, reasoning: 1.029 },
|
||||
"sakana/fugu-ultra": { input: 5.0, output: 30.0, cached: 0.5, reasoning: 30.0 },
|
||||
"seed-2-0-code-preview-260328": { input: 1.0, output: 6.0, cached: 0.2, cache_creation: 0.008333, reasoning: 6.0 },
|
||||
"seed-2-0-lite-260428": { input: 0.5, output: 4.0, cached: 0.1, cache_creation: 0.008333, reasoning: 4.0 },
|
||||
"seed-2-0-mini-260428": { input: 0.2, output: 0.8, cached: 0.04, cache_creation: 0.00833, reasoning: 0.8 },
|
||||
"seed-2-0-pro-260328": { input: 1.0, output: 6.0, cached: 0.2, cache_creation: 0.008333, reasoning: 6.0 },
|
||||
"stepfun/step-3.5-flash": { input: 0.1, output: 0.3, cached: 0.02, reasoning: 0.3 },
|
||||
"stepfun/step-3.7-flash": { input: 0.2, output: 1.15, cached: 0.04, reasoning: 1.15 },
|
||||
"tencent/hy3-preview": { input: 0.066, output: 0.26, cached: 0.029, reasoning: 0.26 },
|
||||
"x-ai/grok-4.1-fast": { input: 0.2, output: 0.5, cached: 0.05, reasoning: 0.5 },
|
||||
"x-ai/grok-4.20-beta": { input: 2, output: 6, cached: 0.2, reasoning: 6 },
|
||||
"x-ai/grok-4.3": { input: 1.25, output: 2.5, cached: 0.2, reasoning: 2.5 },
|
||||
"x-ai/grok-4.5": { input: 2, output: 6, cached: 0.5, reasoning: 6 },
|
||||
"x-ai/grok-build-0.1": { input: 1.0, output: 2.0, cached: 0.2, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2-flash": { input: 0.1, output: 0.3, cached: 0.01, reasoning: 0.3 },
|
||||
"xiaomi/mimo-v2-omni": { input: 0.4, output: 2.0, cached: 0.08, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2-pro": { input: 1.0, output: 3.0, cached: 0.2, reasoning: 3.0 },
|
||||
"xiaomi/mimo-v2.5": { input: 0.4, output: 2.0, cached: 0.08, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2.5-pro": { input: 1.0, output: 3.0, cached: 0.2, reasoning: 3.0 },
|
||||
"z-ai/glm-4.5-air": { input: 0.13, output: 0.85, cached: 0.025, reasoning: 0.85 },
|
||||
"z-ai/glm-4.6": { input: 0.6, output: 2.2, cached: 0.11, reasoning: 2.2 },
|
||||
"z-ai/glm-4.6v": { input: 0.3, output: 0.9, reasoning: 0.9 },
|
||||
"z-ai/glm-4.7": { input: 0.6, output: 2.2, cached: 0.11, reasoning: 2.2 },
|
||||
"z-ai/glm-5": { input: 1.0, output: 3.2, cached: 0.2, reasoning: 3.2 },
|
||||
"z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 },
|
||||
"z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 },
|
||||
"z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 },
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,35 +0,0 @@
|
||||
// Token Plan — credit subscription keys on token-plan.<region>.maas.aliyuncs.com.
|
||||
// Fourth Alibaba key type: Coding Plan (alicode/alicode-intl) and Model Studio
|
||||
// (alims-intl) both reject these keys, and they reject Model Studio keys back.
|
||||
// Singapore is the only region that serves the plan; eu-central-1 answers
|
||||
// IllegalEndpoint. The Anthropic surface (/apps/anthropic/v1/messages) is not
|
||||
// authorized for this plan, so OpenAI-compatible mode is the only transport.
|
||||
export default {
|
||||
id: "alitp-intl",
|
||||
priority: 11,
|
||||
alias: "alitp-intl",
|
||||
display: {
|
||||
name: "Alibaba Token Plan",
|
||||
icon: "cloud",
|
||||
color: "#FF6A00",
|
||||
textIcon: "ATP",
|
||||
website: "https://www.alibabacloud.com/campaign/ai-landing-page-token",
|
||||
notice: {
|
||||
apiKeyUrl: "https://modelstudio.console.alibabacloud.com/?apiKey=1",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1/chat/completions",
|
||||
headers: {},
|
||||
quirks: { preserveCacheControl: true },
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3.8-max-preview", name: "Qwen3.8 Max Preview" },
|
||||
{ id: "qwen3.7-max", name: "Qwen3.7 Max" },
|
||||
{ id: "qwen3.7-plus", name: "Qwen3.7 Plus" },
|
||||
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
],
|
||||
};
|
||||
@@ -45,9 +45,6 @@ export default {
|
||||
clientSecret: "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",
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" },
|
||||
{ id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" },
|
||||
{ id: "gemini-3.6-flash-high", name: "Gemini 3.6 Flash (High)", upstreamModelId: "gemini-3.6-flash-tiered(high)" },
|
||||
{ id: "gemini-3.6-flash-medium", name: "Gemini 3.6 Flash (Medium)", upstreamModelId: "gemini-3.6-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.6-flash-low", name: "Gemini 3.6 Flash (Low)", upstreamModelId: "gemini-3.6-flash-tiered(low)" },
|
||||
@@ -79,7 +76,8 @@ export default {
|
||||
apiVersion: "v1internal",
|
||||
loadCodeAssistEndpoint: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
|
||||
onboardUserEndpoint: "https://cloudcode-pa.googleapis.com/v1internal:onboardUser",
|
||||
loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT,
|
||||
loadCodeAssistUserAgent: "google-api-nodejs-client/9.15.1",
|
||||
loadCodeAssistApiClient: "google-cloud-sdk vscode_cloudshelleditor/0.1",
|
||||
refreshLeadMs: 300000,
|
||||
},
|
||||
features: {
|
||||
|
||||
@@ -49,6 +49,9 @@ export default {
|
||||
header: "Authorization",
|
||||
scheme: "bearer",
|
||||
},
|
||||
hooks: [
|
||||
"claudeOverlay",
|
||||
],
|
||||
},
|
||||
usage: {
|
||||
oauthUrl: "https://api.anthropic.com/api/oauth/usage",
|
||||
|
||||
@@ -19,8 +19,6 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authType: "apikey",
|
||||
authModes: ["apikey"],
|
||||
hasProviderSpecificData: true,
|
||||
transport: {
|
||||
baseUrl: "https://api.cloudflare.com/client/v4/accounts/{accountId}/ai/v1/chat/completions",
|
||||
|
||||
@@ -38,10 +38,6 @@ export default {
|
||||
header: "Authorization",
|
||||
scheme: "bearer",
|
||||
},
|
||||
// Intl billing endpoint mirrors CN shape (data.Response.Data.Accounts[]).
|
||||
usage: {
|
||||
url: "https://www.codebuddy.ai/v2/billing/meter/get-user-resource",
|
||||
},
|
||||
},
|
||||
// Same model lineup exposed by the CN gateway — intl backend is the same catalog.
|
||||
models: [
|
||||
|
||||
@@ -23,9 +23,24 @@ export default {
|
||||
format: "commandcode",
|
||||
forceStream: true,
|
||||
headers: {
|
||||
"x-command-code-version": "0.25.7",
|
||||
"x-command-code-version": "1.10.0",
|
||||
"x-cli-environment": "cli",
|
||||
"User-Agent": "cli",
|
||||
},
|
||||
// Quota/billing endpoints (same alpha API the official CLI /usage calls).
|
||||
// whoami resolves orgId; credits+subscription+usage/summary then report the
|
||||
// 5-hour/weekly windows, plan, and period credits. See services/usage/commandcode.js.
|
||||
usage: {
|
||||
baseUrl: "https://api.commandcode.ai",
|
||||
whoamiUrl: "/alpha/whoami",
|
||||
creditsUrl: "/alpha/billing/credits",
|
||||
subscriptionsUrl: "/alpha/billing/subscriptions",
|
||||
summaryUrl: "/alpha/usage/summary",
|
||||
},
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
models: [
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
// Fish Audio TTS — the model id travels in an HTTP `model` header rather than the
|
||||
// JSON body, and the voice is a reference_id (a cloned or preset voice model).
|
||||
export default {
|
||||
id: "fish-audio",
|
||||
alias: "fish",
|
||||
display: {
|
||||
name: "Fish Audio",
|
||||
icon: "record_voice_over",
|
||||
color: "#1E9BF0",
|
||||
textIcon: "FA",
|
||||
website: "https://fish.audio",
|
||||
notice: {
|
||||
apiKeyUrl: "https://fish.audio/app/api-keys/",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
serviceKinds: ["tts"],
|
||||
ttsConfig: {
|
||||
baseUrl: "https://api.fish.audio/v1/tts",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "fish-audio",
|
||||
models: [
|
||||
{ id: "s2.1-pro-free", name: "S2.1 Pro Free" },
|
||||
{ id: "s2.1-pro", name: "S2.1 Pro" },
|
||||
{ id: "s2-pro", name: "S2 Pro" },
|
||||
{ id: "s1", name: "S1" },
|
||||
],
|
||||
},
|
||||
};
|
||||
@@ -36,7 +36,6 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" },
|
||||
{ id: "gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
|
||||
{ id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
|
||||
|
||||
@@ -21,7 +21,6 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "glm-5.3", name: "GLM 5.3" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
|
||||
@@ -45,7 +45,6 @@ export default {
|
||||
},
|
||||
],
|
||||
models: [
|
||||
{ id: "glm-5.3", name: "GLM 5.3" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
|
||||
@@ -75,6 +75,7 @@ import p72 from "./perplexity.js";
|
||||
import p73 from "./perplexity-agent.js";
|
||||
import p74 from "./playht.js";
|
||||
import p75 from "./qoder.js";
|
||||
import p76 from "./qwen.js";
|
||||
import p77 from "./recraft.js";
|
||||
import p78 from "./runwayml.js";
|
||||
import p79 from "./sdwebui.js";
|
||||
@@ -115,12 +116,6 @@ import p113 from "./morph.js";
|
||||
// import p114 from "./devin-cli.js";
|
||||
// import p104 from "./windsurf.js";
|
||||
import p115 from "./poolside.js";
|
||||
import p116 from "./tokenrouter.js";
|
||||
import p117 from "./selfhosted-stt.js";
|
||||
import p118 from "./selfhosted-tts.js";
|
||||
import p119 from "./selfhosted-embedding.js";
|
||||
import p120 from "./fish-audio.js";
|
||||
import p121 from "./alitp-intl.js";
|
||||
|
||||
export default [
|
||||
p0,
|
||||
@@ -199,6 +194,7 @@ export default [
|
||||
p73,
|
||||
p74,
|
||||
p75,
|
||||
p76,
|
||||
p77,
|
||||
p78,
|
||||
p79,
|
||||
@@ -237,10 +233,4 @@ export default [
|
||||
// p114, // devin-cli — hidden, spawns local agent with shell/fs access
|
||||
// p104, // windsurf — hidden, no tool calling
|
||||
p115,
|
||||
p116,
|
||||
p117,
|
||||
p118,
|
||||
p119,
|
||||
p120,
|
||||
p121,
|
||||
];
|
||||
|
||||
@@ -14,7 +14,7 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authModes: ["oauth", "apikey"],
|
||||
authModes: ["oauth"],
|
||||
hasOAuth: true,
|
||||
transport: {
|
||||
baseUrl: "https://llm.kimchi.dev/openai/v1/chat/completions",
|
||||
|
||||
@@ -15,8 +15,6 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authType: "apikey",
|
||||
authModes: ["apikey"],
|
||||
transport: {
|
||||
baseUrl: "https://ollama.com/api/chat",
|
||||
validateUrl: "https://ollama.com/api/tags",
|
||||
@@ -34,6 +32,5 @@ export default {
|
||||
serviceKinds: ["llm"],
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -22,28 +22,20 @@ export default {
|
||||
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
|
||||
headers: {},
|
||||
},
|
||||
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
|
||||
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
|
||||
// opencode-go models differ in endpoint support.
|
||||
transports: [
|
||||
{ format: "openai", baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
|
||||
{ format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
|
||||
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
|
||||
],
|
||||
models: [
|
||||
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", targetFormat: "claude" },
|
||||
{ id: "minimax-m2.7", name: "MiniMax M2.7", targetFormat: "claude" },
|
||||
{ id: "minimax-m2.5", name: "MiniMax M2.5", targetFormat: "claude" },
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", targetFormat: "claude" },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", targetFormat: "claude" },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", targetFormat: "claude" },
|
||||
],
|
||||
};
|
||||
|
||||
@@ -52,7 +52,5 @@ export default {
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
// PAT (apikey) connections also carry quota usage (via job-token exchange).
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
33
open-sse/providers/registry/qwen.js
Normal file
33
open-sse/providers/registry/qwen.js
Normal file
@@ -0,0 +1,33 @@
|
||||
export default {
|
||||
id: "qwen",
|
||||
hidden: true,
|
||||
priority: 130,
|
||||
alias: "qw",
|
||||
display: {
|
||||
name: "Qwen Code",
|
||||
icon: "psychology",
|
||||
color: "#10B981",
|
||||
website: "https://chat.qwen.ai",
|
||||
notice: {
|
||||
signupUrl: "https://chat.qwen.ai",
|
||||
},
|
||||
},
|
||||
category: "oauth",
|
||||
transport: {
|
||||
baseUrl: "https://portal.qwen.ai/v1/chat/completions",
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3-coder-plus", name: "Qwen3 Coder Plus" },
|
||||
{ id: "qwen3-coder-flash", name: "Qwen3 Coder Flash" },
|
||||
{ id: "vision-model", name: "Qwen3 Vision Model" },
|
||||
{ id: "coder-model", name: "Qwen3.6 Coder Model" },
|
||||
],
|
||||
oauth: {
|
||||
clientId: "f0304373b74a44d2b584a3fb70ca9e56",
|
||||
deviceCodeUrl: "https://chat.qwen.ai/api/v1/oauth2/device/code",
|
||||
tokenUrl: "https://chat.qwen.ai/api/v1/oauth2/token",
|
||||
scope: "openid profile email model.completion",
|
||||
codeChallengeMethod: "S256",
|
||||
refreshLeadMs: 1200000,
|
||||
},
|
||||
};
|
||||
@@ -1,73 +0,0 @@
|
||||
// Self-hosted, OpenAI-compatible embeddings (llama.cpp / llama-server, vLLM,
|
||||
// Infinity, text-embeddings-inference, ...) — the embeddings counterpart of
|
||||
// selfhosted-stt and selfhosted-tts.
|
||||
//
|
||||
// Routing a self-hosted embeddings server already WORKS today, via a custom
|
||||
// provider node: getEmbeddingAdapter() matches `openai-compatible-*` and
|
||||
// `custom-embedding-*` and returns openaiCompatNode, whose buildUrl reads
|
||||
// creds.providerSpecificData.baseUrl. What is missing is a first-class provider,
|
||||
// and the gap is visible rather than functional:
|
||||
//
|
||||
// /v1/embeddings on such a node -> 200, correct vectors
|
||||
// the Embedding page in the dashboard -> the node is not listed at all
|
||||
//
|
||||
// The page renders getProvidersByKind("embedding") plus provider nodes filtered
|
||||
// to `type === "custom-embedding"`. A node created as `openai-compatible` — the
|
||||
// natural choice when ONE endpoint serves chat and embeddings behind the same
|
||||
// front door — satisfies neither, so a working self-hosted embeddings endpoint is
|
||||
// invisible on the page whose job is to show embeddings providers. Diagnosed on a
|
||||
// deployment serving Qwen3-Embedding-8B at 4096 dimensions through exactly that
|
||||
// shape (2026-08-04).
|
||||
//
|
||||
// Declaring it as a provider with serviceKinds: ["embedding"] puts it on the page
|
||||
// beside Voyage, Jina and the rest, and keeps the per-connection baseUrl that
|
||||
// makes self-hosting possible at all.
|
||||
//
|
||||
// authType is "apikey" rather than "none" for the same reason as the STT and TTS
|
||||
// entries: it is what gives the connection a credentials record, and
|
||||
// providerSpecificData.baseUrl lives there. Local servers ignore the key itself;
|
||||
// any non-empty value works.
|
||||
export default {
|
||||
id: "selfhosted-embedding",
|
||||
priority: 50,
|
||||
hasFree: true,
|
||||
alias: "selfhosted-embedding",
|
||||
display: {
|
||||
name: "Self-hosted Embedding",
|
||||
icon: "cloud",
|
||||
color: "#ffffffff",
|
||||
textIcon: "SE",
|
||||
website: "https://github.com/ggml-org/llama.cpp",
|
||||
},
|
||||
category: "apikey",
|
||||
auth: {
|
||||
apiKey: {
|
||||
// Note the /v1: the adapter appends "/embeddings" to whatever it is given,
|
||||
// so a bare http://host:8080 resolves to http://host:8080/embeddings and
|
||||
// misses the OpenAI route entirely. Give it the OpenAI base, the same value
|
||||
// an OpenAI client would use. A trailing /embeddings is tolerated.
|
||||
text: "Set providerSpecificData.baseUrl to the OpenAI base URL, e.g. http://host:8080/v1 — /embeddings is appended. The API key is not checked by local servers; any value works.",
|
||||
},
|
||||
},
|
||||
// A self-hosted server serves whatever model it was started with, so the id
|
||||
// here is a placeholder for the UI: the request passes `model` straight
|
||||
// through, and llama-server ignores an unknown value rather than rejecting it.
|
||||
// Dimensions are deliberately NOT declared — they are a property of the loaded
|
||||
// weights, and asserting a number here would be a guess that silently
|
||||
// contradicts the server.
|
||||
models: [
|
||||
{ id: "embedding", name: "Self-hosted embedding model", kind: "embedding" },
|
||||
],
|
||||
serviceKinds: ["embedding"],
|
||||
embeddingConfig: {
|
||||
// Declared for shape-consistency with the other embedding providers, and
|
||||
// read by the UI — but NOT by the request path. openaiCompatNode resolves the
|
||||
// URL purely from creds.providerSpecificData.baseUrl (falling back to
|
||||
// api.openai.com), so unlike a fixed cloud provider this baseUrl never
|
||||
// reaches the wire. Stated plainly because a reader would otherwise
|
||||
// reasonably assume it is the default endpoint.
|
||||
baseUrl: "http://localhost:8080/v1/embeddings",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
},
|
||||
};
|
||||
@@ -1,48 +0,0 @@
|
||||
// Self-hosted, OpenAI-compatible speech-to-text (whisper.cpp, faster-whisper,
|
||||
// Speaches, vLLM-served Whisper, ...).
|
||||
//
|
||||
// Every other STT provider here is a named cloud service with a fixed endpoint.
|
||||
// This one exists so a locally-served /v1/audio/transcriptions can be used at
|
||||
// all: set the connection's providerSpecificData.baseUrl to the full URL of the
|
||||
// endpoint, exactly as the custom embedding providers already work.
|
||||
//
|
||||
// sttCore dispatches on `format`; anything that is not one of the five named
|
||||
// cloud shapes falls through to transcribeOpenAICompatible, which POSTs the
|
||||
// standard multipart body (file, model, and optional language / prompt /
|
||||
// response_format / temperature). That is precisely what whisper.cpp's OpenAI
|
||||
// endpoint accepts.
|
||||
//
|
||||
// authType is "apikey" rather than "none" so the connection carries a
|
||||
// credentials record — which is where providerSpecificData.baseUrl lives. Local
|
||||
// servers ignore the key itself; any non-empty value works.
|
||||
export default {
|
||||
id: "selfhosted-stt",
|
||||
priority: 50,
|
||||
hasFree: true,
|
||||
alias: "selfhosted-stt",
|
||||
display: {
|
||||
name: "Self-hosted STT",
|
||||
icon: "cloud",
|
||||
color: "#ffffffff",
|
||||
textIcon: "ST",
|
||||
website: "https://github.com/ggml-org/whisper.cpp",
|
||||
},
|
||||
category: "apikey",
|
||||
auth: {
|
||||
apiKey: {
|
||||
text: "Set providerSpecificData.baseUrl to the full transcriptions URL, e.g. http://host:8080/v1/audio/transcriptions. The API key is not checked by local servers; any value works.",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "whisper-1", name: "Whisper (self-hosted)", params: ["language", "response_format", "temperature", "prompt"], kind: "stt" },
|
||||
],
|
||||
serviceKinds: ["stt"],
|
||||
sttConfig: {
|
||||
// Overridden per connection by providerSpecificData.baseUrl; this default
|
||||
// only makes the provider usable out of the box on a same-host deployment.
|
||||
baseUrl: "http://localhost:8080/v1/audio/transcriptions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "openai",
|
||||
},
|
||||
};
|
||||
@@ -1,44 +0,0 @@
|
||||
// Self-hosted, OpenAI-compatible text-to-speech (Kokoro-FastAPI, openedai-speech,
|
||||
// vLLM-served TTS, ...) — the TTS counterpart of selfhosted-stt.
|
||||
//
|
||||
// Every other self-hostable TTS provider here (coqui, tortoise) carries a FIXED
|
||||
// localhost baseUrl in its registry entry and `authType: "none"`, and the generic
|
||||
// dispatcher reads `ttsConfig.baseUrl` from that entry rather than from the
|
||||
// connection. So there was no way to point TTS at a server on another host.
|
||||
//
|
||||
// `authType: "apikey"` is what makes the override possible at all: it gives the
|
||||
// connection a credentials record, which is where providerSpecificData.baseUrl
|
||||
// lives. Local servers ignore the key; any non-empty value works.
|
||||
export default {
|
||||
id: "selfhosted-tts",
|
||||
priority: 50,
|
||||
hasFree: true,
|
||||
alias: "selfhosted-tts",
|
||||
display: {
|
||||
name: "Self-hosted TTS",
|
||||
icon: "cloud",
|
||||
color: "#ffffffff",
|
||||
textIcon: "TT",
|
||||
website: "https://github.com/remsky/Kokoro-FastAPI",
|
||||
},
|
||||
category: "apikey",
|
||||
auth: {
|
||||
apiKey: {
|
||||
text: "Set providerSpecificData.baseUrl to the server root, e.g. http://host:8080 — /v1/audio/speech is appended. The API key is not checked by local servers; any value works.",
|
||||
},
|
||||
},
|
||||
// Voice is selected as "<model>/<voice>", the same convention the OpenAI TTS
|
||||
// adapter uses, so existing clients need no special casing.
|
||||
models: [
|
||||
{ id: "kokoro", name: "Kokoro (self-hosted)", params: ["voice", "response_format", "speed"], kind: "tts" },
|
||||
],
|
||||
serviceKinds: ["tts"],
|
||||
ttsConfig: {
|
||||
// Overridden per connection by providerSpecificData.baseUrl; this default
|
||||
// only makes the provider usable on a same-host deployment.
|
||||
baseUrl: "http://localhost:8880",
|
||||
defaultModel: "kokoro",
|
||||
authType: "apikey",
|
||||
format: "openai-speech",
|
||||
},
|
||||
};
|
||||
@@ -1,162 +0,0 @@
|
||||
export default {
|
||||
id: "tokenrouter",
|
||||
alias: "tokenrouter",
|
||||
aliases: ["tr"],
|
||||
uiAlias: "tokenrouter",
|
||||
display: {
|
||||
name: "TokenRouter",
|
||||
icon: "hub",
|
||||
color: "#0EA5E9",
|
||||
textIcon: "TR",
|
||||
website: "https://www.tokenrouter.com",
|
||||
notice: {
|
||||
text: "OpenAI-compatible gateway. 300+ models (OpenAI, Claude, Gemini, Qwen, DeepSeek, Kimi, GLM, dsb).",
|
||||
apiKeyUrl: "https://www.tokenrouter.com",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
thinkingConfig: {
|
||||
options: ["low", "medium", "high", "xhigh", "max"],
|
||||
defaultMode: "high",
|
||||
},
|
||||
transport: {
|
||||
baseUrl: "https://api.tokenrouter.com/v1/chat/completions",
|
||||
validateUrl: "https://api.tokenrouter.com/v1/models",
|
||||
thinkingFormat: "tokenrouter",
|
||||
},
|
||||
// Seed snapshot from live /v1/models (120 entries). Latest catalogue is
|
||||
// fetched via modelsFetcher; other ids still accepted via passthroughModels.
|
||||
models: [
|
||||
{ id: "MiniMax-Hailuo-2.3", name: "Minimax Hailuo 2.3", kind: "video" },
|
||||
{ id: "MiniMax-M3", name: "Minimax M3" },
|
||||
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5" },
|
||||
{ id: "anthropic/claude-haiku-4.5", name: "Claude Haiku 4.5" },
|
||||
{ id: "anthropic/claude-opus-4.5", name: "Claude Opus 4.5" },
|
||||
{ id: "anthropic/claude-opus-4.6", name: "Claude Opus 4.6" },
|
||||
{ id: "anthropic/claude-opus-4.7", name: "Claude Opus 4.7" },
|
||||
{ id: "anthropic/claude-opus-4.7-fast", name: "Claude Opus 4.7 Fast" },
|
||||
{ id: "anthropic/claude-opus-4.8", name: "Claude Opus 4.8" },
|
||||
{ id: "anthropic/claude-opus-4.8-fast", name: "Claude Opus 4.8 Fast" },
|
||||
{ id: "anthropic/claude-opus-5", name: "Claude Opus 5" },
|
||||
{ id: "anthropic/claude-opus-5-fast", name: "Claude Opus 5 Fast" },
|
||||
{ id: "anthropic/claude-sonnet-4", name: "Claude Sonnet 4" },
|
||||
{ id: "anthropic/claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
|
||||
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "bytedance-seed/seedream-4.5", name: "Seedream 4.5", kind: "image" },
|
||||
{ id: "bytedance-seed/seedream-5.0-lite", name: "Seedream 5.0 Lite", kind: "image" },
|
||||
{ id: "bytedance-seed/seedream-5.0-pro", name: "Seedream 5.0 Pro", kind: "image" },
|
||||
{ id: "claude-haiku-4-5", name: "Claude Haiku 4 5" },
|
||||
{ id: "claude-opus-4-8-m-aws", name: "Claude Opus 4 8 M Aws" },
|
||||
{ id: "deepseek/deepseek-v3.2", name: "Deepseek V3.2" },
|
||||
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-flash-0731", name: "Deepseek V4 Flash 0731" },
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
|
||||
{ id: "ex/gpt-5.4", name: "Gpt 5.4" },
|
||||
{ id: "google/gemini-2.5-flash-image", name: "Gemini 2.5 Flash Image" },
|
||||
{ id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
|
||||
{ id: "google/gemini-3-pro-image-preview", name: "Gemini 3 Pro Image Preview" },
|
||||
{ id: "google/gemini-3.1-flash-image-preview", name: "Gemini 3.1 Flash Image Preview" },
|
||||
{ id: "google/gemini-3.1-flash-lite-image", name: "Gemini 3.1 Flash Lite Image" },
|
||||
{ id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
|
||||
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
|
||||
{ id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
|
||||
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "google/gemini-embedding-2", name: "Gemini Embedding 2" },
|
||||
{ id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B It" },
|
||||
{ id: "happyhorse-1.0-t2v", name: "Happyhorse 1.0 T2V", kind: "video" },
|
||||
{ id: "kling-3.0-turbo", name: "Kling 3.0 Turbo", kind: "video" },
|
||||
{ id: "kling-v2-6", name: "Kling V2 6", kind: "video" },
|
||||
{ id: "kling-v3", name: "Kling V3", kind: "video" },
|
||||
{ id: "kling-v3-omni", name: "Kling V3 Omni", kind: "video" },
|
||||
{ id: "microsoft/mai-image-2.5", name: "Mai Image 2.5" },
|
||||
{ id: "minimax/minimax-m2-her", name: "Minimax M2 Her" },
|
||||
{ id: "minimax/minimax-m2.1", name: "Minimax M2.1" },
|
||||
{ id: "minimax/minimax-m2.1-highspeed", name: "Minimax M2.1 Highspeed" },
|
||||
{ id: "minimax/minimax-m2.5", name: "Minimax M2.5" },
|
||||
{ id: "minimax/minimax-m2.7", name: "Minimax M2.7" },
|
||||
{ id: "minimax/minimax-m2.7-highspeed", name: "Minimax M2.7 Highspeed" },
|
||||
{ id: "miromind/mirothinker-1-7-deepresearch", name: "Mirothinker 1 7 Deepresearch" },
|
||||
{ id: "miromind/mirothinker-1-7-deepresearch-mini", name: "Mirothinker 1 7 Deepresearch Mini" },
|
||||
{ id: "mistralai/devstral-2512", name: "Devstral 2512" },
|
||||
{ id: "mistralai/mistral-medium-3-5", name: "Mistral Medium 3 5" },
|
||||
{ id: "mistralai/mistral-small-2603", name: "Mistral Small 2603" },
|
||||
{ id: "mistralai/voxtral-small-24b-2507", name: "Voxtral Small 24B 2507" },
|
||||
{ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5" },
|
||||
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
|
||||
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "moonshotai/kimi-k3", name: "Kimi K3" },
|
||||
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
|
||||
{ id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", name: "Nemotron 3 Nano Omni 30B A3B Reasoning:Free" },
|
||||
{ id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B" },
|
||||
{ id: "openai/gpt-4o-mini", name: "Gpt 4O Mini" },
|
||||
{ id: "openai/gpt-5", name: "Gpt 5" },
|
||||
{ id: "openai/gpt-5-image", name: "Gpt 5 Image" },
|
||||
{ id: "openai/gpt-5-image-mini", name: "Gpt 5 Image Mini" },
|
||||
{ id: "openai/gpt-5-mini", name: "Gpt 5 Mini" },
|
||||
{ id: "openai/gpt-5.2", name: "Gpt 5.2" },
|
||||
{ id: "openai/gpt-5.4", name: "Gpt 5.4" },
|
||||
{ id: "openai/gpt-5.4-image-2", name: "Gpt 5.4 Image 2", kind: "image" },
|
||||
{ id: "openai/gpt-5.4-mini", name: "Gpt 5.4 Mini" },
|
||||
{ id: "openai/gpt-5.4-nano", name: "Gpt 5.4 Nano" },
|
||||
{ id: "openai/gpt-5.4-pro", name: "Gpt 5.4 Pro" },
|
||||
{ id: "openai/gpt-5.5", name: "Gpt 5.5" },
|
||||
{ id: "openai/gpt-5.5-pro", name: "Gpt 5.5 Pro" },
|
||||
{ id: "openai/gpt-5.6-luna", name: "Gpt 5.6 Luna" },
|
||||
{ id: "openai/gpt-5.6-sol", name: "Gpt 5.6 Sol" },
|
||||
{ id: "openai/gpt-5.6-terra", name: "Gpt 5.6 Terra" },
|
||||
{ id: "openai/gpt-audio", name: "Gpt Audio", kind: "audio" },
|
||||
{ id: "openai/gpt-audio-mini", name: "Gpt Audio Mini", kind: "audio" },
|
||||
{ id: "openai/gpt-oss-120b", name: "Gpt Oss 120B" },
|
||||
{ id: "qwen/qwen3-coder-next", name: "Qwen3 Coder Next" },
|
||||
{ id: "qwen/qwen3.5-122b-a10b", name: "Qwen3.5 122B A10B" },
|
||||
{ id: "qwen/qwen3.5-35b-a3b", name: "Qwen3.5 35B A3B" },
|
||||
{ id: "qwen/qwen3.5-397b-a17b", name: "Qwen3.5 397B A17B" },
|
||||
{ id: "qwen/qwen3.5-9b", name: "Qwen3.5 9B" },
|
||||
{ id: "qwen/qwen3.5-flash", name: "Qwen3.5 Flash" },
|
||||
{ id: "qwen/qwen3.5-plus-02-15", name: "Qwen3.5 Plus 02 15" },
|
||||
{ id: "qwen/qwen3.6-plus", name: "Qwen3.6 Plus" },
|
||||
{ id: "qwen/qwen3.7-max", name: "Qwen3.7 Max" },
|
||||
{ id: "qwen/qwen3.7-plus", name: "Qwen3.7 Plus" },
|
||||
{ id: "qwen/qwen3.8-max", name: "Qwen3.8 Max" },
|
||||
{ id: "qwen3.5-omni-plus", name: "Qwen3.5 Omni Plus" },
|
||||
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
|
||||
{ id: "sakana/fugu-ultra", name: "Fugu Ultra" },
|
||||
{ id: "seed-2-0-code-preview-260328", name: "Seed 2 0 Code Preview 260328" },
|
||||
{ id: "seed-2-0-lite-260428", name: "Seed 2 0 Lite 260428" },
|
||||
{ id: "seed-2-0-mini-260428", name: "Seed 2 0 Mini 260428" },
|
||||
{ id: "seed-2-0-pro-260328", name: "Seed 2 0 Pro 260328" },
|
||||
{ id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash" },
|
||||
{ id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash" },
|
||||
{ id: "tencent/hy3-preview", name: "Hy3 Preview" },
|
||||
{ id: "x-ai/grok-4.1-fast", name: "Grok 4.1 Fast" },
|
||||
{ id: "x-ai/grok-4.20-beta", name: "Grok 4.20 Beta" },
|
||||
{ id: "x-ai/grok-4.3", name: "Grok 4.3" },
|
||||
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
|
||||
{ id: "x-ai/grok-build-0.1", name: "Grok Build 0.1" },
|
||||
{ id: "xiaomi/mimo-v2-flash", name: "Mimo V2 Flash" },
|
||||
{ id: "xiaomi/mimo-v2-omni", name: "Mimo V2 Omni" },
|
||||
{ id: "xiaomi/mimo-v2-pro", name: "Mimo V2 Pro" },
|
||||
{ id: "xiaomi/mimo-v2.5", name: "Mimo V2.5" },
|
||||
{ id: "xiaomi/mimo-v2.5-pro", name: "Mimo V2.5 Pro" },
|
||||
{ id: "z-ai/glm-4.5-air", name: "Glm 4.5 Air" },
|
||||
{ id: "z-ai/glm-4.6", name: "Glm 4.6" },
|
||||
{ id: "z-ai/glm-4.6v", name: "Glm 4.6V" },
|
||||
{ id: "z-ai/glm-4.7", name: "Glm 4.7" },
|
||||
{ id: "z-ai/glm-5", name: "Glm 5" },
|
||||
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
|
||||
{ id: "z-ai/glm-5.1", name: "Glm 5.1" },
|
||||
{ id: "z-ai/glm-5.2", name: "Glm 5.2" },
|
||||
],
|
||||
serviceKinds: ["llm", "embedding", "image"],
|
||||
embeddingConfig: {
|
||||
baseUrl: "https://api.tokenrouter.com/v1/embeddings",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
},
|
||||
imageConfig: {
|
||||
baseUrl: "https://api.tokenrouter.com/v1/images/generations",
|
||||
},
|
||||
modelsFetcher: { url: "https://api.tokenrouter.com/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -25,17 +25,48 @@ export default {
|
||||
clientId: "b1a00492-073a-47ea-816f-4c329264a828",
|
||||
tokenUrl: "https://auth.x.ai/oauth2/token",
|
||||
refreshUrl: "https://auth.x.ai/oauth2/token",
|
||||
// OAuth-only SuperGrok quota surfaces:
|
||||
// - url: monthly API usage allotment (JSON)
|
||||
// - creditsUrl: weekly SuperGrok limit (grpc-web)
|
||||
// - settingsUrl: plan label (subscription_tier_display)
|
||||
usage: {
|
||||
url: "https://cli-chat-proxy.grok.com/v1/billing",
|
||||
creditsUrl: "https://grok.com/grok_api_v2.GrokBuildBilling/GetGrokCreditsConfig",
|
||||
settingsUrl: "https://cli-chat-proxy.grok.com/v1/settings",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "grok-4", name: "Grok 4" },
|
||||
{ id: "grok-4-fast-reasoning", name: "Grok 4 Fast Reasoning" },
|
||||
{ id: "grok-code-fast-1", name: "Grok Code Fast" },
|
||||
{ id: "grok-3", name: "Grok 3" },
|
||||
{ id: "grok-2-image-1212", name: "Grok 2 Image", params: ["n","response_format"], kind: "image" },
|
||||
{ id: "grok-imagine-video", name: "Grok Imagine Video", params: ["duration","aspect_ratio","resolution"], kind: "video" },
|
||||
{
|
||||
id: "grok-imagine-image-quality",
|
||||
name: "Grok Imagine Image Quality",
|
||||
capabilities: ["text2img", "edit"],
|
||||
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
|
||||
kind: "image",
|
||||
},
|
||||
{
|
||||
id: "grok-2-image-1212",
|
||||
name: "Grok 2 Image",
|
||||
capabilities: ["text2img", "edit"],
|
||||
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
|
||||
kind: "image",
|
||||
},
|
||||
{
|
||||
id: "grok-imagine-video",
|
||||
name: "Grok Imagine Video",
|
||||
params: ["duration", "aspect_ratio", "resolution"],
|
||||
kind: "video",
|
||||
},
|
||||
],
|
||||
serviceKinds: ["llm","imageToText","webSearch","image","video"],
|
||||
imageConfig: { baseUrl: "https://api.x.ai/v1/images/generations", bodyFields: ["model","prompt","n","response_format"] },
|
||||
serviceKinds: ["llm", "imageToText", "webSearch", "image", "video"],
|
||||
imageConfig: {
|
||||
baseUrl: "https://api.x.ai/v1/images/generations",
|
||||
editsUrl: "https://api.x.ai/v1/images/edits",
|
||||
bodyFields: ["model", "prompt", "n", "response_format", "aspect_ratio", "resolution", "image", "images"],
|
||||
},
|
||||
// Async video jobs (POST returns { request_id }, GET polls until done/failed).
|
||||
// Docs: https://docs.x.ai/developers/rest-api-reference/inference/videos
|
||||
videoConfig: { baseUrl: "https://api.x.ai/v1/videos" },
|
||||
@@ -44,4 +75,7 @@ export default {
|
||||
endpoint: "https://api.x.ai/v1/responses",
|
||||
pricingUrl: "https://x.ai/api#pricing",
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -15,11 +15,10 @@ export default {
|
||||
textIcon: "XM",
|
||||
website: "https://xiaomimimo.com",
|
||||
notice: {
|
||||
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
|
||||
apiKeyUrl: "https://xiaomimimo.com",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
serviceKinds: ["llm", "tts"],
|
||||
transport: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
validateUrl: "https://api.xiaomimimo.com/v1/models",
|
||||
@@ -43,12 +42,5 @@ export default {
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
|
||||
{ id: "mimo-v2-flash", name: "MiMo V2 Flash" },
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", kind: "tts" },
|
||||
],
|
||||
ttsConfig: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "xiaomi-mimo-tts",
|
||||
},
|
||||
};
|
||||
|
||||
@@ -47,26 +47,6 @@ export const CLAUDE_CLI_SPOOF_HEADERS = {
|
||||
"X-Stainless-Timeout": "600"
|
||||
};
|
||||
|
||||
const ANTHROPIC_BETA_BASE = [
|
||||
"claude-code-20250219",
|
||||
"oauth-2025-04-20",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"structured-outputs-2025-12-15",
|
||||
"fast-mode-2026-02-01",
|
||||
"redact-thinking-2026-02-12",
|
||||
"token-efficient-tools-2026-03-28",
|
||||
];
|
||||
const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"];
|
||||
|
||||
// Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them.
|
||||
export function selectAnthropicBeta(model = "") {
|
||||
const flags = [...ANTHROPIC_BETA_BASE];
|
||||
if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT);
|
||||
return flags.join(",");
|
||||
}
|
||||
|
||||
// Shared baseUrls
|
||||
export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
|
||||
|
||||
|
||||
@@ -31,13 +31,10 @@ const FORMAT_LEVELS = {
|
||||
step: L.base,
|
||||
};
|
||||
|
||||
const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
||||
|
||||
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
|
||||
const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
// gpt-5.6-sol accepts max (maps to xhigh on wire); live probe rejected ultra.
|
||||
{ pattern: "*gpt-5.6-sol*", levels: ["none", "minimal", "low", "medium", "high", "xhigh", "max"] },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
];
|
||||
|
||||
@@ -46,9 +43,7 @@ export function getThinkingLevels(provider, model) {
|
||||
if (provider === "kiro" && resolveKiroEffortPath(model) === null) return null;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (!caps.reasoning) return null;
|
||||
const hit = PATTERN_THINKING.find((entry) =>
|
||||
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
|
||||
);
|
||||
const hit = PATTERN_THINKING.find((p) => matchPattern(p.pattern, model));
|
||||
let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
|
||||
if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none");
|
||||
return levels;
|
||||
|
||||
@@ -25,17 +25,9 @@ function messagePayload(body) {
|
||||
|
||||
function captureSizeSnapshot(body) {
|
||||
const messages = messagePayload(body);
|
||||
const toolHistory = messages?.filter((message) =>
|
||||
message?.role === "tool"
|
||||
|| message?.role === "function"
|
||||
|| message?.tool_calls?.length
|
||||
|| message?.content?.some?.((part) => part?.type === "tool_use" || part?.type === "tool_result")
|
||||
) || [];
|
||||
return {
|
||||
bodyBytes: jsonBytes(body),
|
||||
messageBytes: messages ? jsonBytes(messages) : 0,
|
||||
toolSchemaBytes: jsonBytes(body?.tools || []),
|
||||
toolHistoryBytes: jsonBytes(toolHistory),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -344,10 +336,7 @@ export function formatHeadroomSizeLog(diagnostics) {
|
||||
const before = diagnostics?.before;
|
||||
const after = diagnostics?.after;
|
||||
if (!before || !after) return "";
|
||||
const effective = before.bodyBytes > 0
|
||||
? (((before.bodyBytes - after.bodyBytes) / before.bodyBytes) * 100).toFixed(1)
|
||||
: "0.0";
|
||||
return `body=${before.bodyBytes}B→${after.bodyBytes}B messages=${before.messageBytes}B→${after.messageBytes}B tools=${before.toolSchemaBytes || 0}B→${after.toolSchemaBytes || 0}B toolHistory=${before.toolHistoryBytes || 0}B→${after.toolHistoryBytes || 0}B effective=${effective}%`;
|
||||
return `body=${before.bodyBytes}B→${after.bodyBytes}B messages=${before.messageBytes}B→${after.messageBytes}B`;
|
||||
}
|
||||
|
||||
export function isHeadroomPhantomSavings(stats, diagnostics, minShrinkRatio = 0.05) {
|
||||
|
||||
@@ -1,173 +0,0 @@
|
||||
/**
|
||||
* Capacity Adapter — global fallback pools of models per input-modality capability
|
||||
* (vision / pdf / audioInput / videoInput).
|
||||
*
|
||||
* The pool models are appended as extra fallback candidates behind whatever models
|
||||
* were already going to be tried (a combo's members, or a single target model).
|
||||
* combo.js's existing reorderByCapabilities then floats a capable pool model to the
|
||||
* front only when none of the original models can handle the request — so this
|
||||
* never overrides a combo that already has a member covering the capability.
|
||||
*/
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
|
||||
const CAPABILITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
const HARD_CAPS = new Set(CAPABILITY_KEYS);
|
||||
const DEFAULT_FALLBACK_MODEL = "oc/mimo-v2.5-free";
|
||||
|
||||
// Normalize a capability entry to { enabled, roundRobin, models }. Backward-compat:
|
||||
// accept the legacy array form [{model, enabled}] (treated as enabled, fallback).
|
||||
function normalizeCapEntry(entry) {
|
||||
if (Array.isArray(entry)) {
|
||||
return { enabled: true, roundRobin: false, models: entry.map((e) => e?.model || e).filter(Boolean) };
|
||||
}
|
||||
if (entry && typeof entry === "object") {
|
||||
return {
|
||||
enabled: entry.enabled !== false,
|
||||
roundRobin: !!entry.roundRobin,
|
||||
models: Array.isArray(entry.models) ? entry.models.filter(Boolean) : [],
|
||||
};
|
||||
}
|
||||
return { enabled: false, roundRobin: false, models: [] };
|
||||
}
|
||||
|
||||
// Resolve one capability's full config. Enabled pools with no models fall back
|
||||
// to DEFAULT_FALLBACK_MODEL so the toggle is never a no-op.
|
||||
export function getCapacityAdapterConfig(cap, settings) {
|
||||
const entry = normalizeCapEntry(settings?.capacityAdapter?.[cap]);
|
||||
if (entry.enabled && entry.models.length === 0) {
|
||||
return { ...entry, models: [DEFAULT_FALLBACK_MODEL] };
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
// Flatten enabled models across all capability pools, in priority order, deduped.
|
||||
export function getCapacityAdapterModels(settings) {
|
||||
const seen = new Set();
|
||||
const models = [];
|
||||
for (const cap of CAPABILITY_KEYS) {
|
||||
const { enabled, models: pool } = getCapacityAdapterConfig(cap, settings);
|
||||
if (!enabled) continue;
|
||||
for (const m of pool) {
|
||||
if (!seen.has(m)) {
|
||||
seen.add(m);
|
||||
models.push(m);
|
||||
}
|
||||
}
|
||||
}
|
||||
return models;
|
||||
}
|
||||
|
||||
// Strategy for a capability: "round-robin" when enabled+roundRobin, else "fallback".
|
||||
export function getCapacityAdapterStrategy(cap, settings) {
|
||||
const { enabled, roundRobin } = getCapacityAdapterConfig(cap, settings);
|
||||
return enabled && roundRobin ? "round-robin" : "fallback";
|
||||
}
|
||||
|
||||
// Strategy from the request's required capabilities: picks the first capability
|
||||
// whose adapter pool is enabled and can satisfy a hard requirement.
|
||||
export function getActiveAdapterStrategy(requiredCapabilities, settings) {
|
||||
const hard = [...(requiredCapabilities || [])].filter((c) => HARD_CAPS.has(c));
|
||||
for (const cap of hard) {
|
||||
const { enabled, models } = getCapacityAdapterConfig(cap, settings);
|
||||
if (!enabled || models.length === 0) continue;
|
||||
return getCapacityAdapterStrategy(cap, settings);
|
||||
}
|
||||
return "fallback";
|
||||
}
|
||||
|
||||
function modelSatisfies(modelStr, requiredHard) {
|
||||
const slash = modelStr.indexOf("/");
|
||||
const provider = slash > 0 ? modelStr.slice(0, slash) : "";
|
||||
const model = slash > 0 ? modelStr.slice(slash + 1) : modelStr;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
return requiredHard.every((c) => caps[c] === true);
|
||||
}
|
||||
|
||||
// Prepend capacity-adapter models as priority candidates when NONE of the
|
||||
// original models (combo members, or the single target model) can satisfy the
|
||||
// request's required capabilities. Adapter models go FIRST (priority); the
|
||||
// original models follow as fallback. Leaves `models` untouched when the
|
||||
// original list already covers it (combo.js's reorderByCapabilities handles
|
||||
// that case via autoSwitch).
|
||||
export function augmentModelsWithCapacityAdapter(models, requiredCapabilities, settings) {
|
||||
const hard = [...(requiredCapabilities || [])].filter((c) => HARD_CAPS.has(c));
|
||||
if (hard.length === 0 || !Array.isArray(models) || models.length === 0) return models;
|
||||
if (models.some((m) => modelSatisfies(m, hard))) return models;
|
||||
|
||||
const pool = getCapacityAdapterModels(settings).filter((m) => !models.includes(m) && modelSatisfies(m, hard));
|
||||
if (pool.length === 0) return models;
|
||||
return [...pool, ...models];
|
||||
}
|
||||
|
||||
const CHARS_PER_TOKEN = 4; // rough estimate; avoids pulling in a tokenizer dependency
|
||||
const HEAD_KEEP = 6; // messages after system kept verbatim before dropping the middle
|
||||
|
||||
function blockLength(content) {
|
||||
if (typeof content === "string") return content.length;
|
||||
if (Array.isArray(content)) {
|
||||
return content.reduce((sum, b) => sum + (typeof b?.text === "string" ? b.text.length : 50), 0);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Trim history to fit a (possibly smaller) context window by dropping the MIDDLE.
|
||||
// Preserves: all system/instruction messages (head), and the trailing user run
|
||||
// carrying the media the switch happened for (tail). Older middle turns between
|
||||
// the head instructions and the current turn are dropped first.
|
||||
export function stripHistoryForContext(body, contextWindow) {
|
||||
const key = Array.isArray(body.messages) ? "messages"
|
||||
: Array.isArray(body.input) ? "input"
|
||||
: Array.isArray(body.contents) ? "contents"
|
||||
: null;
|
||||
if (!key) return body;
|
||||
const arr = body[key];
|
||||
if (!arr || arr.length === 0) return body;
|
||||
|
||||
const isSystem = (r) => r === "system" || r === "developer";
|
||||
const systemMsgs = arr.filter((m) => isSystem(m?.role));
|
||||
const rest = arr.filter((m) => !isSystem(m?.role));
|
||||
if (rest.length === 0) return body;
|
||||
|
||||
const isAssistant = (r) => r === "assistant" || r === "model";
|
||||
let i = rest.length - 1;
|
||||
while (i >= 0 && !isAssistant(rest[i]?.role)) i--;
|
||||
const tail = rest.slice(i + 1); // current user turn (has media) — always kept
|
||||
const older = rest.slice(0, i + 1); // everything before it
|
||||
if (older.length === 0) return body;
|
||||
|
||||
const contentOf = (m) => m.content ?? m.parts;
|
||||
// Cap at 80% of the adapter model's context window — leaves room for the response.
|
||||
const budgetChars = (contextWindow || 200000) * 0.8 * CHARS_PER_TOKEN;
|
||||
|
||||
// Prefer keeping the first HEAD_KEEP messages (initial instructions/context) verbatim;
|
||||
// only trim further if even that exceeds the adapter model's context window.
|
||||
const headKept = older.slice(0, HEAD_KEEP);
|
||||
let total = systemMsgs.concat(headKept, tail).reduce((s, m) => s + blockLength(contentOf(m)), 0);
|
||||
|
||||
// If head + tail overflow, drop head turns from the end (closest to middle) first.
|
||||
let head = headKept;
|
||||
while (total > budgetChars && head.length > 0) {
|
||||
const dropped = head.pop();
|
||||
total -= blockLength(contentOf(dropped));
|
||||
}
|
||||
|
||||
if (head.length === older.length) return body;
|
||||
return { ...body, [key]: [...systemMsgs, ...head, ...tail] };
|
||||
}
|
||||
|
||||
// Wrap a handleSingleModel callback so calls to a capacity-adapter model strip
|
||||
// history to fit its context window first. No-op passthrough when the pool is empty.
|
||||
export function withCapacityAdapterStripping(handleSingleModel, adapterModels) {
|
||||
const adapterSet = new Set(adapterModels);
|
||||
if (adapterSet.size === 0) return handleSingleModel;
|
||||
return (body, modelStr, ...rest) => {
|
||||
if (adapterSet.has(modelStr)) {
|
||||
const slash = modelStr.indexOf("/");
|
||||
const provider = slash > 0 ? modelStr.slice(0, slash) : "";
|
||||
const model = slash > 0 ? modelStr.slice(slash + 1) : modelStr;
|
||||
const { contextWindow } = getCapabilitiesForModel(provider, model);
|
||||
body = stripHistoryForContext(body, contextWindow);
|
||||
}
|
||||
return handleSingleModel(body, modelStr, ...rest);
|
||||
};
|
||||
}
|
||||
@@ -23,13 +23,23 @@ function flattenToolHistory(messages) {
|
||||
.filter((msg) => msg)
|
||||
.map((msg) => {
|
||||
if (msg.role === "tool" || msg.role === "function") {
|
||||
return { role: "assistant", content: `${TOOL_RESULT_PREFIX}${extractTextContent(msg.content) || String(msg.content ?? "")}]` };
|
||||
return {
|
||||
role: "assistant",
|
||||
content: `${TOOL_RESULT_PREFIX}${extractTextContent(msg.content) || String(msg.content ?? "")}]`,
|
||||
};
|
||||
}
|
||||
if (msg.role === "assistant" && Array.isArray(msg.tool_calls)) {
|
||||
const { tool_calls, ...rest } = msg;
|
||||
const names = tool_calls.map((c) => c?.function?.name || c?.name || "tool").join(", ");
|
||||
const base = extractTextContent(rest.content) || (typeof rest.content === "string" ? rest.content : "");
|
||||
return { ...rest, content: `${base}${base ? "\n" : ""}${TOOL_CALL_PREFIX}${names}]` };
|
||||
const names = tool_calls
|
||||
.map((c) => c?.function?.name || c?.name || "tool")
|
||||
.join(", ");
|
||||
const base =
|
||||
extractTextContent(rest.content) ||
|
||||
(typeof rest.content === "string" ? rest.content : "");
|
||||
return {
|
||||
...rest,
|
||||
content: `${base}${base ? "\n" : ""}${TOOL_CALL_PREFIX}${names}]`,
|
||||
};
|
||||
}
|
||||
if (Array.isArray(msg.content)) {
|
||||
const hasToolUse = msg.content.some((c) => c.type === "tool_use");
|
||||
@@ -41,7 +51,11 @@ function flattenToolHistory(messages) {
|
||||
for (const block of msg.content) {
|
||||
if (block.type === "text" && block.text) textParts.push(block.text);
|
||||
if (block.type === "tool_use") toolNames.push(block.name || "tool");
|
||||
if (block.type === "tool_result") toolResults.push(extractTextContent(block.content) || String(block.content ?? ""));
|
||||
if (block.type === "tool_result")
|
||||
toolResults.push(
|
||||
extractTextContent(block.content) ||
|
||||
String(block.content ?? ""),
|
||||
);
|
||||
}
|
||||
const { ...rest } = msg;
|
||||
let newContent = textParts.join("\n");
|
||||
@@ -61,7 +75,13 @@ function flattenToolHistory(messages) {
|
||||
// Reorder combo models by capability fit. Stable; never drops a model (fallback intact).
|
||||
// Tier 0: satisfies all hard + all soft. Tier 1: all hard only. Tier 2: rest.
|
||||
export function reorderByCapabilities(models, required) {
|
||||
if (!required || required.size === 0 || !Array.isArray(models) || models.length <= 1) return models;
|
||||
if (
|
||||
!required ||
|
||||
required.size === 0 ||
|
||||
!Array.isArray(models) ||
|
||||
models.length <= 1
|
||||
)
|
||||
return models;
|
||||
const hard = [...required].filter((c) => HARD_CAPS.has(c));
|
||||
const soft = [...required].filter((c) => !HARD_CAPS.has(c));
|
||||
|
||||
@@ -106,74 +126,26 @@ export function detectRequiredCapabilities(body) {
|
||||
const required = new Set();
|
||||
if (!body || typeof body !== "object") return required;
|
||||
|
||||
const addByMime = (mime) => {
|
||||
if (typeof mime !== "string") return;
|
||||
if (mime.startsWith("image/")) required.add("vision");
|
||||
else if (mime === "application/pdf") required.add("pdf");
|
||||
else if (mime.startsWith("audio/")) required.add("audioInput");
|
||||
else if (mime.startsWith("video/")) required.add("videoInput");
|
||||
};
|
||||
|
||||
const scanBlock = (b) => {
|
||||
if (!b || typeof b !== "object") return;
|
||||
const t = b.type;
|
||||
if (t === "image_url" || t === "image" || t === "input_image") required.add("vision");
|
||||
if (t === "input_audio" || t === "audio_url" || t === "audio") required.add("audioInput");
|
||||
if (t === "input_video" || t === "video_url" || t === "video") required.add("videoInput");
|
||||
if (t === "file" || t === "document" || t === "input_file") {
|
||||
// Infer modality from embedded mime when available; fall back to pdf for generic files.
|
||||
let fmime = null;
|
||||
if (b.input_audio?.format) fmime = `audio/${b.input_audio.format}`;
|
||||
else if (b.file?.file_data) fmime = String(b.file.file_data).match(/^data:([^;,]+)/)?.[1];
|
||||
else if (b.source?.media_type) fmime = b.source.media_type;
|
||||
else if (b.source?.data) fmime = String(b.source.data).match(/^data:([^;,]+)/)?.[1];
|
||||
if (fmime) addByMime(fmime);
|
||||
else required.add("pdf");
|
||||
}
|
||||
if (t === "image_url" || t === "image" || t === "input_image")
|
||||
required.add("vision");
|
||||
if (t === "file" || t === "document" || t === "input_file")
|
||||
required.add("pdf");
|
||||
// gemini parts: inlineData/fileData carry a mime
|
||||
addByMime(b.inlineData?.mimeType || b.fileData?.mimeType);
|
||||
const mime = b.inlineData?.mimeType || b.fileData?.mimeType;
|
||||
if (typeof mime === "string" && mime.startsWith("image/"))
|
||||
required.add("vision");
|
||||
if (mime === "application/pdf") required.add("pdf");
|
||||
};
|
||||
|
||||
const scanContent = (content) => {
|
||||
if (Array.isArray(content)) for (const b of content) scanBlock(b);
|
||||
};
|
||||
|
||||
const scanMessage = (m) => {
|
||||
if (!m || typeof m !== "object") return;
|
||||
|
||||
// Ollama / Hermes images array (strings or objects)
|
||||
if (Array.isArray(m.images) && m.images.length > 0) {
|
||||
required.add("vision");
|
||||
}
|
||||
|
||||
// Vercel AI SDK / Hermes attachments / experimental_attachments
|
||||
const attachments = m.experimental_attachments || m.attachments;
|
||||
if (Array.isArray(attachments)) {
|
||||
for (const att of attachments) {
|
||||
if (!att) continue;
|
||||
const mime = att.contentType || att.mediaType || (typeof att.url === "string" && att.url.match(/^data:([^;,]+)/)?.[1]);
|
||||
if (mime) addByMime(mime);
|
||||
else if (att.url || att.data) required.add("vision");
|
||||
}
|
||||
}
|
||||
|
||||
// Direct message-level modality properties
|
||||
if (m.image_url || m.image) required.add("vision");
|
||||
if (m.audio_url || m.audio) required.add("audioInput");
|
||||
|
||||
// Scan array content blocks
|
||||
scanContent(m.content);
|
||||
|
||||
// Scan string content for embedded data URIs
|
||||
if (typeof m.content === "string") {
|
||||
if (m.content.includes("data:image/")) required.add("vision");
|
||||
else if (m.content.includes("data:audio/")) required.add("audioInput");
|
||||
else if (m.content.includes("data:application/pdf")) required.add("pdf");
|
||||
}
|
||||
};
|
||||
|
||||
// Modalities: current user turn only (trailing user run across each known shape).
|
||||
for (const m of trailingUserItems(body.messages)) scanMessage(m); // openai / claude / hermes / ollama
|
||||
for (const m of trailingUserItems(body.messages)) scanContent(m.content); // openai / claude
|
||||
for (const it of trailingUserItems(body.input)) scanContent(it.content); // responses
|
||||
const contents = body.contents || body.request?.contents; // gemini / antigravity
|
||||
for (const c of trailingUserItems(contents)) scanContent(c.parts);
|
||||
@@ -213,9 +185,10 @@ export function getRotatedModels(models, comboName, strategy, stickyLimit = 1) {
|
||||
const rotationKey = comboName || "__default__";
|
||||
const normalizedStickyLimit = normalizeStickyLimit(stickyLimit);
|
||||
const existingState = comboRotationState.get(rotationKey);
|
||||
const state = typeof existingState === "number"
|
||||
const state =
|
||||
typeof existingState === "number"
|
||||
? { index: existingState, consecutiveUseCount: 0 }
|
||||
: (existingState || { index: 0, consecutiveUseCount: 0 });
|
||||
: existingState || { index: 0, consecutiveUseCount: 0 };
|
||||
|
||||
const currentIndex = state.index % models.length;
|
||||
const rotatedModels = rotateModelsFromIndex(models, currentIndex);
|
||||
@@ -256,10 +229,17 @@ export function getComboModelsFromData(modelStr, combosData) {
|
||||
if (modelStr.includes("/")) return null;
|
||||
|
||||
// Handle both array and object formats
|
||||
const combos = Array.isArray(combosData) ? combosData : (combosData?.combos || []);
|
||||
const combos = Array.isArray(combosData)
|
||||
? combosData
|
||||
: combosData?.combos || [];
|
||||
|
||||
const combo = combos.find(c => c.name === modelStr);
|
||||
if (combo && combo.models && combo.models.length > 0) {
|
||||
const combo = combos.find((c) => c.name === modelStr);
|
||||
if (
|
||||
combo &&
|
||||
combo.enabled !== false &&
|
||||
combo.models &&
|
||||
combo.models.length > 0
|
||||
) {
|
||||
return combo.models;
|
||||
}
|
||||
return null;
|
||||
@@ -277,9 +257,23 @@ export function getComboModelsFromData(modelStr, combosData) {
|
||||
* @param {number|string} [options.comboStickyLimit=1] - Requests per combo model before switching
|
||||
* @returns {Promise<Response>}
|
||||
*/
|
||||
export async function handleComboChat({ body, models, handleSingleModel, log, comboName, comboStrategy, comboStickyLimit = 1, autoSwitch = true }) {
|
||||
export async function handleComboChat({
|
||||
body,
|
||||
models,
|
||||
handleSingleModel,
|
||||
log,
|
||||
comboName,
|
||||
comboStrategy,
|
||||
comboStickyLimit = 1,
|
||||
autoSwitch = true,
|
||||
}) {
|
||||
// Apply rotation strategy if enabled
|
||||
let rotatedModels = getRotatedModels(models, comboName, comboStrategy, comboStickyLimit);
|
||||
let rotatedModels = getRotatedModels(
|
||||
models,
|
||||
comboName,
|
||||
comboStrategy,
|
||||
comboStickyLimit,
|
||||
);
|
||||
|
||||
// Auto-switch: float models that satisfy the request's required capabilities to the front.
|
||||
if (autoSwitch) {
|
||||
@@ -287,7 +281,10 @@ export async function handleComboChat({ body, models, handleSingleModel, log, co
|
||||
if (required.size > 0) {
|
||||
const reordered = reorderByCapabilities(rotatedModels, required);
|
||||
if (reordered[0] !== rotatedModels[0]) {
|
||||
log.info("COMBO", `auto-switch for [${[...required].join(",")}] → ${reordered[0]}`);
|
||||
log.info(
|
||||
"COMBO",
|
||||
`auto-switch for [${[...required].join(",")}] → ${reordered[0]}`,
|
||||
);
|
||||
}
|
||||
rotatedModels = reordered;
|
||||
}
|
||||
@@ -299,7 +296,10 @@ export async function handleComboChat({ body, models, handleSingleModel, log, co
|
||||
|
||||
for (let i = 0; i < rotatedModels.length; i++) {
|
||||
const modelStr = rotatedModels[i];
|
||||
log.info("COMBO", `Trying model ${i + 1}/${rotatedModels.length}: ${modelStr}`);
|
||||
log.info(
|
||||
"COMBO",
|
||||
`Trying model ${i + 1}/${rotatedModels.length}: ${modelStr}`,
|
||||
);
|
||||
|
||||
try {
|
||||
const result = await handleSingleModel(body, modelStr);
|
||||
@@ -315,48 +315,78 @@ export async function handleComboChat({ body, models, handleSingleModel, log, co
|
||||
let retryAfter = null;
|
||||
try {
|
||||
const errorBody = await result.clone().json();
|
||||
errorText = errorBody?.error?.message || errorBody?.error || errorBody?.message || errorText;
|
||||
errorText =
|
||||
errorBody?.error?.message ||
|
||||
errorBody?.error ||
|
||||
errorBody?.message ||
|
||||
errorText;
|
||||
retryAfter = errorBody?.retryAfter || null;
|
||||
} catch {
|
||||
// Ignore JSON parse errors
|
||||
}
|
||||
|
||||
// Track earliest retryAfter across all combo models
|
||||
if (retryAfter && (!earliestRetryAfter || new Date(retryAfter) < new Date(earliestRetryAfter))) {
|
||||
if (
|
||||
retryAfter &&
|
||||
(!earliestRetryAfter ||
|
||||
new Date(retryAfter) < new Date(earliestRetryAfter))
|
||||
) {
|
||||
earliestRetryAfter = retryAfter;
|
||||
}
|
||||
|
||||
// Normalize error text to string (Worker-safe)
|
||||
if (typeof errorText !== "string") {
|
||||
try { errorText = JSON.stringify(errorText); } catch { errorText = String(errorText); }
|
||||
try {
|
||||
errorText = JSON.stringify(errorText);
|
||||
} catch {
|
||||
errorText = String(errorText);
|
||||
}
|
||||
}
|
||||
|
||||
// Check if should fallback to next model
|
||||
const { shouldFallback, cooldownMs } = checkFallbackError(result.status, errorText);
|
||||
const { shouldFallback, cooldownMs } = checkFallbackError(
|
||||
result.status,
|
||||
errorText,
|
||||
);
|
||||
|
||||
if (!shouldFallback) {
|
||||
log.warn("COMBO", `Model ${modelStr} failed (no fallback)`, { status: result.status });
|
||||
log.warn("COMBO", `Model ${modelStr} failed (no fallback)`, {
|
||||
status: result.status,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
|
||||
// For transient errors (503/502/504), wait for cooldown before falling through
|
||||
// so a briefly-overloaded provider gets a chance to recover rather than being
|
||||
// skipped immediately (fixes: combo falls through on transient 503)
|
||||
if (cooldownMs && cooldownMs > 0 && cooldownMs <= 5000 &&
|
||||
(result.status === 503 || result.status === 502 || result.status === 504)) {
|
||||
log.info("COMBO", `Model ${modelStr} transient ${result.status}, waiting ${cooldownMs}ms before next`);
|
||||
await new Promise(r => setTimeout(r, cooldownMs));
|
||||
if (
|
||||
cooldownMs &&
|
||||
cooldownMs > 0 &&
|
||||
cooldownMs <= 5000 &&
|
||||
(result.status === 503 ||
|
||||
result.status === 502 ||
|
||||
result.status === 504)
|
||||
) {
|
||||
log.info(
|
||||
"COMBO",
|
||||
`Model ${modelStr} transient ${result.status}, waiting ${cooldownMs}ms before next`,
|
||||
);
|
||||
await new Promise((r) => setTimeout(r, cooldownMs));
|
||||
}
|
||||
|
||||
// Fallback to next model
|
||||
lastError = errorText || String(result.status);
|
||||
if (!lastStatus) lastStatus = result.status;
|
||||
log.warn("COMBO", `Model ${modelStr} failed, trying next`, { status: result.status });
|
||||
log.warn("COMBO", `Model ${modelStr} failed, trying next`, {
|
||||
status: result.status,
|
||||
});
|
||||
} catch (error) {
|
||||
// Catch unexpected exceptions to ensure fallback continues
|
||||
lastError = error.message || String(error);
|
||||
if (!lastStatus) lastStatus = 500;
|
||||
log.warn("COMBO", `Model ${modelStr} threw error, trying next`, { error: lastError });
|
||||
log.warn("COMBO", `Model ${modelStr} threw error, trying next`, {
|
||||
error: lastError,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -364,8 +394,9 @@ export async function handleComboChat({ body, models, handleSingleModel, log, co
|
||||
// Use 503 (Service Unavailable) rather than 406 (Not Acceptable) — 406 implies
|
||||
// the request itself is invalid, but here the providers are simply unavailable
|
||||
// or have no active credentials. 503 is more accurate and retryable by clients.
|
||||
const allDisabled = lastError && lastError.toLowerCase().includes("no credentials");
|
||||
const status = allDisabled ? 503 : (lastStatus || 503);
|
||||
const allDisabled =
|
||||
lastError && lastError.toLowerCase().includes("no credentials");
|
||||
const status = allDisabled ? 503 : lastStatus || 503;
|
||||
const msg = lastError || "All combo models unavailable";
|
||||
|
||||
if (earliestRetryAfter) {
|
||||
@@ -375,10 +406,10 @@ export async function handleComboChat({ body, models, handleSingleModel, log, co
|
||||
}
|
||||
|
||||
log.warn("COMBO", `All models failed | ${msg}`);
|
||||
return new Response(
|
||||
JSON.stringify({ error: { message: msg } }),
|
||||
{ status, headers: { "Content-Type": "application/json" } }
|
||||
);
|
||||
return new Response(JSON.stringify({ error: { message: msg } }), {
|
||||
status,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -396,7 +427,8 @@ function extractPanelText(json) {
|
||||
const msg = choice.message ?? choice.delta ?? {};
|
||||
const t = extractTextContent(msg.content);
|
||||
if (t.trim()) return t;
|
||||
if (typeof choice.text === "string" && choice.text.trim()) return choice.text;
|
||||
if (typeof choice.text === "string" && choice.text.trim())
|
||||
return choice.text;
|
||||
}
|
||||
|
||||
// Claude messages (text blocks share OpenAI's {type:"text"} shape)
|
||||
@@ -413,7 +445,9 @@ function extractPanelText(json) {
|
||||
// OpenAI Responses API
|
||||
if (Array.isArray(json.output)) {
|
||||
const t = json.output
|
||||
.flatMap((o) => (Array.isArray(o.content) ? o.content.map((c) => c?.text || "") : []))
|
||||
.flatMap((o) =>
|
||||
Array.isArray(o.content) ? o.content.map((c) => c?.text || "") : [],
|
||||
)
|
||||
.join("");
|
||||
if (t.trim()) return t;
|
||||
}
|
||||
@@ -480,8 +514,14 @@ function withTimeout(promise, ms) {
|
||||
return new Promise((resolve) => {
|
||||
const t = setTimeout(() => resolve({ __timeout: true }), ms);
|
||||
Promise.resolve(promise)
|
||||
.then((v) => { clearTimeout(t); resolve(v); })
|
||||
.catch((e) => { clearTimeout(t); resolve({ __error: e }); });
|
||||
.then((v) => {
|
||||
clearTimeout(t);
|
||||
resolve(v);
|
||||
})
|
||||
.catch((e) => {
|
||||
clearTimeout(t);
|
||||
resolve({ __error: e });
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -492,7 +532,10 @@ function withTimeout(promise, ms) {
|
||||
* still preferring a full panel when everyone is fast. Bounded by a hard timeout.
|
||||
* Returns a sparse array aligned to `calls` (undefined = not yet / dropped).
|
||||
*/
|
||||
function collectPanel(calls, { minPanel, stragglerGraceMs, panelHardTimeoutMs }) {
|
||||
function collectPanel(
|
||||
calls,
|
||||
{ minPanel, stragglerGraceMs, panelHardTimeoutMs },
|
||||
) {
|
||||
return new Promise((resolve) => {
|
||||
const out = new Array(calls.length);
|
||||
let settled = 0;
|
||||
@@ -509,13 +552,18 @@ function collectPanel(calls, { minPanel, stragglerGraceMs, panelHardTimeoutMs })
|
||||
const hardTimer = setTimeout(finish, panelHardTimeoutMs);
|
||||
calls.forEach((p, i) => {
|
||||
Promise.resolve(p)
|
||||
.then((v) => { out[i] = v; })
|
||||
.catch((e) => { out[i] = { __error: e }; })
|
||||
.then((v) => {
|
||||
out[i] = v;
|
||||
})
|
||||
.catch((e) => {
|
||||
out[i] = { __error: e };
|
||||
})
|
||||
.finally(() => {
|
||||
settled++;
|
||||
if (out[i] && out[i].ok) ok++;
|
||||
if (settled === calls.length) return finish();
|
||||
if (ok >= minPanel && !graceTimer) graceTimer = setTimeout(finish, stragglerGraceMs);
|
||||
if (ok >= minPanel && !graceTimer)
|
||||
graceTimer = setTimeout(finish, stragglerGraceMs);
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -544,12 +592,20 @@ function collectPanel(calls, { minPanel, stragglerGraceMs, panelHardTimeoutMs })
|
||||
* @param {Object} [options.tuning] - Override FUSION_DEFAULTS (minPanel, grace, timeout)
|
||||
* @returns {Promise<Response>}
|
||||
*/
|
||||
export async function handleFusionChat({ body, models, handleSingleModel, log, comboName, judgeModel, tuning }) {
|
||||
export async function handleFusionChat({
|
||||
body,
|
||||
models,
|
||||
handleSingleModel,
|
||||
log,
|
||||
comboName,
|
||||
judgeModel,
|
||||
tuning,
|
||||
}) {
|
||||
const panel = Array.isArray(models) ? models.filter(Boolean) : [];
|
||||
if (panel.length === 0) {
|
||||
return new Response(
|
||||
JSON.stringify({ error: { message: "Fusion combo has no models" } }),
|
||||
{ status: 400, headers: { "Content-Type": "application/json" } }
|
||||
{ status: 400, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
|
||||
@@ -561,13 +617,13 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
||||
const cfg = { ...FUSION_DEFAULTS, ...(tuning || {}) };
|
||||
const minPanel = Math.min(Math.max(2, cfg.minPanel), panel.length);
|
||||
const judge = judgeModel && judgeModel.trim() ? judgeModel.trim() : panel[0];
|
||||
log.info("FUSION", `Combo "${comboName}" | panel=${panel.length} [${panel.join(", ")}] | judge=${judge} | quorum=${minPanel}`);
|
||||
log.info(
|
||||
"FUSION",
|
||||
`Combo "${comboName}" | panel=${panel.length} [${panel.join(", ")}] | judge=${judge} | quorum=${minPanel}`,
|
||||
);
|
||||
|
||||
// 1. Fan out to the panel in parallel: non-streaming, tools stripped (we want prose).
|
||||
const { tools, tool_choice, stream_options, ...rest } = body;
|
||||
// Fusion runs panel models non-streaming; drop stream_options too, or providers
|
||||
// like DeepSeek reject it with "stream_options should be set along with stream = true".
|
||||
// See issue #3024.
|
||||
const { tools, tool_choice, ...rest } = body;
|
||||
const panelBody = { ...rest, stream: false };
|
||||
|
||||
// Flatten tool turns to prose so panel models keep context without emitting tool_calls.
|
||||
@@ -578,7 +634,9 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
||||
}
|
||||
|
||||
const t0 = Date.now();
|
||||
const calls = panel.map((m) => withTimeout(handleSingleModel(panelBody, m, true), cfg.panelHardTimeoutMs));
|
||||
const calls = panel.map((m) =>
|
||||
withTimeout(handleSingleModel(panelBody, m, true), cfg.panelHardTimeoutMs),
|
||||
);
|
||||
const settled = await collectPanel(calls, { ...cfg, minPanel });
|
||||
log.info("FUSION", `fan-out collected in ${Date.now() - t0}ms`);
|
||||
|
||||
@@ -587,10 +645,24 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
||||
for (let i = 0; i < settled.length; i++) {
|
||||
const res = settled[i];
|
||||
const model = panel[i];
|
||||
if (!res) { log.warn("FUSION", `Panel ${model} dropped (straggler/timeout)`); continue; }
|
||||
if (res.__timeout) { log.warn("FUSION", `Panel ${model} timed out`); continue; }
|
||||
if (res.__error) { log.warn("FUSION", `Panel ${model} threw`, { error: res.__error?.message || String(res.__error) }); continue; }
|
||||
if (!res.ok) { log.warn("FUSION", `Panel ${model} failed`, { status: res.status }); continue; }
|
||||
if (!res) {
|
||||
log.warn("FUSION", `Panel ${model} dropped (straggler/timeout)`);
|
||||
continue;
|
||||
}
|
||||
if (res.__timeout) {
|
||||
log.warn("FUSION", `Panel ${model} timed out`);
|
||||
continue;
|
||||
}
|
||||
if (res.__error) {
|
||||
log.warn("FUSION", `Panel ${model} threw`, {
|
||||
error: res.__error?.message || String(res.__error),
|
||||
});
|
||||
continue;
|
||||
}
|
||||
if (!res.ok) {
|
||||
log.warn("FUSION", `Panel ${model} failed`, { status: res.status });
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
const json = await res.clone().json();
|
||||
const text = extractPanelText(json);
|
||||
@@ -601,7 +673,9 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
||||
log.warn("FUSION", `Panel ${model} returned empty content`);
|
||||
}
|
||||
} catch (e) {
|
||||
log.warn("FUSION", `Panel ${model} unparseable`, { error: e.message || String(e) });
|
||||
log.warn("FUSION", `Panel ${model} unparseable`, {
|
||||
error: e.message || String(e),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -610,11 +684,14 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
||||
log.warn("FUSION", "All panel models failed");
|
||||
return new Response(
|
||||
JSON.stringify({ error: { message: "All fusion panel models failed" } }),
|
||||
{ status: 503, headers: { "Content-Type": "application/json" } }
|
||||
{ status: 503, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
if (answers.length === 1) {
|
||||
log.info("FUSION", `Only ${answers[0].model} succeeded — answering directly (no fusion)`);
|
||||
log.info(
|
||||
"FUSION",
|
||||
`Only ${answers[0].model} succeeded — answering directly (no fusion)`,
|
||||
);
|
||||
return handleSingleModel(body, answers[0].model);
|
||||
}
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
* This significantly reduces the risk of being flagged by Google's anti-abuse systems.
|
||||
*/
|
||||
|
||||
import { CLOUD_CODE_API, LOAD_CODE_ASSIST_HEADERS, ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS, LOAD_CODE_ASSIST_METADATA } from "../config/appConstants.js";
|
||||
import { CLOUD_CODE_API, LOAD_CODE_ASSIST_HEADERS, LOAD_CODE_ASSIST_METADATA } from "../config/appConstants.js";
|
||||
|
||||
// ─── Cache ────────────────────────────────────────────────────────────────────
|
||||
// connectionId -> { projectId: string, fetchedAt: number }
|
||||
@@ -157,10 +157,9 @@ export function removeConnection(connectionId) {
|
||||
*/
|
||||
async function fetchProjectId(accessToken, signal, provider) {
|
||||
const endpoints = CLOUD_CODE_API[provider] || CLOUD_CODE_API["gemini-cli"];
|
||||
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
|
||||
const response = await fetch(endpoints.loadCodeAssist, {
|
||||
method: "POST",
|
||||
headers: { ...headers, "Authorization": `Bearer ${accessToken}` },
|
||||
headers: { ...LOAD_CODE_ASSIST_HEADERS, "Authorization": `Bearer ${accessToken}` },
|
||||
body: JSON.stringify({ metadata: LOAD_CODE_ASSIST_METADATA }),
|
||||
signal
|
||||
});
|
||||
@@ -187,7 +186,7 @@ async function fetchProjectId(accessToken, signal, provider) {
|
||||
}
|
||||
}
|
||||
|
||||
return onboardUser(accessToken, tierID, signal, endpoints, provider);
|
||||
return onboardUser(accessToken, tierID, signal, endpoints);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -198,11 +197,10 @@ async function fetchProjectId(accessToken, signal, provider) {
|
||||
* @param {AbortSignal} externalSignal – propagated from the connection's AbortController
|
||||
* @returns {Promise<string|null>}
|
||||
*/
|
||||
async function onboardUser(accessToken, tierID, externalSignal, endpoints, provider) {
|
||||
async function onboardUser(accessToken, tierID, externalSignal, endpoints) {
|
||||
console.log(`[ProjectId] Onboarding user with tier: ${tierID}`);
|
||||
|
||||
const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA };
|
||||
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
|
||||
const MAX_ATTEMPTS = 5;
|
||||
|
||||
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
|
||||
@@ -218,7 +216,7 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
|
||||
try {
|
||||
const response = await fetch(endpoints.onboardUser, {
|
||||
method: "POST",
|
||||
headers: { ...headers, "Authorization": `Bearer ${accessToken}` },
|
||||
headers: { ...LOAD_CODE_ASSIST_HEADERS, "Authorization": `Bearer ${accessToken}` },
|
||||
body: JSON.stringify(reqBody),
|
||||
signal: localCtrl.signal
|
||||
});
|
||||
|
||||
@@ -19,15 +19,9 @@ function isAnthropicCompatible(provider) {
|
||||
return typeof provider === "string" && provider.startsWith(ANTHROPIC_COMPATIBLE_PREFIX);
|
||||
}
|
||||
|
||||
// Resolve the API type (chat vs responses) for an openai-compatible node.
|
||||
// The stored apiType on the connection's providerSpecificData (kept in sync with
|
||||
// the node on create/update) is authoritative. Falls back to the node ID
|
||||
// substring for legacy nodes created before apiType was persisted — their IDs
|
||||
// embed the type: openai-compatible-<chat|responses>-<uuid>.
|
||||
export function resolveOpenAICompatibleApiType(provider, credentials = null) {
|
||||
const stored = credentials?.providerSpecificData?.apiType;
|
||||
if (stored === "chat" || stored === "responses") return stored;
|
||||
return typeof provider === "string" && provider.includes("responses") ? "responses" : "chat";
|
||||
function getOpenAICompatibleType(provider) {
|
||||
if (!isOpenAICompatible(provider)) return "chat";
|
||||
return provider.includes("responses") ? "responses" : "chat";
|
||||
}
|
||||
|
||||
// Detect request format from body structure
|
||||
@@ -111,9 +105,9 @@ export function detectFormat(body) {
|
||||
}
|
||||
|
||||
// Get provider config (internal — no external runtime consumer)
|
||||
function getProviderConfig(provider, credentials = null) {
|
||||
function getProviderConfig(provider) {
|
||||
if (isOpenAICompatible(provider)) {
|
||||
const apiType = resolveOpenAICompatibleApiType(provider, credentials);
|
||||
const apiType = getOpenAICompatibleType(provider);
|
||||
return {
|
||||
...PROVIDERS.openai,
|
||||
format: apiType === "responses" ? "openai-responses" : "openai",
|
||||
@@ -131,14 +125,14 @@ function getProviderConfig(provider, credentials = null) {
|
||||
}
|
||||
|
||||
// Get target format for provider
|
||||
export function getTargetFormat(provider, credentials = null) {
|
||||
export function getTargetFormat(provider) {
|
||||
if (isOpenAICompatible(provider)) {
|
||||
return resolveOpenAICompatibleApiType(provider, credentials) === "responses" ? "openai-responses" : "openai";
|
||||
return getOpenAICompatibleType(provider) === "responses" ? "openai-responses" : "openai";
|
||||
}
|
||||
if (isAnthropicCompatible(provider)) {
|
||||
return "claude";
|
||||
}
|
||||
const config = getProviderConfig(provider, credentials);
|
||||
const config = getProviderConfig(provider);
|
||||
return config.format || "openai";
|
||||
}
|
||||
|
||||
|
||||
56
open-sse/services/providerTimeout.js
Normal file
56
open-sse/services/providerTimeout.js
Normal file
@@ -0,0 +1,56 @@
|
||||
/**
|
||||
* Per-provider connect timeout overrides from user settings.
|
||||
* Settings are read from the DB lazily and cached with a short TTL
|
||||
* so UI changes take effect without requiring a restart.
|
||||
*/
|
||||
|
||||
let cached = {};
|
||||
let cacheTs = 0;
|
||||
const CACHE_TTL_MS = 10_000; // 10s — responsive enough for dashboard changes
|
||||
|
||||
async function refreshCache() {
|
||||
const now = Date.now();
|
||||
if (now - cacheTs < CACHE_TTL_MS && Object.keys(cached).length > 0) return cached;
|
||||
|
||||
try {
|
||||
const { getSettings } = await import("@/lib/localDb");
|
||||
// Return full settings so we can read providerTimeouts + globalTimeoutMs
|
||||
cached = await getSettings();
|
||||
cacheTs = now;
|
||||
} catch {
|
||||
// If DB is unavailable, keep stale cache — don't throw on hot path
|
||||
}
|
||||
return cached;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the effective connect timeout for a provider.
|
||||
* Priority: per-provider override > global default timeout (settings) > registry config > env default.
|
||||
* @param {string} providerId
|
||||
* @param {number} configTimeoutMs - timeoutMs from the static provider registry config
|
||||
* @param {number} envDefaultMs - global default from env (FETCH_CONNECT_TIMEOUT_MS)
|
||||
* @returns {number} timeout in milliseconds
|
||||
*/
|
||||
export async function resolveProviderTimeoutMs(providerId, configTimeoutMs, envDefaultMs) {
|
||||
const overrides = await refreshCache();
|
||||
|
||||
// 1. Per-provider override (set in provider detail page)
|
||||
const providerOverride = overrides.providerTimeouts?.[providerId];
|
||||
if (providerOverride?.timeoutMs && Number.isFinite(providerOverride.timeoutMs) && providerOverride.timeoutMs > 0) {
|
||||
return providerOverride.timeoutMs;
|
||||
}
|
||||
|
||||
// 2. Global default timeout (set in Profile / Settings page)
|
||||
const globalDefault = overrides.defaultTimeoutMs;
|
||||
if (globalDefault && Number.isFinite(globalDefault) && globalDefault > 0) {
|
||||
return globalDefault;
|
||||
}
|
||||
|
||||
// 3. Registry per-provider config
|
||||
if (configTimeoutMs && Number.isFinite(configTimeoutMs) && configTimeoutMs > 0) {
|
||||
return configTimeoutMs;
|
||||
}
|
||||
|
||||
// 4. Env default
|
||||
return envDefaultMs;
|
||||
}
|
||||
@@ -10,12 +10,6 @@
|
||||
*
|
||||
* On any error the live cache stays empty and chatExecuteCall surfaces the
|
||||
* problem to the user as "model config not yet fetched, retry shortly".
|
||||
*
|
||||
* PAT (Personal Access Token, pt-...) connections: a PAT cannot sign COSY
|
||||
* requests directly, so we exchange it for a short-lived job token (jt-...)
|
||||
* via openapi.qoder.sh/api/v1/jobToken/exchange (plain JSON POST), then use
|
||||
* that job token for signing. Job-token traffic must hit api2.qoder.sh —
|
||||
* api3 rejects jt- with "Login expired" (403).
|
||||
*/
|
||||
|
||||
import { createHash } from "crypto";
|
||||
@@ -24,30 +18,11 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { buildCosyHeaders } from "../shared/qoder/cosy.js";
|
||||
import {
|
||||
QODER_MODEL_LIST_URL,
|
||||
QODER_CHAT_BASE_ALT,
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
QODER_USERINFO_URL,
|
||||
QODER_IDE_VERSION,
|
||||
QODER_CLIENT_TYPE,
|
||||
} from "../shared/qoder/constants.js";
|
||||
|
||||
const FETCH_TIMEOUT_MS = 15_000;
|
||||
const CACHE_TTL_MS = 60 * 60 * 1000; // 1h, same as the Kiro catalog
|
||||
|
||||
const PAT_PREFIX = "pt-";
|
||||
|
||||
// PAT → job-token cache: a job token is short-lived (24h), so we keep it per
|
||||
// PAT and re-exchange once it is within 5 minutes of expiry.
|
||||
const PAT_REFRESH_BUFFER_MS = 5 * 60 * 1000;
|
||||
const PAT_DEFAULT_TTL_MS = 24 * 60 * 60 * 1000;
|
||||
|
||||
export function isQoderPat(token) {
|
||||
return typeof token === "string" && token.startsWith(PAT_PREFIX);
|
||||
}
|
||||
|
||||
/** @type {Map<string, { accessToken: string, userId: string, expiresAt: number }>} */
|
||||
const patJobCache = new Map();
|
||||
|
||||
/** @type {Map<string, { expiresAt: number, models: any[], rawConfigs: Map<string, object>, fetched: boolean }>} */
|
||||
const catalogCache = new Map();
|
||||
|
||||
@@ -59,109 +34,6 @@ const catalogCache = new Map();
|
||||
*/
|
||||
const inflight = new Map();
|
||||
|
||||
/**
|
||||
* Exchange a Qoder PAT (pt-...) for a short-lived job token (jt-...).
|
||||
* This endpoint is plain JSON POST — NOT COSY-signed.
|
||||
*/
|
||||
async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
{
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
"Cosy-Version": QODER_IDE_VERSION,
|
||||
"Cosy-ClientType": QODER_CLIENT_TYPE,
|
||||
},
|
||||
body: JSON.stringify({ personal_token: pat }),
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
throw new Error(`qoder PAT exchange failed: ${res.status} ${text.slice(0, 200)}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
if (!data.token) throw new Error("qoder PAT exchange returned no job token");
|
||||
|
||||
let expiresAt = Date.now() + PAT_DEFAULT_TTL_MS;
|
||||
if (data.expires_at) {
|
||||
const parsed = Date.parse(data.expires_at);
|
||||
if (!Number.isNaN(parsed)) expiresAt = parsed;
|
||||
} else if (typeof data.expires_in === "number" && data.expires_in > 0) {
|
||||
expiresAt = Date.now() + data.expires_in;
|
||||
}
|
||||
return { jobToken: data.token, jobRefreshToken: data.refresh_token || "", expiresAt };
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the Qoder userId for a job token (needed for COSY signing).
|
||||
* Returns "" on any failure — callers fall back to the stored userId.
|
||||
*/
|
||||
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null) {
|
||||
try {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_USERINFO_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${jobToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
},
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
const data = await res.json().catch(() => ({}));
|
||||
return data.id || data.userId || data.user_id || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a PAT to a job-token credential, cached per-PAT.
|
||||
*/
|
||||
async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
|
||||
const cached = patJobCache.get(pat);
|
||||
if (cached && cached.expiresAt - Date.now() > PAT_REFRESH_BUFFER_MS) return cached;
|
||||
|
||||
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal);
|
||||
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal);
|
||||
const resolved = { accessToken: jobToken, userId, expiresAt };
|
||||
patJobCache.set(pat, resolved);
|
||||
return resolved;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve connection credentials to COSY-signable form:
|
||||
* - PAT (pt-...) connections → exchanged to a job token (jt-...) + userId
|
||||
* - everything else → passed through unchanged
|
||||
*/
|
||||
export async function resolveQoderCredentials(credentials, proxyOptions = null, signal = null) {
|
||||
const raw = credentials?.apiKey || credentials?.accessToken;
|
||||
if (isQoderPat(raw)) {
|
||||
const resolved = await resolvePatCredential(raw, proxyOptions, signal);
|
||||
return {
|
||||
...credentials,
|
||||
accessToken: resolved.accessToken,
|
||||
apiKey: undefined,
|
||||
providerSpecificData: {
|
||||
authMethod: "pat",
|
||||
...(credentials?.providerSpecificData || {}),
|
||||
userId: resolved.userId || credentials?.providerSpecificData?.userId || "",
|
||||
machineId: credentials?.providerSpecificData?.machineId || "",
|
||||
},
|
||||
};
|
||||
}
|
||||
return credentials;
|
||||
}
|
||||
|
||||
/**
|
||||
* Stable cache key per credential (so different login sessions for the same
|
||||
* account share an entry).
|
||||
@@ -196,16 +68,10 @@ async function fetchQoderCatalogRaw(credentials, signal, proxyOptions = null) {
|
||||
const creds = cosyCredsFromConnection(credentials);
|
||||
if (!creds.userId || !creds.authToken) return null;
|
||||
|
||||
// Job-token traffic is rejected by api3 ("Login expired" 403) — the
|
||||
// official qodercli serves it from api2 instead.
|
||||
const modelListUrl = String(creds.authToken).startsWith("jt-")
|
||||
? `${QODER_CHAT_BASE_ALT}/algo/api/v2/model/list`
|
||||
: QODER_MODEL_LIST_URL;
|
||||
|
||||
const headers = {
|
||||
Accept: "application/json",
|
||||
"Accept-Encoding": "identity",
|
||||
...buildCosyHeaders(Buffer.alloc(0), modelListUrl, creds),
|
||||
...buildCosyHeaders(Buffer.alloc(0), QODER_MODEL_LIST_URL, creds),
|
||||
};
|
||||
|
||||
const controller = new AbortController();
|
||||
@@ -226,7 +92,7 @@ async function fetchQoderCatalogRaw(credentials, signal, proxyOptions = null) {
|
||||
}
|
||||
}
|
||||
response = await proxyAwareFetch(
|
||||
modelListUrl,
|
||||
QODER_MODEL_LIST_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers,
|
||||
@@ -293,16 +159,11 @@ export async function getQoderModelConfig(credentials, modelKey, options = {}) {
|
||||
* one upstream request per credential.
|
||||
*/
|
||||
export async function resolveQoderModels(credentials, options = {}) {
|
||||
let resolved;
|
||||
try {
|
||||
resolved = await resolveQoderCredentials(credentials, options.proxyOptions, options.signal);
|
||||
} catch (error) {
|
||||
options.log?.warn?.("QODER", `PAT exchange failed: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
if (!resolved?.accessToken || !(resolved.providerSpecificData || {}).userId) return null;
|
||||
if (!credentials?.accessToken) return null;
|
||||
const psd = credentials.providerSpecificData || {};
|
||||
if (!psd.userId) return null;
|
||||
|
||||
const key = cacheKey(resolved);
|
||||
const key = cacheKey(credentials);
|
||||
const now = Date.now();
|
||||
if (!options.forceRefresh) {
|
||||
const cached = catalogCache.get(key);
|
||||
@@ -319,7 +180,7 @@ export async function resolveQoderModels(credentials, options = {}) {
|
||||
}
|
||||
|
||||
const fetchPromise = (async () => {
|
||||
const fetched = await fetchQoderCatalogRaw(resolved, options.signal, options.proxyOptions);
|
||||
const fetched = await fetchQoderCatalogRaw(credentials, options.signal, options.proxyOptions);
|
||||
if (!fetched) return null;
|
||||
const entry = {
|
||||
expiresAt: Date.now() + CACHE_TTL_MS,
|
||||
|
||||
@@ -6,6 +6,7 @@ import {
|
||||
refreshKimiToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshKiroToken,
|
||||
refreshIflowToken,
|
||||
@@ -25,6 +26,7 @@ export {
|
||||
refreshKimiToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshKiroToken,
|
||||
refreshIflowToken,
|
||||
@@ -135,6 +137,7 @@ const REFRESH_HANDLERS = {
|
||||
antigravity: (c, log) => refreshGoogleToken(c.refreshToken, PROVIDERS.antigravity.clientId, PROVIDERS.antigravity.clientSecret, log),
|
||||
claude: (c, log) => refreshClaudeOAuthToken(c.refreshToken, log),
|
||||
codex: (c, log) => refreshCodexToken(c.refreshToken, log),
|
||||
qwen: (c, log) => refreshQwenToken(c.refreshToken, log),
|
||||
iflow: (c, log) => refreshIflowToken(c.refreshToken, log),
|
||||
github: (c, log) => refreshGitHubToken(c.refreshToken, log),
|
||||
kiro: (c, log) => refreshKiroToken(c.refreshToken, c.providerSpecificData, log),
|
||||
@@ -202,6 +205,7 @@ export function formatProviderCredentials(provider, credentials, log) {
|
||||
};
|
||||
|
||||
case "codex":
|
||||
case "qwen":
|
||||
case "iflow":
|
||||
case "openai":
|
||||
case "openrouter":
|
||||
|
||||
@@ -40,6 +40,11 @@ const REFRESH_PROFILES = {
|
||||
url: () => OAUTH_ENDPOINTS.anthropic.token,
|
||||
dedupKey: "claude",
|
||||
},
|
||||
qwen: {
|
||||
url: () => OAUTH_ENDPOINTS.qwen.token,
|
||||
dedupKey: "qwen",
|
||||
parse: (tokens) => tokens.resource_url ? { providerSpecificData: { resourceUrl: tokens.resource_url } } : {},
|
||||
},
|
||||
iflow: {
|
||||
url: () => OAUTH_ENDPOINTS.iflow.token,
|
||||
dedupKey: "iflow",
|
||||
@@ -186,6 +191,11 @@ export async function refreshGoogleToken(refreshToken, clientId, clientSecret, l
|
||||
}, log);
|
||||
}
|
||||
|
||||
// Qwen: form body + clientId, surfaces resource_url. Delegate to refreshAccessToken("qwen", ...).
|
||||
export async function refreshQwenToken(refreshToken, log) {
|
||||
return refreshAccessToken("qwen", refreshToken, {}, log);
|
||||
}
|
||||
|
||||
export function classifyOAuthRefreshError(errorText = "", status = 0) {
|
||||
let parsed = null;
|
||||
try {
|
||||
|
||||
@@ -10,12 +10,14 @@ import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitReset
|
||||
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
|
||||
import { getKiroUsage } from "./usage/kiro.js";
|
||||
import { getMiniMaxUsage } from "./usage/minimax.js";
|
||||
import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js";
|
||||
import { getCodeBuddyCnUsage } from "./usage/codebuddy-cn.js";
|
||||
import { getXaiUsage } from "./usage/xai.js";
|
||||
import { getGrokCliUsage } from "./usage/grok-cli.js";
|
||||
import { getKimiUsage } from "./usage/kimi.js";
|
||||
import { getDeepseekUsage } from "./usage/deepseek.js";
|
||||
import { resolveQoderCredentials } from "./qoderModels.js";
|
||||
import { getCommandCodeUsage } from "./usage/commandcode.js";
|
||||
import {
|
||||
getQwenUsage,
|
||||
getIflowUsage,
|
||||
getOllamaUsage,
|
||||
getGlmUsage,
|
||||
@@ -33,30 +35,27 @@ const USAGE_HANDLERS = {
|
||||
github: (c) => getGitHubUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
"gemini-cli": (c) => getGeminiUsage(c.accessToken, c.providerDataWithProjectId, c.proxyOptions),
|
||||
antigravity: (c) => getAntigravityUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
claude: (c) => getClaudeUsage(c.accessToken, c.proxyOptions, { force: c.force }),
|
||||
claude: (c) => getClaudeUsage(c.accessToken, c.proxyOptions),
|
||||
codex: (c) => getCodexUsage(c.accessToken, c.proxyOptions),
|
||||
kiro: (c) => getKiroUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
qoder: async (c) => {
|
||||
// PAT (pt-...) connections must be exchanged to a job token before the
|
||||
// quota endpoint accepts them.
|
||||
const resolved = await resolveQoderCredentials(c, c.proxyOptions).catch(() => null);
|
||||
return getQoderUsage(resolved?.accessToken || c.accessToken, c.proxyOptions);
|
||||
},
|
||||
qoder: (c) => getQoderUsage(c.accessToken, c.proxyOptions),
|
||||
qwen: (c) => getQwenUsage(c.accessToken, c.providerSpecificData),
|
||||
iflow: (c) => getIflowUsage(c.accessToken),
|
||||
ollama: (c) => getOllamaUsage(c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
ollama: (c) => getOllamaUsage(c.accessToken),
|
||||
glm: (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
"glm-cn": (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
minimax: (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
"minimax-cn": (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
"vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions),
|
||||
"codebuddy-cn": (c) => getCodeBuddyCnUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
"codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
xai: (c) => getXaiUsage(c.accessToken, c.proxyOptions),
|
||||
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData),
|
||||
deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions),
|
||||
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
|
||||
};
|
||||
|
||||
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {
|
||||
export async function getUsageForProvider(connection, proxyOptions = null) {
|
||||
const { provider, accessToken, apiKey, providerSpecificData, projectId } = connection;
|
||||
const providerDataWithProjectId = {
|
||||
...(providerSpecificData || {}),
|
||||
@@ -65,13 +64,5 @@ export async function getUsageForProvider(connection, proxyOptions = null, optio
|
||||
|
||||
const handler = USAGE_HANDLERS[provider];
|
||||
if (!handler) return { message: `Usage API not implemented for ${provider}` };
|
||||
return await handler({
|
||||
provider,
|
||||
accessToken,
|
||||
apiKey,
|
||||
providerSpecificData,
|
||||
providerDataWithProjectId,
|
||||
proxyOptions,
|
||||
force: options.force === true,
|
||||
});
|
||||
return await handler({ provider, accessToken, apiKey, providerSpecificData, providerDataWithProjectId, proxyOptions });
|
||||
}
|
||||
|
||||
@@ -19,43 +19,7 @@ const CLAUDE_CONFIG = {
|
||||
const OAUTH_429_COOLDOWN_MS = 180000;
|
||||
const oauthCooldown = new Map();
|
||||
|
||||
// Dedup + short TTL cache per access token. Many tabs / many accounts / auto-refresh
|
||||
// all funnel through here; without this each call hits Anthropic and triggers 429.
|
||||
const USAGE_CACHE_TTL_MS = 300000;
|
||||
const usageCache = new Map(); // token -> { promise } | { result, expiresAt }
|
||||
|
||||
export async function getClaudeUsage(accessToken, proxyOptions = null, options = {}) {
|
||||
const force = options?.force === true;
|
||||
|
||||
// Serve in-flight or fresh cached result (skip on manual force)
|
||||
if (!force && accessToken) {
|
||||
const hit = usageCache.get(accessToken);
|
||||
if (hit?.promise) return hit.promise;
|
||||
if (hit && hit.expiresAt > Date.now()) return hit.result;
|
||||
}
|
||||
|
||||
const stale = (!force && accessToken && usageCache.get(accessToken)?.result) || null;
|
||||
|
||||
const promise = (async () => {
|
||||
const result = await fetchClaudeUsageRaw(accessToken, proxyOptions);
|
||||
// Only cache real quota data, not soft-failure {message: ...} payloads
|
||||
if (accessToken && result?.quotas) {
|
||||
usageCache.set(accessToken, {
|
||||
result,
|
||||
expiresAt: Date.now() + USAGE_CACHE_TTL_MS,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
// Soft failure (429/error): prefer the last good read over a transient error
|
||||
if (stale) return stale;
|
||||
return result;
|
||||
})();
|
||||
|
||||
if (accessToken) usageCache.set(accessToken, { promise });
|
||||
return promise;
|
||||
}
|
||||
|
||||
async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
|
||||
export async function getClaudeUsage(accessToken, proxyOptions = null) {
|
||||
try {
|
||||
// Skip OAuth usage call while this token is cooling down from a recent 429
|
||||
const cooldownUntil = oauthCooldown.get(accessToken);
|
||||
|
||||
@@ -43,17 +43,17 @@ function refillCadence(acc) {
|
||||
return "Monthly";
|
||||
}
|
||||
|
||||
async function getCodeBuddyUsage(providerId, accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
const token = accessToken || apiKey;
|
||||
if (!token) {
|
||||
return { message: `CodeBuddy (${providerId}) credential not available.` };
|
||||
return { message: "CodeBuddy CN credential not available." };
|
||||
}
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(U(providerId).url, {
|
||||
const response = await proxyAwareFetch(U(PROVIDER_ID).url, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
...(PROVIDERS[providerId]?.headers || {}),
|
||||
...(PROVIDERS[PROVIDER_ID]?.headers || {}),
|
||||
Authorization: `Bearer ${token}`,
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
@@ -129,18 +129,10 @@ async function getCodeBuddyUsage(providerId, accessToken, apiKey, providerSpecif
|
||||
});
|
||||
|
||||
const basePkg = refills[0] || accounts[0] || {};
|
||||
const plan = basePkg.PackageName || basePkg.SubProductName || "CodeBuddy";
|
||||
const plan = basePkg.PackageName || basePkg.SubProductName || "CodeBuddy CN";
|
||||
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: `CodeBuddy (${providerId}) error: ${error.message}` };
|
||||
return { message: `CodeBuddy CN error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
return getCodeBuddyUsage(PROVIDER_ID, accessToken, apiKey, providerSpecificData, proxyOptions);
|
||||
}
|
||||
|
||||
export async function getCodeBuddyIntlUsage(accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
return getCodeBuddyUsage("codebuddy-intl", accessToken, apiKey, providerSpecificData, proxyOptions);
|
||||
}
|
||||
|
||||
203
open-sse/services/usage/commandcode.js
Normal file
203
open-sse/services/usage/commandcode.js
Normal file
@@ -0,0 +1,203 @@
|
||||
/**
|
||||
* CommandCode usage handler
|
||||
*
|
||||
* Mirrors the official command-code CLI /usage command: it calls the alpha API
|
||||
* to surface the 5-hour + weekly usage windows, the subscription plan, and the
|
||||
* credits consumed in the current billing period.
|
||||
*
|
||||
* GET /alpha/whoami → org.id (org-scoped billing; null for personal)
|
||||
* GET /alpha/billing/credits → { credits: { monthlyCredits, purchasedCredits,
|
||||
* freeCredits }, windowLimits: { fiveHour, weekly } }
|
||||
* GET /alpha/billing/subscriptions → { data: { planId, currentPeriodStart, ... } }
|
||||
* GET /alpha/usage/summary?since= → period token/cost totals
|
||||
*
|
||||
* The CLI fetches whoami first (for orgId), then credits + subscription in
|
||||
* parallel, then the summary with since = currentPeriodStart. We keep the same
|
||||
* order/dependencies: window limits live on credits, and the plan period start
|
||||
* determines the summary window.
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U, parseResetTime } from "./shared.js";
|
||||
|
||||
const USAGE = U("commandcode");
|
||||
const BASE = USAGE.baseUrl || "https://api.commandcode.ai";
|
||||
const WHOAMI_URL = BASE + (USAGE.whoamiUrl || "/alpha/whoami");
|
||||
const CREDITS_URL = BASE + (USAGE.creditsUrl || "/alpha/billing/credits");
|
||||
const SUBSCRIPTIONS_URL =
|
||||
BASE + (USAGE.subscriptionsUrl || "/alpha/billing/subscriptions");
|
||||
const SUMMARY_URL = BASE + (USAGE.summaryUrl || "/alpha/usage/summary");
|
||||
|
||||
function buildHeaders(token) {
|
||||
return {
|
||||
Authorization: `Bearer ${token}`,
|
||||
Accept: "application/json",
|
||||
};
|
||||
}
|
||||
|
||||
/** Build a normalized quota row. `unit` is "$" — the API reports currency credits. */
|
||||
function makeQuota({ used, total, resetAt, unlimited = false, unit = "$" }) {
|
||||
const safeTotal = Math.max(0, Number(total) || 0);
|
||||
const safeUsed = Math.max(0, Number(used) || 0);
|
||||
if (unlimited || safeTotal === 0) {
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: 0,
|
||||
remainingPercentage: unlimited ? 100 : 0,
|
||||
resetAt: resetAt || null,
|
||||
unit,
|
||||
unlimited: true,
|
||||
};
|
||||
}
|
||||
const remaining = Math.max(0, safeTotal - safeUsed);
|
||||
const remainingPercentage = (remaining / safeTotal) * 100;
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: safeTotal,
|
||||
remainingPercentage,
|
||||
resetAt: resetAt || null,
|
||||
unit,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} apiKey - commandcode API key (user_...)
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getCommandCodeUsage(apiKey, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "CommandCode credential not available." };
|
||||
}
|
||||
|
||||
const headers = buildHeaders(apiKey);
|
||||
|
||||
try {
|
||||
// whoami resolves the org id (billing is org-scoped; null for personal).
|
||||
const whoamiRes = await proxyAwareFetch(
|
||||
WHOAMI_URL,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
|
||||
return { message: "CommandCode credential invalid or expired." };
|
||||
}
|
||||
if (!whoamiRes.ok) {
|
||||
return { message: `CommandCode whoami API error (${whoamiRes.status}).` };
|
||||
}
|
||||
const whoami = await whoamiRes.json().catch(() => null);
|
||||
const orgId = whoami?.org?.id ?? null;
|
||||
|
||||
const orgQuery = orgId ? `?orgId=${encodeURIComponent(orgId)}` : "";
|
||||
|
||||
const [creditsRes, subsRes] = await Promise.all([
|
||||
proxyAwareFetch(
|
||||
CREDITS_URL + orgQuery,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
proxyAwareFetch(
|
||||
SUBSCRIPTIONS_URL + orgQuery,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
]);
|
||||
|
||||
if (
|
||||
creditsRes.status === 401 ||
|
||||
creditsRes.status === 403 ||
|
||||
subsRes.status === 401 ||
|
||||
subsRes.status === 403
|
||||
) {
|
||||
return { message: "CommandCode credential invalid or expired." };
|
||||
}
|
||||
if (!creditsRes.ok) {
|
||||
return {
|
||||
message: `CommandCode credits API error (${creditsRes.status}).`,
|
||||
};
|
||||
}
|
||||
|
||||
const credits = await creditsRes.json().catch(() => null);
|
||||
const subs = await subsRes.json().catch(() => null);
|
||||
|
||||
const subData = subs?.data;
|
||||
const planId = subData?.planId ?? null;
|
||||
const periodStart = subData?.currentPeriodStart ?? null;
|
||||
|
||||
// Summary needs `since`; the CLI falls back to first-of-month when the
|
||||
// subscription period start is unavailable.
|
||||
const since = periodStart || firstOfMonth();
|
||||
const summaryRes = await proxyAwareFetch(
|
||||
`${SUMMARY_URL}?since=${encodeURIComponent(since)}`,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
const summary = summaryRes.ok
|
||||
? await summaryRes.json().catch(() => null)
|
||||
: null;
|
||||
|
||||
const quotas = {};
|
||||
const windowLimits = credits?.windowLimits || {};
|
||||
|
||||
const fiveHour = windowLimits.fiveHour;
|
||||
if (fiveHour && Number(fiveHour.cap) > 0) {
|
||||
quotas["5-hour window"] = makeQuota({
|
||||
used: fiveHour.used,
|
||||
total: fiveHour.cap,
|
||||
resetAt: parseResetTime(fiveHour.resetAt),
|
||||
});
|
||||
}
|
||||
|
||||
const weekly = windowLimits.weekly;
|
||||
if (weekly && Number(weekly.cap) > 0) {
|
||||
quotas["Weekly window"] = makeQuota({
|
||||
used: weekly.used,
|
||||
total: weekly.cap,
|
||||
resetAt: parseResetTime(weekly.resetAt),
|
||||
});
|
||||
}
|
||||
|
||||
// Monthly credits consumed this billing period (from summary when present,
|
||||
// else the credits object's monthlyCredits as a fallback).
|
||||
const monthlyUsed =
|
||||
typeof summary?.totalCredits === "number"
|
||||
? summary.totalCredits
|
||||
: typeof credits?.credits?.monthlyCredits === "number"
|
||||
? credits.credits.monthlyCredits
|
||||
: 0;
|
||||
|
||||
const monthlyTotal =
|
||||
typeof credits?.credits?.monthlyCredits === "number"
|
||||
? credits.credits.monthlyCredits
|
||||
: 0;
|
||||
|
||||
if (monthlyTotal > 0 || monthlyUsed > 0) {
|
||||
quotas["Monthly credits"] = makeQuota({
|
||||
used: monthlyUsed,
|
||||
total: monthlyTotal,
|
||||
resetAt: periodStart ? undefined : null,
|
||||
});
|
||||
}
|
||||
|
||||
if (Object.keys(quotas).length === 0) {
|
||||
return {
|
||||
plan: planId || "CommandCode",
|
||||
message: "CommandCode connected, but no quota was reported.",
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
plan: planId || "CommandCode",
|
||||
quotas,
|
||||
periodBasis: summary?.periodBasis || "billing-period",
|
||||
};
|
||||
} catch (error) {
|
||||
return { message: `CommandCode usage error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
function firstOfMonth() {
|
||||
const now = new Date();
|
||||
return new Date(now.getFullYear(), now.getMonth(), 1).toISOString();
|
||||
}
|
||||
@@ -161,12 +161,7 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
if (data.models) {
|
||||
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
|
||||
const importantModels = [
|
||||
'gemini-3.7-flash-high',
|
||||
'gemini-3.7-flash-medium',
|
||||
'gemini-3.7-flash-low',
|
||||
'gemini-3.6-flash-high',
|
||||
'gemini-3.6-flash-medium',
|
||||
'gemini-3.6-flash-low',
|
||||
'gemini-3-flash-agent',
|
||||
'gemini-3.5-flash-low',
|
||||
'gemini-3.5-flash-extra-low',
|
||||
'gemini-pro-agent',
|
||||
@@ -174,8 +169,10 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
'claude-sonnet-4-6',
|
||||
'claude-opus-4-6-thinking',
|
||||
'gpt-oss-120b-medium',
|
||||
'gemini-3-flash',
|
||||
// Image generation models
|
||||
'gemini-3.1-flash-image',
|
||||
'gemini-3-pro-image',
|
||||
];
|
||||
|
||||
for (const [modelKey, info] of Object.entries(data.models)) {
|
||||
|
||||
@@ -91,24 +91,6 @@ function resolvePlan(user, config) {
|
||||
return "Grok Build";
|
||||
}
|
||||
|
||||
// Display only; upstream remains authoritative for access and quota enforcement.
|
||||
function planFromAccessToken(accessToken) {
|
||||
try {
|
||||
const payload = JSON.parse(Buffer.from(accessToken.split(".")[1], "base64url"));
|
||||
return {
|
||||
0: "Free",
|
||||
1: "SuperGrok",
|
||||
2: "X Basic",
|
||||
3: "X Premium",
|
||||
4: "X Premium Plus",
|
||||
5: "SuperGrok Heavy",
|
||||
6: "SuperGrok Lite",
|
||||
}[payload.tier] || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
function makeQuota({ used, total, resetAt, unlimited = false }) {
|
||||
const safeTotal = Math.max(0, toFiniteNumber(total, 0));
|
||||
const safeUsed = Math.max(0, toFiniteNumber(used, 0));
|
||||
@@ -389,7 +371,6 @@ export async function getGrokCliUsage(accessToken, providerSpecificData = null,
|
||||
}
|
||||
|
||||
const parsed = parseGrokCliBilling(billing, user);
|
||||
parsed.plan = planFromAccessToken(accessToken) || parsed.plan;
|
||||
|
||||
if (!parsed.quotas || Object.keys(parsed.quotas).length === 0) {
|
||||
// Paid SuperGrok often returns cap=0 over REST but exposes the shared
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* Misc usage handlers (iFlow, Ollama, GLM, Vercel AI Gateway, Qoder)
|
||||
* Misc usage handlers (Qwen, iFlow, Ollama, GLM, Vercel AI Gateway, Qoder)
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
@@ -15,6 +15,23 @@ const GLM_QUOTA_URLS = {
|
||||
// Returns { balance: "95.50", total_used: "4.50" } (USD as decimal strings).
|
||||
const VERCEL_AI_GATEWAY_CREDITS_URL = U("vercel-ai-gateway").url;
|
||||
|
||||
/**
|
||||
* Qwen Usage
|
||||
*/
|
||||
export async function getQwenUsage(accessToken, providerSpecificData) {
|
||||
try {
|
||||
const resourceUrl = providerSpecificData?.resourceUrl;
|
||||
if (!resourceUrl) {
|
||||
return { message: "Qwen connected. No resource URL available." };
|
||||
}
|
||||
|
||||
// Qwen may have usage endpoint at resource URL
|
||||
return { message: "Qwen connected. Usage tracked per request." };
|
||||
} catch (error) {
|
||||
return { message: "Unable to fetch Qwen usage." };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* iFlow Usage
|
||||
*/
|
||||
@@ -29,86 +46,23 @@ export async function getIflowUsage(accessToken) {
|
||||
|
||||
/**
|
||||
* Ollama Cloud Usage
|
||||
* GET https://ollama.com/api/usage — session (5h) + weekly (7d) `usage` is a 0..1
|
||||
* ratio (1.0 = limit reached, e.g. weekly 100% used). No reset timestamp exposed.
|
||||
* POST https://ollama.com/api/me — plan label (fail-open).
|
||||
* Auth: Authorization: Bearer <apiKey>
|
||||
* Ollama Cloud uses an API key from ollama.com/settings/keys
|
||||
* and has no public usage API — free tier has light usage limits (resets every 5h & 7d).
|
||||
* This returns an informational message with the plan details.
|
||||
*/
|
||||
export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "Ollama Cloud API key not available." };
|
||||
}
|
||||
|
||||
export async function getOllamaUsage(accessToken, providerSpecificData) {
|
||||
try {
|
||||
const response = await proxyAwareFetch("https://ollama.com/api/usage", {
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
Accept: "application/json",
|
||||
},
|
||||
}, proxyOptions);
|
||||
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
return { message: "Ollama Cloud API key invalid or expired." };
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
return { message: `Ollama Cloud usage API error (${response.status}).` };
|
||||
}
|
||||
|
||||
let data;
|
||||
try {
|
||||
data = await response.json();
|
||||
} catch {
|
||||
return { message: "Ollama Cloud usage response was not JSON." };
|
||||
}
|
||||
|
||||
// Best-effort plan label from /api/me
|
||||
const me = await proxyAwareFetch("https://ollama.com/api/me", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
Accept: "application/json",
|
||||
"Content-Length": "0",
|
||||
},
|
||||
}, proxyOptions).then((r) => (r.ok ? r.json() : null)).catch(() => null);
|
||||
|
||||
const planRaw = typeof me?.Plan === "string" ? me.Plan : "";
|
||||
const plan = planRaw
|
||||
? planRaw.charAt(0).toUpperCase() + planRaw.slice(1).toLowerCase()
|
||||
: "Ollama Cloud";
|
||||
|
||||
const limits = data?.limits && typeof data.limits === "object" ? data.limits : {};
|
||||
|
||||
// Ollama `usage` is a 0..1 ratio (1.0 = limit reached). Convert to a 0..100
|
||||
// bar. Do NOT set absolute `remaining` — QuotaTable reads remainingPercentage.
|
||||
function ratioQuota(usageRatio, resetAt = null) {
|
||||
const ratio = Math.max(0, Math.min(1, Number(usageRatio) || 0));
|
||||
const usedPct = Math.round(ratio * 100);
|
||||
return { used: usedPct, total: 100, remainingPercentage: 100 - usedPct, resetAt, unlimited: false };
|
||||
}
|
||||
|
||||
const sessionRaw = limits.session?.usage;
|
||||
const weeklyRaw = limits.weekly?.usage;
|
||||
const sessionNum = Number(sessionRaw);
|
||||
const weeklyNum = Number(weeklyRaw);
|
||||
const hasSession = sessionRaw !== undefined && sessionRaw !== null && !Number.isNaN(sessionNum);
|
||||
const hasWeekly = weeklyRaw !== undefined && weeklyRaw !== null && !Number.isNaN(weeklyNum);
|
||||
|
||||
if (!hasSession && !hasWeekly) {
|
||||
// Ollama Cloud does not expose a public quota/usage API.
|
||||
// The provider is configured as noAuth with a notice explaining limits.
|
||||
// We return a graceful message so the UI shows a friendly state instead of an error.
|
||||
const plan = providerSpecificData?.plan || "Free";
|
||||
return {
|
||||
plan,
|
||||
message: "Ollama Cloud connected. No usage limits reported.",
|
||||
quotas: {},
|
||||
message: "Ollama Cloud uses a free tier with light usage limits (resets every 5h & 7d). For detailed usage tracking, visit ollama.com/settings/keys.",
|
||||
quotas: [],
|
||||
};
|
||||
}
|
||||
|
||||
const quotas = {};
|
||||
if (hasSession) quotas["Session (5h)"] = ratioQuota(sessionNum);
|
||||
if (hasWeekly) quotas["Weekly (7d)"] = ratioQuota(weeklyNum);
|
||||
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: `Ollama Cloud error: ${error.message}` };
|
||||
return { message: "Unable to fetch Ollama Cloud usage." };
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
326
open-sse/services/usage/xai.js
Normal file
326
open-sse/services/usage/xai.js
Normal file
@@ -0,0 +1,326 @@
|
||||
/**
|
||||
* xAI (Grok) OAuth usage handler
|
||||
*
|
||||
* SuperGrok quota is split across two upstream surfaces (OAuth only):
|
||||
*
|
||||
* 1) Monthly / API usage allotment (JSON)
|
||||
* GET https://cli-chat-proxy.grok.com/v1/billing
|
||||
* {
|
||||
* "config": {
|
||||
* "monthlyLimit": { "val": 15000 },
|
||||
* "used": { "val": 733 },
|
||||
* "onDemandCap": { "val": 0 },
|
||||
* "billingPeriodStart": "...",
|
||||
* "billingPeriodEnd": "..."
|
||||
* }
|
||||
* }
|
||||
*
|
||||
* 2) Weekly SuperGrok limit (grpc-web protobuf)
|
||||
* POST https://grok.com/grok_api_v2.GrokBuildBilling/GetGrokCreditsConfig
|
||||
* Empty request frame; response message1 contains:
|
||||
* usedPercent (float32), window start/end timestamps, nested windows.
|
||||
* This is what the grok.com usage page labels "Weekly limit" / "Resets …".
|
||||
*
|
||||
* Plan label comes from cli-chat-proxy settings:
|
||||
* GET https://cli-chat-proxy.grok.com/v1/settings → subscription_tier_display
|
||||
*
|
||||
* Note: grok.com/rest/rate-limits is a short chat-window (e.g. 2h query count)
|
||||
* behind Cloudflare browser cookies — not usable with pure OAuth bearer.
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U, parseResetTime, toFiniteNumber } from "./shared.js";
|
||||
|
||||
// Empty grpc-web request frame: flag(0) + length(0) + no payload.
|
||||
const GRPC_WEB_EMPTY_FRAME = Buffer.from([0x00, 0x00, 0x00, 0x00, 0x00]);
|
||||
|
||||
function moneyVal(wrapper) {
|
||||
if (wrapper == null) return null;
|
||||
if (typeof wrapper === "number") return toFiniteNumber(wrapper, null);
|
||||
if (typeof wrapper === "object" && wrapper.val != null) {
|
||||
return toFiniteNumber(wrapper.val, null);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function authHeaders(accessToken, extra = {}) {
|
||||
return {
|
||||
Authorization: `Bearer ${accessToken}`,
|
||||
...extra,
|
||||
};
|
||||
}
|
||||
|
||||
function readVarint(buf, offset) {
|
||||
let val = 0;
|
||||
let shift = 0;
|
||||
let i = offset;
|
||||
while (i < buf.length) {
|
||||
const b = buf[i++];
|
||||
val |= (b & 0x7f) << shift;
|
||||
if ((b & 0x80) === 0) return { value: val >>> 0, offset: i };
|
||||
shift += 7;
|
||||
if (shift > 35) break;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal protobuf decoder for GetGrokCreditsConfig.
|
||||
* Only understands varint / fixed32 / fixed64 / length-delimited.
|
||||
*/
|
||||
function parseProtobufFields(buf) {
|
||||
const fields = [];
|
||||
let i = 0;
|
||||
while (i < buf.length) {
|
||||
const key = readVarint(buf, i);
|
||||
if (!key) break;
|
||||
i = key.offset;
|
||||
const field = key.value >>> 3;
|
||||
const wt = key.value & 7;
|
||||
|
||||
if (wt === 0) {
|
||||
const v = readVarint(buf, i);
|
||||
if (!v) break;
|
||||
i = v.offset;
|
||||
fields.push({ field, type: "varint", value: v.value });
|
||||
} else if (wt === 1) {
|
||||
if (i + 8 > buf.length) break;
|
||||
fields.push({ field, type: "fixed64", value: buf.subarray(i, i + 8) });
|
||||
i += 8;
|
||||
} else if (wt === 5) {
|
||||
if (i + 4 > buf.length) break;
|
||||
fields.push({
|
||||
field,
|
||||
type: "fixed32",
|
||||
value: buf.readFloatLE(i),
|
||||
});
|
||||
i += 4;
|
||||
} else if (wt === 2) {
|
||||
const ln = readVarint(buf, i);
|
||||
if (!ln) break;
|
||||
i = ln.offset;
|
||||
if (i + ln.value > buf.length) break;
|
||||
fields.push({
|
||||
field,
|
||||
type: "bytes",
|
||||
value: buf.subarray(i, i + ln.value),
|
||||
});
|
||||
i += ln.value;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return fields;
|
||||
}
|
||||
|
||||
function parseTimestamp(bytes) {
|
||||
if (!bytes || !bytes.length) return null;
|
||||
const fields = parseProtobufFields(bytes);
|
||||
const seconds = fields.find((f) => f.field === 1 && f.type === "varint")?.value;
|
||||
if (!Number.isFinite(seconds) || seconds <= 0) return null;
|
||||
return new Date(seconds * 1000).toISOString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse grpc-web response bytes from GetGrokCreditsConfig.
|
||||
* Returns { usedPercent, resetAt, periodStart } or null.
|
||||
*/
|
||||
export function parseGrokCreditsConfig(raw) {
|
||||
if (!raw || !raw.length) return null;
|
||||
const buf = Buffer.isBuffer(raw) ? raw : Buffer.from(raw);
|
||||
|
||||
// grpc-web data frame: 1-byte flag + 4-byte big-endian length + message
|
||||
if (buf.length < 5) return null;
|
||||
const flag = buf[0];
|
||||
// Data frames have flag 0; ignore trailer frames (flag 0x80).
|
||||
if (flag !== 0) return null;
|
||||
const msgLen = buf.readUInt32BE(1);
|
||||
if (msgLen <= 0 || 5 + msgLen > buf.length) return null;
|
||||
const msg = buf.subarray(5, 5 + msgLen);
|
||||
|
||||
// Response is typically { 1: CreditsConfig }
|
||||
const top = parseProtobufFields(msg);
|
||||
const configBytes = top.find((f) => f.field === 1 && f.type === "bytes")?.value || msg;
|
||||
const fields = parseProtobufFields(configBytes);
|
||||
|
||||
const usedPercentRaw = fields.find((f) => f.field === 1 && f.type === "fixed32")?.value;
|
||||
const periodStart = parseTimestamp(fields.find((f) => f.field === 4 && f.type === "bytes")?.value);
|
||||
const periodEnd = parseTimestamp(fields.find((f) => f.field === 5 && f.type === "bytes")?.value);
|
||||
|
||||
if (usedPercentRaw == null || !Number.isFinite(usedPercentRaw)) return null;
|
||||
|
||||
const usedPercent = Math.max(0, Math.min(100, usedPercentRaw));
|
||||
return {
|
||||
usedPercent,
|
||||
remainingPercent: Math.max(0, 100 - usedPercent),
|
||||
periodStart,
|
||||
resetAt: periodEnd,
|
||||
};
|
||||
}
|
||||
|
||||
async function fetchBilling(accessToken, billingUrl, proxyOptions) {
|
||||
const response = await proxyAwareFetch(billingUrl, {
|
||||
method: "GET",
|
||||
headers: authHeaders(accessToken, { Accept: "application/json" }),
|
||||
}, proxyOptions);
|
||||
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
return { error: "auth", status: response.status };
|
||||
}
|
||||
if (!response.ok) {
|
||||
return { error: "http", status: response.status };
|
||||
}
|
||||
|
||||
const data = await response.json().catch(() => null);
|
||||
if (!data || typeof data !== "object") {
|
||||
return { error: "json" };
|
||||
}
|
||||
return { data };
|
||||
}
|
||||
|
||||
async function fetchWeeklyCredits(accessToken, creditsUrl, proxyOptions) {
|
||||
if (!creditsUrl) return null;
|
||||
try {
|
||||
const response = await proxyAwareFetch(creditsUrl, {
|
||||
method: "POST",
|
||||
headers: authHeaders(accessToken, {
|
||||
"Content-Type": "application/grpc-web+proto",
|
||||
"x-grpc-web": "1",
|
||||
"x-user-agent": "connect-es/2.1.1",
|
||||
Accept: "*/*",
|
||||
Origin: "https://grok.com",
|
||||
Referer: "https://grok.com/?_s=usage",
|
||||
}),
|
||||
body: GRPC_WEB_EMPTY_FRAME,
|
||||
}, proxyOptions);
|
||||
|
||||
if (!response.ok) return null;
|
||||
const ab = await response.arrayBuffer();
|
||||
return parseGrokCreditsConfig(Buffer.from(ab));
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchPlanLabel(accessToken, settingsUrl, proxyOptions) {
|
||||
if (!settingsUrl) return null;
|
||||
try {
|
||||
const response = await proxyAwareFetch(settingsUrl, {
|
||||
method: "GET",
|
||||
headers: authHeaders(accessToken, { Accept: "application/json" }),
|
||||
}, proxyOptions);
|
||||
if (!response.ok) return null;
|
||||
const data = await response.json().catch(() => null);
|
||||
const label = data?.subscription_tier_display;
|
||||
return typeof label === "string" && label.trim() ? label.trim() : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} accessToken - xAI OAuth access token
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getXaiUsage(accessToken, proxyOptions = null) {
|
||||
if (!accessToken) {
|
||||
return { message: "xAI usage unavailable: no access token. Re-authorize the connection." };
|
||||
}
|
||||
|
||||
const cfg = U("xai") || {};
|
||||
const billingUrl = cfg.url;
|
||||
if (!billingUrl) {
|
||||
return { message: "xAI usage endpoint is not configured." };
|
||||
}
|
||||
|
||||
try {
|
||||
const [billingResult, weekly, planLabel] = await Promise.all([
|
||||
fetchBilling(accessToken, billingUrl, proxyOptions),
|
||||
fetchWeeklyCredits(accessToken, cfg.creditsUrl, proxyOptions),
|
||||
fetchPlanLabel(accessToken, cfg.settingsUrl, proxyOptions),
|
||||
]);
|
||||
|
||||
if (billingResult.error === "auth") {
|
||||
return { message: "xAI OAuth token expired or unauthorized. Please re-authorize." };
|
||||
}
|
||||
|
||||
const quotas = {};
|
||||
let periodStart = null;
|
||||
let periodEnd = null;
|
||||
let onDemandCap = 0;
|
||||
|
||||
if (billingResult.data) {
|
||||
const config =
|
||||
billingResult.data.config && typeof billingResult.data.config === "object"
|
||||
? billingResult.data.config
|
||||
: billingResult.data;
|
||||
const monthlyLimit = moneyVal(config.monthlyLimit);
|
||||
const used = moneyVal(config.used);
|
||||
onDemandCap = moneyVal(config.onDemandCap) ?? 0;
|
||||
periodEnd = parseResetTime(config.billingPeriodEnd);
|
||||
periodStart = parseResetTime(config.billingPeriodStart);
|
||||
|
||||
// Absolute credit counts — do NOT put remaining credits on `remaining`
|
||||
// (QuotaTable treats remaining as a 0-100 percentage; same pitfall as Qoder).
|
||||
if (monthlyLimit != null && monthlyLimit > 0) {
|
||||
const usedSafe = Math.max(0, used ?? 0);
|
||||
quotas.api_usage = {
|
||||
used: usedSafe,
|
||||
total: monthlyLimit,
|
||||
remainingCredits: Math.max(0, monthlyLimit - usedSafe),
|
||||
unit: "credits",
|
||||
resetAt: periodEnd,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
if (onDemandCap > 0) {
|
||||
quotas.on_demand = {
|
||||
used: 0,
|
||||
total: onDemandCap,
|
||||
remainingCredits: onDemandCap,
|
||||
unit: "credits",
|
||||
resetAt: periodEnd,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Weekly SuperGrok window — percentage-based like Claude/Codex windows.
|
||||
if (weekly) {
|
||||
quotas.weekly = {
|
||||
used: weekly.usedPercent,
|
||||
total: 100,
|
||||
remaining: weekly.remainingPercent,
|
||||
remainingPercentage: weekly.remainingPercent,
|
||||
resetAt: weekly.resetAt || null,
|
||||
unlimited: false,
|
||||
};
|
||||
if (!periodStart && weekly.periodStart) periodStart = weekly.periodStart;
|
||||
}
|
||||
|
||||
if (Object.keys(quotas).length === 0) {
|
||||
const statusHint =
|
||||
billingResult.error === "http"
|
||||
? ` Billing API temporarily unavailable (${billingResult.status}).`
|
||||
: "";
|
||||
return {
|
||||
plan: planLabel || "xAI",
|
||||
message: `xAI connected. No quota allotment reported for this account.${statusHint}`,
|
||||
periodStart,
|
||||
periodEnd,
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
plan: planLabel || "xAI",
|
||||
periodStart,
|
||||
periodEnd,
|
||||
onDemandCap,
|
||||
quotas,
|
||||
};
|
||||
} catch (error) {
|
||||
return { message: `xAI connected. Unable to fetch billing: ${error.message}` };
|
||||
}
|
||||
}
|
||||
@@ -11,9 +11,6 @@
|
||||
export const QODER_OPENAPI_BASE = "https://openapi.qoder.sh";
|
||||
export const QODER_CENTER_BASE = "https://center.qoder.sh";
|
||||
export const QODER_CHAT_BASE = "https://api3.qoder.sh";
|
||||
// Job-token (jt-...) traffic is rejected by api3 with "Login expired" (403);
|
||||
// the official qodercli serves it from api2 instead.
|
||||
export const QODER_CHAT_BASE_ALT = "https://api2.qoder.sh";
|
||||
|
||||
export const QODER_LOGIN_URL = "https://qoder.com/device/selectAccounts";
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user