Compare commits
363 Commits
gitea/feat
...
ddfa789a31
| Author | SHA1 | Date | |
|---|---|---|---|
| ddfa789a31 | |||
| 6f52d7020c | |||
| 59f17b3725 | |||
| a835771c97 | |||
|
|
eb712ca821 | ||
|
|
11222eff0f | ||
|
|
e214fb1c30 | ||
|
|
f615a83cb2 | ||
|
|
e74db4d0a6 | ||
|
|
77e6a227fe | ||
|
|
1442cc73ce | ||
|
|
fb9fab0206 | ||
|
|
28cfd9facf | ||
|
|
1a3d446831 | ||
|
|
97f3ab97b1 | ||
|
|
0da803eef4 | ||
|
|
81f4f93082 | ||
|
|
cec672d9d9 | ||
|
|
ed963931b4 | ||
|
|
f388b5e56b | ||
|
|
b84681d5a4 | ||
|
|
2ab6a4c949 | ||
|
|
c08efdbe2b | ||
|
|
4eda76e2ab | ||
|
|
e0ffc7e2a1 | ||
|
|
6ab9ca9eb1 | ||
|
|
6efb97904b | ||
|
|
831001c322 | ||
|
|
e7dd72a8d7 | ||
|
|
5caa72f5fb | ||
|
|
e014cb537f | ||
|
|
d1d4e0f02b | ||
|
|
ac98dd9d32 | ||
|
|
a58902e4a7 | ||
|
|
98579f98c1 | ||
|
|
15687d1913 | ||
|
|
acb5c34cdc | ||
|
|
b870b5d41b | ||
|
|
1f190bd00b | ||
|
|
70f15aa50b | ||
|
|
1fe996db6a | ||
|
|
c24a854278 | ||
|
|
1fc2a81d65 | ||
|
|
f6c59d30b0 | ||
|
|
ac9120fde3 | ||
|
|
ee7a961633 | ||
|
|
f68d2f5ee5 | ||
|
|
ed1bd0c528 | ||
|
|
925cb4aade | ||
|
|
b9c92cb83c | ||
|
|
44e4b80bbe | ||
|
|
9d3f7646d1 | ||
|
|
009cac6326 | ||
|
|
38f031f4c9 | ||
|
|
90b52e06ff | ||
|
|
2203cd8f2b | ||
|
|
2fd99eae5d | ||
|
|
df85e16d7a | ||
|
|
dff648496c | ||
|
|
88676b3037 | ||
|
|
bb3cb43e09 | ||
|
|
4a371d1d9f | ||
|
|
d91e8b85e0 | ||
|
|
993c6eb469 | ||
|
|
28d005772a | ||
|
|
2a9213c5bd | ||
|
|
ec6692808b | ||
|
|
e5a13c3ab7 | ||
|
|
67d9182e1a | ||
|
|
9dbdca0e5e | ||
|
|
5a86f6a8d2 | ||
|
|
eb312bd470 | ||
|
|
fcfcced4ab | ||
|
|
56a40765e9 | ||
|
|
cadef6c4ff | ||
|
|
ab044e6d6d | ||
|
|
14401c433c | ||
|
|
e08ac6dada | ||
|
|
f9d82c6575 | ||
|
|
2f17352cc2 | ||
|
|
90a0005845 | ||
|
|
e79ae6e7c5 | ||
|
|
a68ada1c83 | ||
|
|
c4af43faa3 | ||
|
|
d7f7d70dd5 | ||
|
|
9c45b27cd7 | ||
|
|
e6f5724b4b | ||
|
|
d01724556a | ||
|
|
40eed18688 | ||
|
|
0532f00d84 | ||
|
|
9c650e1d54 | ||
|
|
1a3db1efae | ||
|
|
f0a6d35818 | ||
|
|
548e32aacf | ||
|
|
abb20d9f39 | ||
| 86112cee6d | |||
| 5c6048759c | |||
| d0f202a75d | |||
| f0adfb205a | |||
| 1d56e2dbc5 | |||
| 55f10c11e5 | |||
| eedad6c5ea | |||
| 99752a397c | |||
| bb8d67ba9c | |||
| 144dda2ac2 | |||
| bc9719fac7 | |||
| 6770f6ba0b | |||
| c61dc6de46 | |||
| b5c0f10610 | |||
| 1256f29d92 | |||
| de9e00c66d | |||
|
|
699edac327 | ||
|
|
540ebbe682 | ||
|
|
e1115e2839 | ||
|
|
27f3710c8b | ||
|
|
59d858b639 | ||
|
|
92259214db | ||
|
|
b04c03c6b5 | ||
|
|
8b2b2fefb5 | ||
|
|
86694ed8d0 | ||
|
|
8af5e752da | ||
|
|
8ed9da7165 | ||
|
|
7e5f5a8813 | ||
|
|
345cdcf6a5 | ||
|
|
65197ad11c | ||
|
|
e02bde4a70 | ||
|
|
30fec4318e | ||
|
|
67271d859e | ||
|
|
b566b20ade | ||
|
|
6d30ce6de5 | ||
|
|
5b417f9bf2 | ||
|
|
b57c041345 | ||
|
|
8a527fec91 | ||
|
|
70ba0024b0 | ||
|
|
80afb59907 | ||
|
|
10a923da11 | ||
|
|
01858feca0 | ||
|
|
e2a4fe048f | ||
|
|
b44bb09f72 | ||
|
|
456f2a2635 | ||
|
|
cd4003bc8b | ||
|
|
71dcdc1053 | ||
| a3182a7265 | |||
| 386b25ff7f | |||
|
|
15223724c3 | ||
|
|
35f86e5828 | ||
|
|
41588bea01 | ||
|
|
03f8487cc7 | ||
|
|
99639c0540 | ||
|
|
e41d85037d | ||
|
|
02c66fe2bd | ||
|
|
dcdd4628b3 | ||
|
|
6498b3122f | ||
|
|
8e59093db7 | ||
|
|
cd13d904d7 | ||
|
|
fe547f4dc0 | ||
|
|
b480892952 | ||
|
|
3fab15ae3e | ||
|
|
d06e0d26c6 | ||
|
|
b11be8be0a | ||
|
|
42c691b3ea | ||
|
|
a7941ddab4 | ||
|
|
646b3b9ba3 | ||
|
|
baebc9a06e | ||
|
|
25e4bf1c6c | ||
|
|
c570fe33ae | ||
|
|
d0751bcff7 | ||
|
|
948dd8f89b | ||
|
|
86131b9ca4 | ||
|
|
651df2f0e2 | ||
|
|
da8691f866 | ||
|
|
13ed14568d | ||
|
|
1eb37db32d | ||
|
|
d433c0b295 | ||
|
|
3292dfc102 | ||
|
|
9138c99391 | ||
|
|
41606a37a3 | ||
|
|
2abe8b855c | ||
|
|
c06cc08453 | ||
|
|
d6df6576c5 | ||
|
|
786b3013ba | ||
|
|
0648e9e420 | ||
|
|
ae4f76c433 | ||
|
|
918b3c87a1 | ||
|
|
0e5da70cb1 | ||
| 2a37a4085e | |||
| e2f8323ab1 | |||
| 9bd7adc556 | |||
| 34a78f579e | |||
| c8b96a61e7 | |||
| e438a03f96 | |||
| 008e0ef311 | |||
| 5058a402f1 | |||
| 9b27ee2611 | |||
| c5ce1ef140 | |||
| fcd3dcb409 | |||
| 067f18aaa1 | |||
|
|
f260a1817b | ||
| 88faba150a | |||
| 0dbae80930 | |||
|
|
6fcd27337a | ||
|
|
9be6588cc8 | ||
|
|
1319dea620 | ||
|
|
31df0635aa | ||
|
|
baf3356583 | ||
|
|
24fd165b0d | ||
|
|
a8313cd322 | ||
|
|
f8e8039446 | ||
|
|
e3e3e235f6 | ||
|
|
0afe949387 | ||
|
|
5e59790824 | ||
|
|
16cb40fda1 | ||
|
|
44c7b34837 | ||
|
|
15dfd86416 | ||
|
|
8b0fcf4b16 | ||
|
|
6d96e24bd9 | ||
|
|
3b14bf4a49 | ||
|
|
9c9dd7b191 | ||
|
|
f17a68aaee | ||
|
|
6eaa9f8369 | ||
|
|
65ac9b3cec | ||
|
|
72ec06a81d | ||
|
|
de2da19a9e | ||
|
|
aa0448f7e2 | ||
|
|
41c9e6be87 | ||
|
|
8e04fe1734 | ||
|
|
783e271c16 | ||
|
|
3c17d3406b | ||
|
|
007d372724 | ||
|
|
e45bd73d6e | ||
|
|
c85a5c57ba | ||
|
|
57b3b2c175 | ||
|
|
53a8b5ed55 | ||
|
|
039c4dbc72 | ||
|
|
79918c7830 | ||
|
|
6994cd1f70 | ||
|
|
4f48ab8c7f | ||
|
|
c97963c4fb | ||
|
|
cef5dd4d61 | ||
|
|
d587b2a487 | ||
|
|
7c7fae3955 | ||
|
|
9ba8f37486 | ||
|
|
55628eea02 | ||
|
|
c4a120af8f | ||
|
|
eb00222c4f | ||
|
|
43d4abbcf2 | ||
|
|
e0ba667450 | ||
|
|
0513bf393f | ||
| d826e39008 | |||
| 2897cc3972 | |||
|
|
ccb0842d0a | ||
|
|
68566f53dc | ||
|
|
bc252ea802 | ||
|
|
de680e789f | ||
|
|
6acc3bb965 | ||
|
|
8b9cac180e | ||
|
|
30d0f6d3d8 | ||
|
|
59b7828237 | ||
|
|
d6761c6fb0 | ||
|
|
02ccdc2d22 | ||
|
|
9c58ba645e | ||
|
|
70e8dc4974 | ||
|
|
b94685b80d | ||
|
|
0248dd5348 | ||
|
|
27b37705b3 | ||
|
|
7dfb346667 | ||
|
|
a6a41dfb3c | ||
|
|
2629218b04 | ||
|
|
c9926897ba | ||
|
|
88a8c72d2d | ||
|
|
ba508f2506 | ||
|
|
a077ee85bd | ||
|
|
e567ba800f | ||
|
|
542a088c04 | ||
|
|
9173c29b66 | ||
|
|
eceac9d7ae | ||
| ab9a3c1d43 | |||
|
|
837cfec5a9 | ||
| b1d368d960 | |||
|
|
f89ba32d79 | ||
|
|
9845a1702f | ||
|
|
a625ea9fd8 | ||
|
|
b61c50cbb7 | ||
|
|
baafc74c1f | ||
|
|
bb314118f2 | ||
|
|
2d515c8abc | ||
|
|
74d5fedf79 | ||
|
|
dcf1927f22 | ||
|
|
e1f3399b73 | ||
|
|
f1f9d27061 | ||
|
|
90df008f0c | ||
|
|
d2599ebf17 | ||
|
|
5cdcf67484 | ||
|
|
b25e10160d | ||
|
|
0270f6ea70 | ||
|
|
ce6bdf7fc2 | ||
|
|
a11937cdd6 | ||
|
|
c73c419d09 | ||
|
|
a3b267a5cb | ||
|
|
65c65a0f56 | ||
|
|
7610f28f42 | ||
|
|
0d4d4bc261 | ||
|
|
3a7a878f91 | ||
|
|
e79f9eddb4 | ||
|
|
b9e2611045 | ||
|
|
ddd5509e97 | ||
|
|
288940960a | ||
|
|
d75471bbbc | ||
|
|
0c55d49ab6 | ||
|
|
cfbdf06047 | ||
|
|
a4c5fa4e14 | ||
|
|
20b442b708 | ||
|
|
71cd5b2f23 | ||
|
|
b10b807063 | ||
|
|
081c6f2aff | ||
|
|
19281b5524 | ||
|
|
bbae990b92 | ||
|
|
8c068a1f5c | ||
|
|
97a6708651 | ||
|
|
a3cd7c82bc | ||
|
|
bf7da67859 | ||
|
|
da0149de97 | ||
|
|
1885ad7f64 | ||
|
|
481e7e467b | ||
|
|
008de32c06 | ||
|
|
b6454d84da | ||
|
|
46e6c01a01 | ||
|
|
5041494e1c | ||
|
|
4dadab9d5f | ||
|
|
7f436e2792 | ||
|
|
54e3245ace | ||
|
|
960f8a0379 | ||
|
|
5cc4f222f8 | ||
|
|
cd557a2552 | ||
|
|
ced51ed62f | ||
|
|
cb0135b695 | ||
|
|
abc0add031 | ||
|
|
9102c4c6d8 | ||
|
|
ce6120ce7b | ||
|
|
7afaecd617 | ||
|
|
a5363b83b5 | ||
|
|
b08751c4ea | ||
|
|
76752a4396 | ||
|
|
602ee4054b | ||
|
|
8f81f17b99 | ||
|
|
182c849979 | ||
|
|
0b3c794075 | ||
|
|
a9785a5f70 | ||
|
|
373850ee36 | ||
|
|
749c2e3f9c | ||
|
|
7fa2e7f029 | ||
|
|
8d1db46beb | ||
|
|
9e3866658a | ||
|
|
8a664d619d | ||
|
|
2d94fffe3b | ||
|
|
319caa2d7b | ||
|
|
95bfc64f06 | ||
|
|
eff81b1242 | ||
|
|
b66b5c68ce | ||
|
|
fc8722e897 | ||
|
|
713c563765 | ||
|
|
3d20a4ccd2 | ||
|
|
526235872a |
20
.gitignore
vendored
20
.gitignore
vendored
@@ -1,4 +1,5 @@
|
||||
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
|
||||
|
||||
# dependencies
|
||||
/node_modules
|
||||
/.pnp
|
||||
@@ -8,8 +9,10 @@
|
||||
!.yarn/plugins
|
||||
!.yarn/releases
|
||||
!.yarn/versions
|
||||
|
||||
# testing
|
||||
/coverage
|
||||
|
||||
# next.js
|
||||
/.next/
|
||||
/.next-cli-build/
|
||||
@@ -19,22 +22,28 @@ product
|
||||
# production
|
||||
/build
|
||||
.idea/
|
||||
|
||||
# misc
|
||||
.DS_Store
|
||||
*.pem
|
||||
|
||||
# debug
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
.pnpm-debug.log*
|
||||
|
||||
# env files (can opt-in for committing if needed)
|
||||
.env*
|
||||
!.env.example
|
||||
|
||||
# vercel
|
||||
.vercel
|
||||
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
next-env.d.ts
|
||||
|
||||
.bin/*
|
||||
data/
|
||||
logs/*
|
||||
@@ -52,18 +61,23 @@ Thanks.md
|
||||
PUBLIC.en.md
|
||||
PR/*
|
||||
package-lock.json
|
||||
|
||||
|
||||
#Ignore vscode AI rules
|
||||
.github/instructions/codacy.instructions.md
|
||||
README1.md
|
||||
deploy*.sh
|
||||
ecosystem.config.*
|
||||
|
||||
scripts/agSniffer/*
|
||||
gitbooks/*
|
||||
gitbook/README.md
|
||||
|
||||
# Refactor backup reference (do not bundle/lint)
|
||||
open-sse.old/
|
||||
.graphifyignore
|
||||
graphify-out/*
|
||||
|
||||
# Local-only working dirs (notes, vendored repos, scripts, skills)
|
||||
.claude/
|
||||
.docs/
|
||||
@@ -72,8 +86,6 @@ graphify-out/*
|
||||
.codegraph/
|
||||
.PR/
|
||||
.next-analyze/*
|
||||
# CommandCode CLI local state (auth/taste/projects)
|
||||
.commandcode/
|
||||
|
||||
# Pi subagent run artifacts
|
||||
.pi-subagents/
|
||||
# Kiro local workspace state
|
||||
.kiro/
|
||||
167
CHANGELOG.md
167
CHANGELOG.md
@@ -1,3 +1,168 @@
|
||||
# v0.5.69 (2026-09-05)
|
||||
|
||||
## Features
|
||||
- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities
|
||||
- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`)
|
||||
- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys
|
||||
- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819)
|
||||
- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through
|
||||
- **CLI tools**: replace Copilot MITM with VS Code extension setup guide
|
||||
- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace
|
||||
|
||||
## Fixes
|
||||
- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792)
|
||||
- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813)
|
||||
- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797)
|
||||
- **Dashboard**: dynamic mode label for local/remote detection (#3801)
|
||||
- **Codex**: format reset credit API errors cleanly (#3778)
|
||||
- **Security**: guard cowork MCP tools probe against SSRF (#3783)
|
||||
- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800)
|
||||
- **Logger**: suppress noisy background token refresh logs
|
||||
- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory
|
||||
|
||||
# v0.5.65 (2026-09-03)
|
||||
|
||||
## Features
|
||||
- **Fetch**: add Ollama Cloud web fetch provider
|
||||
- **Gemini / Antigravity**: add Gemini 3.8 Flash support and bump IDE fingerprint to 2.11.0
|
||||
- **Claude**: add Claude Fable 5.1 support (adaptive thinking with `output_config.effort`), bump Claude Code fingerprint to 2.1.258 for new-model access
|
||||
- **Providers**: add client-side status filter (All / Active / Inactive / No connection) on the Providers dashboard; add max height and scroll for connection list
|
||||
- **Providers & Models**: streamline tokenrouter model catalog down to 22 flagship/newest models and add missing provider icons; refresh Codebuddy-CN catalog (add hy4-preview/hy3/glm-5.3/kimi-k3-1, drop EOL glm-5.0/glm-4.7)
|
||||
- **Models**: capability toggles (vision, reasoning) when adding custom models with upsert and live caps refresh
|
||||
- **CLI tools**: support saving and managing custom API key presets
|
||||
- **Quota**: add usage and rate-limit tracking for Groq via `x-ratelimit-*` headers
|
||||
- **i18n**: complete Indonesian translation (1391 keys)
|
||||
|
||||
## Fixes
|
||||
- **Security**: close SSRF guard bypasses in `ssrfGuard.js` (alternate IPv6 encodings, hostname trailing dots, wildcard DNS resolution check, safe redirect handling) (#3714)
|
||||
- **Model markers**: strip the `[1m]` context marker Claude Code appends to model names (`claude-opus-5[1m]`) preventing model resolution failures (#3690)
|
||||
- **Claude**: drop `server_tool_use` blocks carrying foreign IDs to avoid Anthropic 400 rejections; never anchor cache breakpoints on `defer_loading` tools (#3567)
|
||||
- **Antigravity**: strike-break optimistic quota readings that keep 429ing by blocking the connection+model pair for 15m after 3 strikes (#3681); preserve client identity on model catalog requests (#3414)
|
||||
- **Auth**: protect root `/responses` rewrite requiring API key validation in dashboardGuard
|
||||
- **Chat & Docker**: return 503 Service Unavailable when all credentials are rate-limited; explicitly bundle `node-machine-id` into standalone Docker runtime image
|
||||
- **OpenCode**: route Muse Spark models to `/zen/v1/responses` and declare vision support; filter inactive free model
|
||||
- **Kiro**: preserve inline images as OpenAI-compatible `image_url` parts in OpenAI MITM; remove redundant top-level `systemPrompt` from payload
|
||||
- **Usage**: read Responses-shape `cached_tokens` in `extractUsageFromResponse` for non-streaming traffic
|
||||
- **Models**: support single model lookup with provider-prefixed IDs (e.g. `cc/claude-sonnet-5`)
|
||||
- **Translator**: route Gemini thinking through `reasoning_effort` on OpenAI-compatible wire; convert `prefixItems` and ensure array items in Gemini schema sanitizer
|
||||
- **UI**: apply persisted theme before first paint to prevent flash on reload; translate combo vision adapter label
|
||||
|
||||
# v0.5.59 (2026-08-29)
|
||||
|
||||
## Features
|
||||
- **Search**: new web search providers — Antigravity (Google Search grounding
|
||||
on the existing OAuth account pool, citations keyed and merged by URL) and
|
||||
Xquik (X search with `x-api-key` auth, cursor pagination, credit-based
|
||||
usage), both on `POST /v1/search`. Based on #3437 by @Nautilaceae
|
||||
- **Search**: ollama-search and zai-search borrow a chat provider's API key
|
||||
instead of requiring their own connection, driven by a new
|
||||
`credentialFallback` registry field. zai-search later folded into the `glm`
|
||||
provider itself so the web search page shows the shared connection
|
||||
- **Models**: daily background sync of model capabilities from models.dev —
|
||||
modalities keyed by model id (majority of sources must declare one),
|
||||
context/output limits keyed by provider + model, strictly additive and
|
||||
sitting below the hand-written tables. ETag + mtime cache, 60s startup
|
||||
delay, `MODEL_CATALOG_SYNC=off` to disable
|
||||
- **Models**: add GLM-5.3-Flash (1M context, natively multimodal), DeepSeek
|
||||
V4 Vision, Grok 4.5/4.6 (500k context); correct glm-4.6v/4.5v video input
|
||||
and output limits, backfill glm-4.6v on glm-cn
|
||||
- **Usage**: show the Zed plan quota on the dashboard — plan, edit
|
||||
predictions, hosted model requests and billing-cycle reset; unlimited rows
|
||||
render as "N used · Unlimited"
|
||||
- **Usage**: track GPT-5.3-Codex-Spark quota windows (spark_session /
|
||||
spark_weekly) from the Codex usage response (#3431)
|
||||
- **Antigravity**: quota-aware routing — on 409/429 fetch live quota for the
|
||||
exact per-model resetAt and skip only the exhausted account/model pair;
|
||||
report the earliest reset when every account is blocked (#3561)
|
||||
- **Antigravity**: map image `size` to the aspect-ratio model suffix (-WxH);
|
||||
add the Gemini 3.7 Flash tiers to MITM defaultModels so they show up in
|
||||
the dashboard model-mapping table
|
||||
- **Dashboard**: bulk import Grok CLI accounts from JSON — paste an array or
|
||||
drag-drop multiple .json files, all OAuth connections created in a single
|
||||
call, mirroring the codex flow
|
||||
- **CLI tools**: endpoint presets shared across every tool card through one
|
||||
live-resyncing store, instead of per-card localStorage copies that never
|
||||
saw each other's saved endpoints
|
||||
- **Token Saver**: configurable compression timeout (`headroomTimeoutMs`) —
|
||||
the fixed 3000 ms made busy machines time out and send inconsistently
|
||||
compressed bodies, hurting prompt caching
|
||||
- **i18n**: pt-BR expanded to 1132 terms
|
||||
|
||||
## Fixes
|
||||
- **Claude Code**: add Claude Fable 5.1 and advertise Claude Code 2.1.258 in
|
||||
both the request header and billing identity; use its permanent adaptive-thinking
|
||||
mode with `output_config.effort`
|
||||
- **Stream**: record usage when a client closes on the terminal event — the
|
||||
Responses API has no [DONE] sentinel, so codex closed the socket on
|
||||
`response.completed` and cancelled the reader before flush() ran its usage
|
||||
side effects; the tail now lives in a once-guarded finalizeStream(). Also
|
||||
stop logging a disconnect for every completed Responses call
|
||||
- **Stream**: parse the trailing NDJSON line an Ollama stream leaves behind
|
||||
without a closing newline — the final chunk carrying `done_reason` and the
|
||||
token counts was dropped
|
||||
- **Session**: read the Claude Code session id from the
|
||||
`x-claude-code-session-id` header — `metadata.user_id` is dropped by
|
||||
Responses translation, splitting one conversation across several
|
||||
`prompt_cache_key` values and missing the upstream prefix cache
|
||||
- **Usage**: preserve nested `cached_tokens` — the top-level-only read
|
||||
persisted `cached_tokens: 0` for every Responses-format provider (codex,
|
||||
grok-cli, …), billing cache hits at the full input rate
|
||||
- **Usage**: GLM quotas accept CREDIT_LIMIT plans and multi-interval windows
|
||||
(5h session / 7d weekly) instead of overwriting a single "session" key
|
||||
- **Models**: the catalog sync no longer erases its own output — deltas were
|
||||
measured against the previous run's writes (the second run cut `providers`
|
||||
from 20 entries to 5); one vote per provider in the modality tally, ETag
|
||||
restored from file on startup, and the worker thread dropped after the
|
||||
bundler rewrote its path into a module-not-found error
|
||||
- **Executor**: CommandCode returns errors as a `type:"error"` event inside
|
||||
an HTTP 200 NDJSON stream — peek the first events before committing, abort
|
||||
and return a real 4xx/5xx so combo/account fallback triggers instead of
|
||||
streaming the error text as content
|
||||
- **Search**: scope failure locks on the credential-fallback path — a failing
|
||||
search locked `modelLock___all` and took the shared glm key offline for
|
||||
chat as well; locks are now attributed to the connection's owner and
|
||||
scoped to `websearch:<provider>`
|
||||
- **Providers**: connection tests get a 15s AbortSignal timeout instead of
|
||||
hanging and exhausting the browser socket pool; guard undefined provider
|
||||
names on the providers page
|
||||
- **Antigravity**: sanitize competing-client branding via a config-driven
|
||||
rule table (Zed's Claude-agent prompt, opencode → antigravity) — upstream
|
||||
answers 429 Quota Exhausted. Applied in the executor so the shared
|
||||
openai-to-gemini translator leaves gemini/vertex/zed untouched
|
||||
- **MiniMax**: preserve images on the sourceFormat-matched OpenAI transport
|
||||
— MiniMax-M3 resolved a Claude-shaped body posted to the OpenAI endpoint,
|
||||
silently dropping `image_url` blocks (#3418)
|
||||
- **Claude**: decloak tool names in same-format streaming passthrough —
|
||||
OAuth-cloaked names (CLAUDE_TOOL_SUFFIX) leaked to the client and every
|
||||
tool call was rejected as unknown
|
||||
- **Tools**: default a missing `tools[].type` to "custom" on Claude-format
|
||||
requests — strict Anthropic-compatible gateways (MiniMax) reject the
|
||||
request with 400 otherwise
|
||||
- **Translator**: zai thinkingFormat sends the top-level `reasoning_effort`
|
||||
object GLM-5.2+ requires — every GLM-5.x request ran at the model default
|
||||
(max); gated on GLM-5.2+ since older GLM does not read it (#2721)
|
||||
- **RTK**: system prompt injection matches each target wire format
|
||||
(Chat/Responses/Claude/Gemini/Kiro) and is exact-idempotent across retries,
|
||||
so distinct prompts sharing a long prefix are no longer collapsed (#3202).
|
||||
Also set the diagnostic before the silent null return on Responses
|
||||
translation failure so the panel is no longer blank
|
||||
- **OpenCode**: route muse-spark through /zen/v1/responses (it 500s on
|
||||
chat/completions), normalizing the Chat fields the Responses API rejects
|
||||
and clamping max/ultra effort to xhigh
|
||||
- **CLI**: install better-sqlite3 without build tools on Node 22+ (N-API
|
||||
13.0.3 ships per-platform prebuilds, `--ignore-scripts` skips the implicit
|
||||
node-gyp build); Node < 22 stays on 12.6.2, working installs untouched
|
||||
- **CLI tools**: send the API key Codex actually reads —
|
||||
`[model_providers.9router.http_headers]` instead of auth.json (which left
|
||||
every request 401 and clobbered an existing ChatGPT login); subagent model
|
||||
moved to `agents.default_subagent_model`
|
||||
- **OAuth**: refresh Cline tokens with the extension JSON contract
|
||||
- **Dashboard**: clamp the API key mask length — keys shorter than 8 chars
|
||||
threw RangeError and crashed the media-provider detail page
|
||||
- **UI**: wait for the Material Symbols font itself before revealing icons —
|
||||
`document.fonts.ready` resolved before the 4MB woff2 even started loading,
|
||||
leaving icons blank until a second load
|
||||
|
||||
# v0.5.55 (2026-08-14)
|
||||
|
||||
## Features
|
||||
@@ -625,4 +790,4 @@
|
||||
# v0.4.46 (2026-05-15)
|
||||
|
||||
## Breaking Changes
|
||||
- Tunnel public URL changed — old tunnel links no longer work, please reconnect to get the new URL
|
||||
- Tunnel public URL changed — old tunnel links no longer work, please reconnect to get the new URL
|
||||
|
||||
10
CLAUDE.md
10
CLAUDE.md
@@ -89,3 +89,13 @@ Pre-translate hooks that compress `tool_result` content in-place to cut tokens.
|
||||
- Security-sensitive env: `JWT_SECRET` (session cookie), `INITIAL_PASSWORD` (default `123456` — must override), `API_KEY_SECRET`, `MACHINE_ID_SALT`. Full env contract in `.env.example` and ARCHITECTURE.md's env matrix.
|
||||
- Binary/protobuf upstreams (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — they're handled inside their own executor, not the translator.
|
||||
- Versioning: root and `cli/` are versioned independently; changes are logged in `CHANGELOG.md`. Commit style is Conventional Commits (`fix(translator): …`, `feat(...)`).
|
||||
|
||||
<!-- BEGIN:nextjs-agent-rules -->
|
||||
|
||||
# This is NOT the Next.js you know
|
||||
|
||||
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` (resolved from this file's directory; in monorepos the `next` package may not be visible from the repo root) before writing any code. Heed deprecation notices.
|
||||
|
||||
This block is written and re-added by `next dev` — verify at `node_modules/next/dist/server/lib/generate-agent-files.js`. Removing it from a diff only re-creates the uncommitted change; committing it with your work keeps the tree clean.
|
||||
|
||||
<!-- END:nextjs-agent-rules -->
|
||||
|
||||
@@ -2,14 +2,15 @@
|
||||
ARG NODE_IMAGE=node:22-alpine
|
||||
FROM ${NODE_IMAGE} AS base
|
||||
WORKDIR /app
|
||||
# CN mirror for apk (used by builder and runner stages)
|
||||
RUN sed -i 's|dl-cdn.alpinelinux.org|mirrors.aliyun.com|g' /etc/apk/repositories
|
||||
|
||||
FROM base AS builder
|
||||
|
||||
RUN apk --no-cache upgrade && apk --no-cache add python3 make g++ linux-headers
|
||||
|
||||
COPY package.json ./
|
||||
RUN --mount=type=cache,target=/root/.npm \
|
||||
npm install
|
||||
RUN npm install --registry=https://registry.npmmirror.com
|
||||
|
||||
COPY . ./
|
||||
ENV NEXT_TELEMETRY_DISABLED=1
|
||||
@@ -40,6 +41,8 @@ COPY --from=builder /app/node_modules/next ./node_modules/next
|
||||
# sql.js loads dist/sql-wasm.wasm by path at runtime; tracing only follows JS imports,
|
||||
# so the last-resort DB driver would abort with ENOENT on the missing binary.
|
||||
COPY --from=builder /app/node_modules/sql.js ./node_modules/sql.js
|
||||
# node-machine-id is createRequire-loaded at runtime; tracing omits it.
|
||||
COPY --from=builder /app/node_modules/node-machine-id ./node_modules/node-machine-id
|
||||
|
||||
RUN mkdir -p /app/data && chown -R node:node /app && \
|
||||
mkdir -p /app/data-home && chown node:node /app/data-home && \
|
||||
|
||||
@@ -6,7 +6,13 @@ const fs = require("fs");
|
||||
const os = require("os");
|
||||
const path = require("path");
|
||||
|
||||
const BETTER_SQLITE3_VERSION = "12.6.2";
|
||||
// Gate the pinned version by Node major, mirroring src/lib/db/driver.js gating
|
||||
// style: 13.x is N-API and ships per-platform prebuilds inside the package, so
|
||||
// it needs no ABI-specific download. It requires Node >= 22; older runtimes stay
|
||||
// on 12.6.2, which fetches an ABI-specific binary via prebuild-install.
|
||||
const [NODE_MAJOR] = process.versions.node.split(".").map(Number);
|
||||
const USE_NAPI_BUILD = NODE_MAJOR >= 22;
|
||||
const BETTER_SQLITE3_VERSION = USE_NAPI_BUILD ? "13.0.3" : "12.6.2";
|
||||
const SQL_JS_VERSION = "1.14.1";
|
||||
|
||||
function getDataDir() {
|
||||
@@ -45,9 +51,23 @@ function hasModule(name) {
|
||||
return fs.existsSync(path.join(getRuntimeNodeModules(), name, "package.json"));
|
||||
}
|
||||
|
||||
function isGlibcRuntime() {
|
||||
try { return Boolean(process.report?.getReport()?.header?.glibcVersionRuntime); } catch { return true; }
|
||||
}
|
||||
|
||||
// 12.x compiles/downloads into build/Release; 13.x ships prebuilds/<platform>-<arch>.node.
|
||||
function getBetterSqliteBinary() {
|
||||
const root = path.join(getRuntimeNodeModules(), "better-sqlite3");
|
||||
const platform = process.platform === "linux" && !isGlibcRuntime() ? "linuxmusl" : process.platform;
|
||||
return [
|
||||
path.join(root, "build", "Release", "better_sqlite3.node"),
|
||||
path.join(root, "prebuilds", `${platform}-${process.arch}.node`),
|
||||
].find((file) => fs.existsSync(file));
|
||||
}
|
||||
|
||||
function isBetterSqliteBinaryValid() {
|
||||
const binary = path.join(getRuntimeNodeModules(), "better-sqlite3", "build", "Release", "better_sqlite3.node");
|
||||
if (!fs.existsSync(binary)) return false;
|
||||
const binary = getBetterSqliteBinary();
|
||||
if (!binary) return false;
|
||||
try {
|
||||
const fd = fs.openSync(binary, "r");
|
||||
const buf = Buffer.alloc(4);
|
||||
@@ -91,6 +111,7 @@ function runNpmInstall({ cwd, pkgs, extraArgs = [], timeout = 180000 }) {
|
||||
function npmInstall(pkgs, opts = {}) {
|
||||
const cwd = ensureRuntimeDir();
|
||||
const extra = opts.optional ? ["--no-save"] : [];
|
||||
if (opts.ignoreScripts) extra.push("--ignore-scripts");
|
||||
if (!opts.silent) console.log("⏳ Installing SQLite engine (first run)...");
|
||||
const res = runNpmInstall({ cwd, pkgs, extraArgs: extra, timeout: opts.timeout || 180000 });
|
||||
if (!res.ok && !opts.silent) {
|
||||
@@ -129,7 +150,10 @@ function ensureSqliteRuntime({ silent = false } = {}) {
|
||||
return { betterSqlite: true, sqlJs: sqlJsOk };
|
||||
}
|
||||
|
||||
const ok = npmInstall([`better-sqlite3@${BETTER_SQLITE3_VERSION}`], { optional: true, silent });
|
||||
// npm injects an implicit `node-gyp rebuild` for any package carrying a
|
||||
// binding.gyp, which would demand build tools even though 13.x already bundles
|
||||
// the binary — skip scripts so the bundled prebuild is used as-is.
|
||||
const ok = npmInstall([`better-sqlite3@${BETTER_SQLITE3_VERSION}`], { optional: true, silent, ignoreScripts: USE_NAPI_BUILD });
|
||||
return {
|
||||
betterSqlite: ok && hasModule("better-sqlite3") && isBetterSqliteBinaryValid(),
|
||||
sqlJs: sqlJsOk,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "9router",
|
||||
"version": "0.5.55",
|
||||
"version": "0.5.69",
|
||||
"description": "9Router CLI - Start and manage 9Router server",
|
||||
"bin": {
|
||||
"9router": "./cli.js"
|
||||
@@ -16,7 +16,7 @@
|
||||
"scripts": {
|
||||
"dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js",
|
||||
"build": "node scripts/build-cli.js",
|
||||
"pack:cli": "npm run build && npm pack --pack-destination ../..",
|
||||
"pack:cli": "npm run build && npm pack --pack-destination ..",
|
||||
"publish:cli": "npm run build && npm publish",
|
||||
"postinstall": "node hooks/postinstall.js",
|
||||
"prepublishOnly": "npm run build"
|
||||
|
||||
@@ -53,6 +53,9 @@ const PROVIDER_MODELS = {
|
||||
{ id: "glm-4.7" },
|
||||
],
|
||||
ag: [
|
||||
{ id: "gemini-3.8-flash-high" },
|
||||
{ id: "gemini-3.8-flash-medium" },
|
||||
{ id: "gemini-3.8-flash-low" },
|
||||
{ id: "gemini-3.7-flash-high" },
|
||||
{ id: "gemini-3.7-flash-medium" },
|
||||
{ id: "gemini-3.7-flash-low" },
|
||||
@@ -101,6 +104,8 @@ const PROVIDER_MODELS = {
|
||||
{ id: "claude-3-5-sonnet-20241022" },
|
||||
],
|
||||
gemini: [
|
||||
{ id: "gemini-3.8-flash" },
|
||||
{ id: "gemini-3.7-flash" },
|
||||
{ id: "gemini-3.6-flash" },
|
||||
{ id: "gemini-3.5-flash-lite" },
|
||||
{ id: "gemini-3-pro-preview" },
|
||||
|
||||
261
docs/superpowers/plans/2026-09-04-opencode-go-session-header.md
Normal file
261
docs/superpowers/plans/2026-09-04-opencode-go-session-header.md
Normal file
@@ -0,0 +1,261 @@
|
||||
# OpenCode Go Session Header Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Send a stable, conversation-scoped `x-opencode-session` header on every OpenCode Go request and install the patched CLI locally.
|
||||
|
||||
**Architecture:** Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. `chatCore` passes the provider-scoped session resolved from the original request plus the detected client tool; the executor derives a request-local upstream session and delegates all existing transport, authentication, retry, and proxy behavior to `DefaultExecutor`.
|
||||
|
||||
**Tech Stack:** Node.js ESM, Vitest, Next.js, npm CLI packaging, GitHub CLI.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Apply the header to OpenCode Go chat completions, Claude Messages, and OpenAI Responses transports.
|
||||
- Preserve a valid native `x-opencode-session`; hash all translated non-OpenCode identities to `ses_<32 lowercase hex>`.
|
||||
- Namespace translated identities by detected client tool, using `generic` when unknown.
|
||||
- Do not keep mutable per-request session state on the executor singleton or mutate the caller's credentials object.
|
||||
- Do not change OpenCode Go models, routing, reasoning, tool behavior, dependencies, or unrelated providers.
|
||||
- Reuse upstream issue #3759 instead of creating a duplicate issue.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Add Failing OpenCode Go Session Tests
|
||||
|
||||
**Files:**
|
||||
- Create: `tests/unit/opencode-go-session.test.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getExecutor(provider)` and `DefaultExecutor.buildHeaders(credentials, stream, url, model)`.
|
||||
- Produces: the required public behavior for `OpenCodeGoExecutor.prepareRequestCredentials({ body, credentials, providerSessionId, clientTool })` and `OpenCodeGoExecutor.execute(args)`.
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
Create a Vitest suite that mocks `proxyAwareFetch`, obtains `getExecutor("opencode-go")`, and asserts:
|
||||
|
||||
```js
|
||||
const prepared = executor.prepareRequestCredentials({
|
||||
body: { messages: [{ role: "user", content: "hello" }] },
|
||||
credentials: { apiKey: "test-key", connectionId: "conn-a", rawHeaders: {} },
|
||||
providerSessionId: "conversation-a",
|
||||
clientTool: "claude",
|
||||
});
|
||||
|
||||
expect(prepared).not.toBe(credentials);
|
||||
expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/);
|
||||
expect(credentials).not.toHaveProperty("_opencodeGoSession");
|
||||
```
|
||||
|
||||
Cover native header preservation, stable values across all three runtime transports, different conversation IDs, different client tools using the same ID, connection fallback, no singleton state, no header on `DefaultExecutor("openai")`, and the final fetch headers returned by `execute()`.
|
||||
|
||||
- [ ] **Step 2: Run the focused test and verify RED**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because `getExecutor("opencode-go")` still returns `DefaultExecutor` and `prepareRequestCredentials` does not exist.
|
||||
|
||||
- [ ] **Step 3: Commit the failing test**
|
||||
|
||||
```bash
|
||||
git add tests/unit/opencode-go-session.test.js
|
||||
git commit -m "test: cover OpenCode Go session headers"
|
||||
```
|
||||
|
||||
### Task 2: Implement the Dedicated Executor
|
||||
|
||||
**Files:**
|
||||
- Create: `open-sse/executors/opencode-go.js`
|
||||
- Modify: `open-sse/executors/index.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `DefaultExecutor`, `resolveSessionId()`, request `credentials.rawHeaders`, `providerSessionId`, and `clientTool`.
|
||||
- Produces: `OpenCodeGoExecutor`, `prepareRequestCredentials()`, and an `execute()` override that delegates with cloned credentials.
|
||||
|
||||
- [ ] **Step 1: Add the minimal executor implementation**
|
||||
|
||||
Implement these rules:
|
||||
|
||||
```js
|
||||
function translatedSessionId(sessionId, clientTool) {
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
|
||||
.digest("hex")
|
||||
.slice(0, 32);
|
||||
return `ses_${digest}`;
|
||||
}
|
||||
```
|
||||
|
||||
`prepareRequestCredentials()` must read a case-insensitive native
|
||||
`x-opencode-session` with the same non-empty, 256-character cap used by the
|
||||
session manager. Otherwise it uses `providerSessionId` or calls
|
||||
`resolveSessionId({ headers, body, connectionId, scope: "opencode-go" })`, then
|
||||
returns `{ ...credentials, _opencodeGoSession: value }`.
|
||||
|
||||
`execute(args)` must call `prepareRequestCredentials(args)` and delegate using
|
||||
`super.execute({ ...args, credentials: prepared })`. `buildHeaders()` must call
|
||||
`super.buildHeaders()` and add the prepared session, with a connection-scoped
|
||||
fallback for direct callers.
|
||||
|
||||
Register `new OpenCodeGoExecutor()` under `"opencode-go"` and export the class.
|
||||
|
||||
- [ ] **Step 2: Run the focused test and verify partial GREEN**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
|
||||
```
|
||||
|
||||
Expected: executor-level tests pass; any chatCore-context assertion remains failing until Task 3.
|
||||
|
||||
- [ ] **Step 3: Commit the executor**
|
||||
|
||||
```bash
|
||||
git add open-sse/executors/opencode-go.js open-sse/executors/index.js tests/unit/opencode-go-session.test.js
|
||||
git commit -m "fix(opencode-go): add stable session header executor"
|
||||
```
|
||||
|
||||
### Task 3: Pass Original Request Session Context
|
||||
|
||||
**Files:**
|
||||
- Modify: `open-sse/handlers/chatCore.js`
|
||||
- Modify: `tests/unit/opencode-go-session.test.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: existing `sessionSeed` and `clientTool` variables in `handleChatCore()`.
|
||||
- Produces: `providerSessionId` and `clientTool` fields on both initial and refreshed-credential calls to `executor.execute()`.
|
||||
|
||||
- [ ] **Step 1: Add or enable the failing integration assertion**
|
||||
|
||||
Use a mocked executor or source request containing a body-only `session_id` and
|
||||
assert the executor receives the provider-scoped session resolved before
|
||||
translation.
|
||||
|
||||
- [ ] **Step 2: Run the focused test and verify RED**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because `handleChatCore()` does not pass `providerSessionId` or
|
||||
`clientTool` to `executor.execute()`.
|
||||
|
||||
- [ ] **Step 3: Pass the request context**
|
||||
|
||||
Add the same fields to both executor calls:
|
||||
|
||||
```js
|
||||
executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
providerSessionId: sessionSeed,
|
||||
clientTool,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run focused and neighboring tests**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js \
|
||||
tests/unit/opencode-go-session.test.js \
|
||||
tests/unit/opencode-go-models.test.js \
|
||||
tests/unit/session-manager.test.js \
|
||||
tests/unit/executor-const-guard.test.js
|
||||
```
|
||||
|
||||
Expected: PASS with zero failed tests.
|
||||
|
||||
- [ ] **Step 5: Commit the context wiring**
|
||||
|
||||
```bash
|
||||
git add open-sse/handlers/chatCore.js tests/unit/opencode-go-session.test.js
|
||||
git commit -m "fix(chat): forward provider session context"
|
||||
```
|
||||
|
||||
### Task 4: Verify and Install the Local CLI Package
|
||||
|
||||
**Files:**
|
||||
- Generated: `9router-0.5.65.tgz`
|
||||
- Packaged output: `cli/app/server.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: completed source changes and existing CLI build scripts.
|
||||
- Produces: a globally installed patched `9router@0.5.65`.
|
||||
|
||||
- [ ] **Step 1: Run source verification**
|
||||
|
||||
```bash
|
||||
git diff --check origin/master...HEAD
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/
|
||||
npm run build
|
||||
```
|
||||
|
||||
Expected: every command exits zero. Record any pre-existing full-suite failures
|
||||
separately rather than hiding them.
|
||||
|
||||
- [ ] **Step 2: Build and package the CLI**
|
||||
|
||||
```bash
|
||||
npm --prefix cli run build
|
||||
npm --prefix cli pack -- --pack-destination ..
|
||||
```
|
||||
|
||||
Expected: `9router-0.5.65.tgz` exists and contains the patched bundled server.
|
||||
|
||||
- [ ] **Step 3: Replace the global npm installation**
|
||||
|
||||
```bash
|
||||
npm install -g ./9router-0.5.65.tgz
|
||||
```
|
||||
|
||||
Expected: `/opt/homebrew/lib/node_modules/9router/package.json` reports `0.5.65`
|
||||
and the installed bundle contains `x-opencode-session` plus the new executor.
|
||||
|
||||
- [ ] **Step 4: Commit any required package-source adjustment**
|
||||
|
||||
Do not commit generated tarballs or CLI build artifacts unless the repository
|
||||
already tracks and requires them.
|
||||
|
||||
### Task 5: Publish the Upstream Pull Request
|
||||
|
||||
**Files:**
|
||||
- No additional source files unless verification finds a required correction.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: verified branch commits and GitHub issue #3759.
|
||||
- Produces: a fork branch and a PR against `decolua/9router:master`.
|
||||
|
||||
- [ ] **Step 1: Create or repair the GitHub fork remote**
|
||||
|
||||
Use `gh repo fork decolua/9router --remote` if the current `fork` remote remains
|
||||
missing, then push `fix/opencode-go-session-header`.
|
||||
|
||||
- [ ] **Step 2: Create the PR**
|
||||
|
||||
Use title:
|
||||
|
||||
```text
|
||||
fix(opencode-go): send stable session header
|
||||
```
|
||||
|
||||
The body must include the root cause, downstream-session translation policy,
|
||||
three covered transports, concurrency behavior, verification evidence,
|
||||
`Fixes #3759`, and a note that this PR is intentionally narrower than #3780.
|
||||
|
||||
- [ ] **Step 3: Verify the published PR**
|
||||
|
||||
Run `gh pr view --json number,title,state,url,headRefName,baseRefName` and report
|
||||
the issue and PR URLs.
|
||||
@@ -0,0 +1,114 @@
|
||||
# OpenCode Go Session Header Design
|
||||
|
||||
## Problem
|
||||
|
||||
OpenCode Go will begin rejecting some requests without an
|
||||
`x-opencode-session` header on September 6, 2026. In 9Router v0.5.65,
|
||||
`opencode-go` uses `DefaultExecutor`, whose generic header builder does not add
|
||||
that header. The specialized OpenCode Free executor already sends it, but that
|
||||
logic does not apply to the paid OpenCode Go provider or its three transports.
|
||||
|
||||
## Goals
|
||||
|
||||
- Add `x-opencode-session` to every OpenCode Go chat, Claude Messages, and
|
||||
OpenAI Responses request.
|
||||
- Translate a downstream conversation identity into a stable upstream identity.
|
||||
- Keep identities isolated across different downstream agents and conversations.
|
||||
- Avoid exposing non-OpenCode downstream session identifiers to OpenCode Go.
|
||||
- Avoid mutable session state on the shared executor singleton.
|
||||
- Leave OpenCode Free and all unrelated providers unchanged.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
- Inferring an exact conversation boundary when a downstream client provides no
|
||||
session or conversation identifier.
|
||||
- Adding or changing OpenCode Go models, routing, reasoning, or tool behavior.
|
||||
- Changing the general session-resolution policy for other providers.
|
||||
|
||||
## Architecture
|
||||
|
||||
Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. The executor
|
||||
keeps the existing generic URL, authentication, translation, retry, and proxy
|
||||
behavior, and overrides only the OpenCode Go session-header concern.
|
||||
|
||||
`handleChatCore` already resolves a provider-scoped session from the original
|
||||
request before translation. It will pass that value and the detected client
|
||||
tool to `executor.execute()` as request context. `OpenCodeGoExecutor.execute()`
|
||||
will create a shallow request-local credentials object containing the resolved
|
||||
OpenCode Go session. It will then delegate to `DefaultExecutor.execute()`.
|
||||
This avoids storing request state on the executor singleton or mutating shared
|
||||
provider credentials.
|
||||
|
||||
## Session Resolution
|
||||
|
||||
The original downstream request remains the source of truth. Existing
|
||||
`resolveSessionId()` behavior recognizes Claude Code, Antigravity, generic
|
||||
session headers, and common body fields before request translation can discard
|
||||
them.
|
||||
|
||||
Resolution rules:
|
||||
|
||||
1. If the downstream request supplies `x-opencode-session`, treat it as an
|
||||
authoritative OpenCode identity after trimming and length validation.
|
||||
2. Otherwise use the provider-scoped session resolved from the original request.
|
||||
3. Namespace the resolved value with the detected downstream agent, falling back
|
||||
to `generic` when the agent is unknown.
|
||||
4. Convert the namespaced value to an opaque deterministic identifier:
|
||||
`ses_` plus the first 32 hexadecimal characters of SHA-256.
|
||||
5. If no explicit downstream identity exists, the existing provider connection
|
||||
fallback guarantees that a header is still sent. It is stable but cannot
|
||||
distinguish multiple conversations sharing that connection.
|
||||
|
||||
The same input conversation produces the same upstream identifier for all three
|
||||
OpenCode Go transports. Different agents using the same raw session value
|
||||
produce different identifiers.
|
||||
|
||||
## Header Injection
|
||||
|
||||
`OpenCodeGoExecutor.buildHeaders()` delegates to
|
||||
`DefaultExecutor.buildHeaders()` and adds only:
|
||||
|
||||
```text
|
||||
x-opencode-session: <stable-session-id>
|
||||
```
|
||||
|
||||
The implementation applies to:
|
||||
|
||||
- `https://opencode.ai/zen/go/v1/chat/completions`
|
||||
- `https://opencode.ai/zen/go/v1/messages`
|
||||
- `https://opencode.ai/zen/go/v1/responses`
|
||||
|
||||
## Error Handling
|
||||
|
||||
Session derivation must not make requests fail. Invalid or oversized native
|
||||
header values are ignored and the normal resolved-session fallback is used.
|
||||
Hashing uses Node's built-in `crypto` module and requires no new dependency.
|
||||
|
||||
## Testing
|
||||
|
||||
Add a focused unit suite that proves:
|
||||
|
||||
- all three OpenCode Go transports receive the header;
|
||||
- the same conversation remains stable across requests and transports;
|
||||
- different conversations produce different values;
|
||||
- different agents using the same raw ID remain isolated;
|
||||
- non-OpenCode session IDs are represented as opaque `ses_<32 hex>` values;
|
||||
- a valid native `x-opencode-session` remains stable;
|
||||
- headerless requests still receive a stable fallback;
|
||||
- OpenCode Free behavior is unchanged;
|
||||
- unrelated `DefaultExecutor` providers do not receive the header;
|
||||
- no request state is retained on the shared executor instance.
|
||||
|
||||
Run the focused unit tests first, then the neighboring executor/session tests,
|
||||
the full offline test suite, the application build, and the CLI package build.
|
||||
|
||||
## Delivery
|
||||
|
||||
Build the CLI with `npm --prefix cli run build`, create a package with
|
||||
`npm --prefix cli pack`, and install the generated tarball globally to replace
|
||||
the current npm-installed `9router@0.5.65`. Verify the installed package version
|
||||
and packaged source contains the new executor.
|
||||
|
||||
Upstream issue #3759 already tracks the problem, so no duplicate issue will be
|
||||
created. The pull request will be narrowly scoped to this fix, reference
|
||||
`Fixes #3759`, and explain how it differs from the broader open PR #3780.
|
||||
@@ -111,6 +111,27 @@ Model: cx/gpt-5.2-codex
|
||||
| `cx/gpt-5.2` | GPT 5.2 | General tasks |
|
||||
| `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding |
|
||||
|
||||
### Image Generation
|
||||
|
||||
The Codex image catalog includes `cx/gpt-5.6-sol-image`,
|
||||
`cx/gpt-5.6-terra-image`, and `cx/gpt-5.6-luna-image`, alongside the existing
|
||||
GPT 5.5, 5.4, and 5.3 image aliases. Select them under **Image → OpenAI Codex**
|
||||
in the dashboard, or discover them with `GET /v1/models/image` after connecting
|
||||
a Codex account.
|
||||
|
||||
```bash
|
||||
curl http://localhost:20128/v1/images/generations \
|
||||
-H "Authorization: Bearer $NINE_ROUTER_API_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"model":"cx/gpt-5.6-sol-image","prompt":"A blue square","size":"1024x1024"}'
|
||||
```
|
||||
|
||||
These are 9Router aliases: the image adapter removes `-image` and sends the
|
||||
underlying model an `image_generation` tool through the Codex Responses API.
|
||||
The same endpoint accepts an `image` reference for edits. Image generation
|
||||
requires an eligible ChatGPT Plus or higher account; availability of each
|
||||
underlying model and its image tool depends on the connected account.
|
||||
|
||||
### Pro Tips
|
||||
|
||||
- **5-hour rolling quota** - Fresh quota every 5 hours
|
||||
|
||||
@@ -16,7 +16,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
|
||||
- `rtk/` — request token-killer. `index.js` compresses `tool_result` content in-place (OpenAI/Claude/Kiro shapes); `filters/` per-tool compressors + `autodetect.js`; `headroom.js` external compress proxy; `caveman.js` system-prompt injector.
|
||||
- `transformer/` — `responsesTransformer.js` (Chat Completions SSE → Codex Responses API SSE), `streamToJsonConverter.js`.
|
||||
- `shared/` — cross-provider auth/identity: `clineAuth.js`, `machineId.js`, `qoder/`.
|
||||
- `services/` — `model.js`, `provider.js`, `accountFallback.js`, `combo.js`, `compact.js`, `tokenRefresh/`+`tokenRefresh.js`, `oauthCredentialManager.js`, `usage/`, `projectId.js`, `kiroModels.js`/`qoderModels.js`.
|
||||
- `services/` — `model.js`, `provider.js`, `accountFallback.js`, `combo.js`, `tokenRefresh/`+`tokenRefresh.js`, `oauthCredentialManager.js`, `usage/`, `projectId.js`, `kiroModels.js`/`qoderModels.js`.
|
||||
- `utils/` — streamHandler, stream, sse, error, sessionManager, claudeCloaking, clientDetector, proxyFetch (patches global fetch), cursorProtobuf/cursorChecksum, ollamaTransform.
|
||||
|
||||
## Conventions
|
||||
@@ -37,3 +37,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
|
||||
- `registry/index.js` is an auto-generated static import list; regenerate it (don't hand-edit) after adding a `registry/{id}.js`. REGISTRY_TEMPLATE is excluded by design.
|
||||
- Special binary/protobuf formats (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — handle in their executor.
|
||||
- `rtk/` + `headroom.js` mutate the request body in-place and are **fail-open**: any error returns null and leaves the body untouched — never throw out of them. RTK skips `is_error`/`status:"error"` tool results to preserve traces.
|
||||
- **HTTP 200 in-stream errors**: some upstreams signal failure INSIDE a 200 stream (AI SDK v5 `{"type":"error"}` events, error text in content). HTTP-level success checks miss these → no fallback, `Status: success` in logs. Three hook points + one config escape hatch:
|
||||
1. **Translator** — never map an error event to content. Emit an OpenAI-shaped `chunk.error = { message, type }` + terminal chunk (`translator/response/commandcode-to-openai.js` is the worked example). Downstream `parseSSEToOpenAIResponse` already detects `chunk?.error`.
|
||||
2. **Executor early-peek** — for streaming fallback, read the first events BEFORE returning the response; an error → non-ok Response (`executors/commandcode.js` `peekForUpstreamError`).
|
||||
3. **Config escape hatch (no code)** — per-provider `streamErrorPatterns` setting (UI: provider page → Stream Error Patterns). Patterns matched against the first ~8KB of the stream and the assembled non-streaming content; see `utils/streamErrorPeek.js` + `utils/streamErrorPatterns.js`.
|
||||
|
||||
@@ -171,6 +171,13 @@ export const LOAD_CODE_ASSIST_METADATA = {
|
||||
|
||||
// System prompts
|
||||
export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official CLI for Claude.";
|
||||
// Rewrite rules applied to Antigravity system prompts: competing-client branding
|
||||
// makes the backend flag the request and answer 429 Quota Exhausted.
|
||||
export const ANTIGRAVITY_PROMPT_REWRITES = [
|
||||
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
|
||||
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
|
||||
];
|
||||
|
||||
export const ANTIGRAVITY_DEFAULT_SYSTEM = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.**Absolute paths only****Proactiveness**";
|
||||
|
||||
// Derive từ registry oauth.refreshLeadMs
|
||||
|
||||
@@ -73,6 +73,17 @@ export const ERROR_RULES = [
|
||||
{ status: 403, cooldownMs: COOLDOWN.long },
|
||||
{ status: 404, cooldownMs: COOLDOWN.long },
|
||||
{ status: 429, backoff: true },
|
||||
// --- Request-scoped errors: the request itself is broken — retrying the same
|
||||
// body on another account/model can never succeed, and locking the account
|
||||
// would punish a healthy credential for our own bad request. Callers use this
|
||||
// to fail fast (no account rotation, no model lock).
|
||||
{ text: "context_length_exceeded", requestScoped: true },
|
||||
{ text: "context window", requestScoped: true },
|
||||
{ text: "maximum context length", requestScoped: true },
|
||||
{ text: "prompt is too long", requestScoped: true },
|
||||
{ text: "input is too long", requestScoped: true },
|
||||
{ text: "max_tokens exceed", requestScoped: true },
|
||||
{ text: "reduce the length", requestScoped: true },
|
||||
];
|
||||
|
||||
// Backward compat: COOLDOWN_MS object (used by index.js re-export)
|
||||
|
||||
@@ -3,7 +3,8 @@ import REGISTRY from "../providers/registry/index.js";
|
||||
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
|
||||
import { PROVIDER_MODELS } from "../providers/index.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js";
|
||||
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
export { PROVIDER_MODELS };
|
||||
|
||||
|
||||
@@ -49,6 +50,9 @@ export function findModelName(aliasOrId, modelId) {
|
||||
}
|
||||
|
||||
export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) {
|
||||
return FORMATS.OPENAI_RESPONSES;
|
||||
}
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelTargetFormat(findModel(models, modelId, aliasOrId));
|
||||
|
||||
@@ -1,12 +1,13 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
|
||||
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
|
||||
|
||||
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
|
||||
function sanitizeFunctionName(name) {
|
||||
@@ -187,6 +188,9 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
};
|
||||
}
|
||||
|
||||
const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" });
|
||||
const sessionId = toNumericSessionId(rawSessionId) || rawSessionId;
|
||||
|
||||
// ─── Standard (non-image) request ───
|
||||
// Fix contents for Claude models via Antigravity
|
||||
const contents = body.request?.contents?.map(c => {
|
||||
@@ -202,17 +206,31 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
return true;
|
||||
});
|
||||
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
|
||||
// don't persist thoughtSignature in their history, so backfill the default signature on any
|
||||
// functionCall part that arrives without one.
|
||||
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
|
||||
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
|
||||
// don't persist thoughtSignature in their history, so backfill from cache or default signature.
|
||||
// In parallel function calls, only the first call needs a signature; siblings stay unsigned.
|
||||
let firstFunctionCallSeen = false;
|
||||
const modifiedParts = parts?.map(p => {
|
||||
if (!p.functionCall) return p;
|
||||
const callId = p.functionCall.id;
|
||||
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId) : null;
|
||||
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
|
||||
firstFunctionCallSeen = true;
|
||||
if (callSig) {
|
||||
return { ...p, thoughtSignature: callSig };
|
||||
}
|
||||
if (p.thoughtSignature && !cachedSig) {
|
||||
// Unsigned sibling call
|
||||
const { thoughtSignature: _, ...rest } = p;
|
||||
return rest;
|
||||
}
|
||||
return p;
|
||||
});
|
||||
|
||||
const partsChanged = parts?.length !== c.parts?.length || modifiedParts?.some((p, idx) => p !== c.parts[idx]);
|
||||
if (role !== c.role || partsChanged) {
|
||||
return {
|
||||
...c, role,
|
||||
parts: needsBackfill
|
||||
? parts.map(p => (p.functionCall && !p.thoughtSignature)
|
||||
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
|
||||
: p)
|
||||
: parts,
|
||||
parts: modifiedParts || parts,
|
||||
};
|
||||
}
|
||||
return c;
|
||||
@@ -246,13 +264,13 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {};
|
||||
stripBlacklisted(requestWithoutTools);
|
||||
|
||||
// Rewrite competitive system prompts (e.g. Zed IDE's Claude prompt) to prevent Antigravity from
|
||||
// flagging the request and immediately blocking it with a 429 Quota Exhausted response.
|
||||
// Rewrite competing-client branding in system prompts (e.g. Zed's Claude prompt,
|
||||
// OpenCode naming) so Antigravity doesn't flag the request with a 429 Quota Exhausted.
|
||||
if (requestWithoutTools.systemInstruction?.parts) {
|
||||
const oldText = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
|
||||
for (const part of requestWithoutTools.systemInstruction.parts) {
|
||||
if (typeof part.text === "string" && part.text.includes(oldText)) {
|
||||
part.text = part.text.split(oldText).join("");
|
||||
if (typeof part.text !== "string") continue;
|
||||
for (const { from, to } of ANTIGRAVITY_PROMPT_REWRITES) {
|
||||
part.text = part.text.replaceAll(from, to);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -267,7 +285,7 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
generationConfig,
|
||||
...(contents && { contents }),
|
||||
...(tools && { tools }),
|
||||
sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }),
|
||||
sessionId,
|
||||
safetySettings: undefined,
|
||||
...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } })
|
||||
};
|
||||
|
||||
@@ -2,6 +2,7 @@ import { HTTP_STATUS, RETRY_CONFIG, DEFAULT_RETRY_CONFIG, resolveRetryEntry, FET
|
||||
import { shouldRefreshCredentials } from "../services/oauthCredentialManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
|
||||
@@ -133,13 +134,14 @@ export class BaseExecutor {
|
||||
|
||||
// Abort if upstream doesn't return response headers within connection timeout
|
||||
const connectCtrl = new AbortController();
|
||||
const timeoutMs = this.config?.timeoutMs || FETCH_CONNECT_TIMEOUT_MS;
|
||||
const timeoutMs = await resolveProviderTimeoutMs(this.provider, this.config?.timeoutMs, FETCH_CONNECT_TIMEOUT_MS);
|
||||
const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs);
|
||||
const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal;
|
||||
let fetchT0 = 0;
|
||||
|
||||
try {
|
||||
const bodyStr = JSON.stringify(transformedBody);
|
||||
const fetchT0 = Date.now();
|
||||
fetchT0 = Date.now();
|
||||
dbg("FETCH", `${this.provider.toUpperCase()} → ${url} | body=${bodyStr.length}B | connectTimeout=${timeoutMs}ms`);
|
||||
const response = await proxyAwareFetch(url, {
|
||||
method: "POST",
|
||||
@@ -165,6 +167,11 @@ export class BaseExecutor {
|
||||
clearTimeout(connectTimer);
|
||||
lastError = error;
|
||||
const isConnectTimeout = connectCtrl.signal.aborted && error.name === "AbortError";
|
||||
// Error diagnostic — only logs on actual upstream failure. Distinguishes
|
||||
// undici connect timeout (UND_ERR_CONNECT_TIMEOUT), DNS (ENOTFOUND),
|
||||
// refused (ECONNREFUSED) vs our own connectCtrl abort (AbortError).
|
||||
const cause = error?.cause || {};
|
||||
console.log(`[FETCH-DIAG] ${this.provider} fetch error | name=${error.name} | code=${error.code ?? cause?.code ?? "none"} | msg=${String(error.message).slice(0, 120)} | connectTimeout=${timeoutMs}ms | elapsed=${Date.now() - fetchT0}ms`);
|
||||
dbg("FETCH", `${this.provider.toUpperCase()} ✖ ${error.name}: ${error.message}${isConnectTimeout ? " (connect timeout)" : ""}`);
|
||||
// Connect timeout is internal — convert to retryable network error, don't propagate AbortError
|
||||
if (error.name === "AbortError" && !isConnectTimeout) throw error;
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { randomUUID } from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { commandCodeToOpenAIResponse } from "../translator/response/commandcode-to-openai.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
|
||||
@@ -14,53 +15,282 @@ import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
* We translate each event to an OpenAI chat.completion.chunk and emit it as SSE so
|
||||
* both the streaming and non-streaming (forced SSE → JSON) downstream handlers in
|
||||
* 9router can consume it without further format translation.
|
||||
*
|
||||
* Terminal upstream failures arrive as `{"type":"error"}` events inside the HTTP
|
||||
* 200 stream, so a plain `response.ok` check cannot see them. We peek the first
|
||||
* events before committing the response (see peekForUpstreamError) so a stream
|
||||
* that starts with an error fails fast — the normal `!response.ok` path then
|
||||
* triggers account/model fallback instead of streaming fake success content.
|
||||
*/
|
||||
export class CommandCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("commandcode", PROVIDERS.commandcode);
|
||||
}
|
||||
constructor() {
|
||||
super("commandcode", PROVIDERS.commandcode);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
body.stream = true;
|
||||
return body;
|
||||
}
|
||||
transformRequest(_model, body, _stream, _credentials) {
|
||||
body.stream = true;
|
||||
return body;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
...(this.config.headers || {}),
|
||||
"x-session-id": randomUUID(),
|
||||
};
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
...(this.config.headers || {}),
|
||||
"x-session-id": randomUUID(),
|
||||
};
|
||||
|
||||
const token = credentials?.apiKey || credentials?.accessToken;
|
||||
if (token) headers["Authorization"] = `Bearer ${token}`;
|
||||
const token = credentials?.apiKey || credentials?.accessToken;
|
||||
if (token) headers["Authorization"] = `Bearer ${token}`;
|
||||
|
||||
if (stream) headers["Accept"] = "text/event-stream";
|
||||
return headers;
|
||||
}
|
||||
if (stream) headers["Accept"] = "text/event-stream";
|
||||
return headers;
|
||||
}
|
||||
|
||||
async execute(opts) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
result.response = wrapNdjsonAsOpenAISse(result.response, opts.model);
|
||||
result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
|
||||
return result;
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
let parsed = null;
|
||||
try {
|
||||
parsed = JSON.parse(bodyText || "{}");
|
||||
} catch {
|
||||
parsed = null;
|
||||
}
|
||||
const errObj = parsed?.error || parsed;
|
||||
const msg = errObj?.message || parsed?.message || bodyText || response.statusText;
|
||||
const status = Number(errObj?.code || errObj?.statusCode || response.status) || response.status;
|
||||
return {
|
||||
status,
|
||||
message: msg || `CommandCode upstream error: ${response.status}`,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
function wrapNdjsonAsOpenAISse(originalResponse, model) {
|
||||
export function parseCommandCodeError(event) {
|
||||
if (!event || typeof event !== "object") {
|
||||
return {
|
||||
statusCode: 503,
|
||||
message: "CommandCode upstream error",
|
||||
type: "server_error",
|
||||
};
|
||||
}
|
||||
|
||||
const errVal = event.error ?? event.message ?? "unknown";
|
||||
let message = "";
|
||||
let statusCode = null;
|
||||
let type = "server_error";
|
||||
|
||||
if (typeof errVal === "object" && errVal !== null) {
|
||||
message = errVal.message || errVal.error || JSON.stringify(errVal);
|
||||
if (errVal.statusCode && Number.isInteger(Number(errVal.statusCode))) {
|
||||
statusCode = Number(errVal.statusCode);
|
||||
} else if (errVal.status && Number.isInteger(Number(errVal.status))) {
|
||||
statusCode = Number(errVal.status);
|
||||
}
|
||||
if (errVal.type) type = errVal.type;
|
||||
} else if (typeof errVal === "string") {
|
||||
message = errVal;
|
||||
} else {
|
||||
message = JSON.stringify(errVal);
|
||||
}
|
||||
|
||||
if (event.statusCode && Number.isInteger(Number(event.statusCode))) {
|
||||
statusCode = Number(event.statusCode);
|
||||
}
|
||||
|
||||
if (!statusCode || statusCode < 400 || statusCode > 599) {
|
||||
const lower = message.toLowerCase();
|
||||
if (lower.includes("rate limit") || lower.includes("too many requests")) {
|
||||
statusCode = 429;
|
||||
type = "rate_limit_error";
|
||||
} else if (lower.includes("unauthorized") || lower.includes("invalid api key") || lower.includes("authentication")) {
|
||||
statusCode = 401;
|
||||
type = "authentication_error";
|
||||
} else if (lower.includes("payment required") || lower.includes("billing")) {
|
||||
statusCode = 402;
|
||||
type = "billing_error";
|
||||
} else if (lower.includes("quota") || lower.includes("forbidden") || lower.includes("permission")) {
|
||||
statusCode = 403;
|
||||
type = "permission_error";
|
||||
} else if (lower.includes("not found")) {
|
||||
statusCode = 404;
|
||||
type = "invalid_request_error";
|
||||
} else if (lower.includes("unavailable") || lower.includes("overloaded") || lower.includes("server error")) {
|
||||
statusCode = 503;
|
||||
type = "server_error";
|
||||
} else {
|
||||
statusCode = 503;
|
||||
}
|
||||
}
|
||||
|
||||
return { statusCode, message, type };
|
||||
}
|
||||
|
||||
export async function inspectAndWrapCommandCodeResponse(originalResponse, model) {
|
||||
const reader = originalResponse.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
const bufferedLines = [];
|
||||
let detectedError = null;
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const { value, done } = await reader.read();
|
||||
if (done) {
|
||||
const trimmed = buffer.trim();
|
||||
if (trimmed) {
|
||||
try {
|
||||
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
const parsed = JSON.parse(jsonStr);
|
||||
if (parsed?.type === "error") {
|
||||
detectedError = parsed;
|
||||
} else {
|
||||
bufferedLines.push(trimmed);
|
||||
}
|
||||
} catch {
|
||||
bufferedLines.push(trimmed);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() || "";
|
||||
|
||||
let stopLoop = false;
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
if (!jsonStr || jsonStr === "[DONE]") {
|
||||
bufferedLines.push(trimmed);
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
|
||||
let event;
|
||||
try {
|
||||
event = JSON.parse(jsonStr);
|
||||
} catch {
|
||||
bufferedLines.push(trimmed);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (event?.type === "error") {
|
||||
detectedError = event;
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
|
||||
bufferedLines.push(trimmed);
|
||||
|
||||
if (
|
||||
event?.type === "text-delta" ||
|
||||
event?.type === "reasoning-delta" ||
|
||||
event?.type === "tool-input-start" ||
|
||||
event?.type === "tool-call" ||
|
||||
event?.type === "finish" ||
|
||||
event?.type === "finish-step"
|
||||
) {
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (stopLoop) break;
|
||||
}
|
||||
} catch {
|
||||
try { reader.releaseLock(); } catch { /* ignore */ }
|
||||
return originalResponse;
|
||||
}
|
||||
|
||||
if (detectedError) {
|
||||
try { await reader.cancel(); } catch { /* ignore */ }
|
||||
const { statusCode, message, type } = parseCommandCodeError(detectedError);
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
error: {
|
||||
message: `[CommandCode error: ${message}]`,
|
||||
type,
|
||||
code: statusCode,
|
||||
},
|
||||
}),
|
||||
{
|
||||
status: statusCode,
|
||||
statusText: statusCode === 503 ? "Service Unavailable" : (statusCode === 429 ? "Too Many Requests" : "Bad Gateway"),
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
const combinedStream = createReplayedStream(bufferedLines, buffer, reader);
|
||||
return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse);
|
||||
}
|
||||
|
||||
function createReplayedStream(bufferedLines, remainingBuffer, reader) {
|
||||
const encoder = new TextEncoder();
|
||||
let replayed = false;
|
||||
|
||||
return new ReadableStream({
|
||||
async pull(controller) {
|
||||
if (!replayed) {
|
||||
replayed = true;
|
||||
let prefix = bufferedLines.join("\n");
|
||||
if (prefix && remainingBuffer) {
|
||||
prefix += "\n" + remainingBuffer;
|
||||
} else if (remainingBuffer) {
|
||||
prefix = remainingBuffer;
|
||||
} else if (prefix) {
|
||||
prefix += "\n";
|
||||
}
|
||||
if (prefix) {
|
||||
controller.enqueue(encoder.encode(prefix));
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const { value, done } = await reader.read();
|
||||
if (done) {
|
||||
controller.close();
|
||||
} else {
|
||||
controller.enqueue(value);
|
||||
}
|
||||
} catch (err) {
|
||||
controller.error(err);
|
||||
}
|
||||
},
|
||||
async cancel(reason) {
|
||||
try {
|
||||
await reader.cancel(reason);
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
function wrapNdjsonAsOpenAISse(streamBody, model, originalResponse = null) {
|
||||
const decoder = new TextDecoder();
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
const state = { model };
|
||||
|
||||
const emitChunks = (chunks, controller) => {
|
||||
if (!chunks) return;
|
||||
const list = Array.isArray(chunks) ? chunks : [chunks];
|
||||
for (const c of list) {
|
||||
if (c == null) continue;
|
||||
controller.enqueue(encoder.encode(`data: ${JSON.stringify(c)}\n\n`));
|
||||
}
|
||||
};
|
||||
const emitChunks = (chunks, controller) => {
|
||||
if (!chunks) return;
|
||||
const list = Array.isArray(chunks) ? chunks : [chunks];
|
||||
for (const c of list) {
|
||||
if (c == null) continue;
|
||||
controller.enqueue(encoder.encode(`data: ${JSON.stringify(c)}\n\n`));
|
||||
}
|
||||
};
|
||||
|
||||
const transform = new TransformStream({
|
||||
transform(chunk, controller) {
|
||||
@@ -70,7 +300,6 @@ function wrapNdjsonAsOpenAISse(originalResponse, model) {
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
// Translate AI SDK v5 NDJSON line to one or more OpenAI chunks
|
||||
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
|
||||
}
|
||||
},
|
||||
@@ -83,11 +312,17 @@ function wrapNdjsonAsOpenAISse(originalResponse, model) {
|
||||
},
|
||||
});
|
||||
|
||||
const newBody = originalResponse.body.pipeThrough(transform);
|
||||
const newBody = streamBody.pipeThrough(transform);
|
||||
return new Response(newBody, {
|
||||
status: originalResponse.status,
|
||||
statusText: originalResponse.statusText,
|
||||
headers: originalResponse.headers,
|
||||
status: originalResponse?.status || 200,
|
||||
statusText: originalResponse?.statusText || "OK",
|
||||
headers: {
|
||||
"Content-Type": "text/event-stream",
|
||||
"Cache-Control": "no-cache",
|
||||
"Connection": "keep-alive",
|
||||
...(originalResponse?.headers ? Object.fromEntries(originalResponse.headers.entries()) : {}),
|
||||
"content-type": "text/event-stream",
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -154,7 +154,18 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
|
||||
applyAuth(headers, desc, credentials);
|
||||
|
||||
if (this.provider === "claude" && model) {
|
||||
// anthropic-compatible-* nodes serving a real Claude model sit in front of
|
||||
// Anthropic itself (a rotating multi-account proxy, a corporate gateway),
|
||||
// so the request needs the same beta flags the `claude` provider sends:
|
||||
// without `context-management-2025-06-27` upstream rejects the
|
||||
// `context_management` block Claude Code puts in every request with
|
||||
// "context_management: Extra inputs are not permitted" (HTTP 400), and the
|
||||
// combo silently falls through to the next model. The model id gates this:
|
||||
// a node fronting Kimi or GLM answers on its own ids and never matches, so
|
||||
// gateways that would choke on unknown beta flags are left untouched.
|
||||
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
|
||||
if (model && (this.provider === "claude"
|
||||
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
|
||||
}
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@ import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
@@ -40,6 +41,7 @@ const executors = {
|
||||
vertex: new VertexExecutor("vertex"),
|
||||
"vertex-partner": new VertexExecutor("vertex-partner"),
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
@@ -84,6 +86,7 @@ export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
export { DefaultExecutor } from "./default.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
|
||||
182
open-sse/executors/opencode-go.js
Normal file
182
open-sse/executors/opencode-go.js
Normal file
@@ -0,0 +1,182 @@
|
||||
import crypto from "node:crypto";
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeGoSession";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
|
||||
const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses";
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function translatedSession(sessionId, clientTool) {
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
|
||||
.digest("hex")
|
||||
.slice(0, 32);
|
||||
return `ses_${digest}`;
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so checks hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
function isResponsesModel(model) {
|
||||
return isMuseSparkModel(baseModelId(model));
|
||||
}
|
||||
|
||||
// Flatten Chat Completions tool declarations into the Responses flat shape and
|
||||
// drop hosted/nameless tools the /responses endpoint rejects.
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
// Mirror the request translator: {type:"object"} without properties is rejected
|
||||
// by strict Responses backends, so fill in the empty properties map.
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Last line of defense for native Responses clients (sourceFormat === targetFormat
|
||||
// skips translation): coerce items in place so malformed tool payloads 400 here
|
||||
// with a clear shape instead of upstream as InputValidationError.
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export class OpenCodeGoExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("opencode-go");
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
|
||||
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
|
||||
return super.buildUrl(model, stream, urlIndex, credentials);
|
||||
}
|
||||
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const native = nativeSession(sourceCredentials.rawHeaders);
|
||||
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
|
||||
headers: sourceCredentials.rawHeaders,
|
||||
body,
|
||||
connectionId: sourceCredentials.connectionId,
|
||||
scope: "opencode-go",
|
||||
});
|
||||
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
|
||||
};
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
const credentials = this.prepareRequestCredentials(args);
|
||||
return super.execute({ ...args, credentials });
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
const headers = super.buildHeaders(credentials || {}, stream, url, model);
|
||||
const prepared = credentials?.[SESSION_FIELD];
|
||||
if (prepared) {
|
||||
headers[SESSION_HEADER] = prepared;
|
||||
return headers;
|
||||
}
|
||||
|
||||
const fallback = this.prepareRequestCredentials({ credentials });
|
||||
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
|
||||
return headers;
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
const out = super.transformRequest(model, body);
|
||||
if (!isResponsesModel(model || body?.model)) return out;
|
||||
const normalized = normalizeResponsesInput(out.input);
|
||||
if (normalized) out.input = normalized;
|
||||
if (!Array.isArray(out.input) || out.input.length === 0) {
|
||||
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses names the output cap max_output_tokens, not max_tokens.
|
||||
if (out.max_output_tokens === undefined) {
|
||||
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
|
||||
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
|
||||
}
|
||||
delete out.max_tokens;
|
||||
delete out.max_completion_tokens;
|
||||
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
|
||||
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
|
||||
}
|
||||
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
|
||||
if (!out.reasoning.summary) out.reasoning.summary = "auto";
|
||||
}
|
||||
delete out.reasoning_effort;
|
||||
out.stream = true;
|
||||
out.store = false;
|
||||
normalizeResponsesTools(out);
|
||||
sanitizeResponsesItems(out);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
@@ -1,11 +1,17 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
|
||||
const OPENCODE_UA = "opencode";
|
||||
const MESSAGES_MODELS = new Set();
|
||||
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
|
||||
const RESPONSES_MODELS = new Set([
|
||||
"muse-spark-1.2-contributor-free",
|
||||
"muse-spark-1.3-contributor-free",
|
||||
]);
|
||||
|
||||
function generateRequestId() {
|
||||
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
@@ -15,19 +21,48 @@ function generateSessionId() {
|
||||
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
}
|
||||
|
||||
// Normalize any resolved id into opencode's ses_ format (stable per-conversation)
|
||||
function toOpencodeSession(id) {
|
||||
const stripped = String(id || "").replace(/^ses_/, "").replace(/-/g, "");
|
||||
return stripped ? `ses_${stripped}` : null;
|
||||
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
function isResponsesModel(model) {
|
||||
const base = baseModelId(model);
|
||||
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials) {
|
||||
return toOpencodeSession(resolveSessionId({
|
||||
headers: credentials?.rawHeaders,
|
||||
const headers = credentials?.rawHeaders || {};
|
||||
return resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
}));
|
||||
generate: generateSessionId,
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeOpencodeReasoning(model, body) {
|
||||
const current = body.reasoning;
|
||||
const currentReasoning = current && typeof current === "object" && !Array.isArray(current)
|
||||
? current
|
||||
: null;
|
||||
const requestedEffort = typeof body.reasoning_effort === "string"
|
||||
? body.reasoning_effort
|
||||
: currentReasoning?.effort;
|
||||
if (typeof requestedEffort !== "string") return;
|
||||
|
||||
const cleanModel = baseModelId(model || body.model);
|
||||
const supportedLevels = getThinkingLevels("opencode", cleanModel);
|
||||
let effort = requestedEffort.toLowerCase().trim();
|
||||
if ((effort === "max" || effort === "ultra") && supportedLevels?.length && !supportedLevels.includes(effort)) {
|
||||
if (effort === "ultra" && supportedLevels.includes("max")) effort = "max";
|
||||
else if (supportedLevels.includes("xhigh")) effort = "xhigh";
|
||||
}
|
||||
|
||||
body.reasoning = { ...currentReasoning, effort };
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
delete body.reasoning_effort;
|
||||
}
|
||||
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
@@ -38,13 +73,24 @@ export class OpenCodeExecutor extends BaseExecutor {
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
this._currentSessionId = resolveOpencodeSession(body, credentials);
|
||||
if (isResponsesModel(model)) {
|
||||
// Responses API names the output cap max_output_tokens and takes thinking
|
||||
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
|
||||
if (body.max_output_tokens === undefined) {
|
||||
if (body.max_completion_tokens !== undefined) body.max_output_tokens = body.max_completion_tokens;
|
||||
else if (body.max_tokens !== undefined) body.max_output_tokens = body.max_tokens;
|
||||
}
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
normalizeOpencodeReasoning(model, body);
|
||||
}
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
return MESSAGES_MODELS.has(model)
|
||||
? `${base}/zen/v1/messages`
|
||||
return isResponsesModel(model)
|
||||
? `${base}/zen/v1/responses`
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@ import { PROVIDERS } from "../config/providers.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import {
|
||||
QODER_CHAT_URL_ENCODED,
|
||||
QODER_CHAT_BASE_ALT,
|
||||
@@ -37,10 +38,13 @@ import {
|
||||
QODER_MODEL_MAP,
|
||||
} from "../shared/qoder/constants.js";
|
||||
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
|
||||
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
|
||||
import { encodeDataUri } from "../translator/concerns/image.js";
|
||||
|
||||
/**
|
||||
* Hoist role:"system" messages out of the messages array (Qoder rejects
|
||||
* system in messages) and flatten any multipart content arrays.
|
||||
* system in messages) and flatten multipart content arrays — EXCEPT image
|
||||
* blocks, which are preserved (see normalizeContent).
|
||||
*/
|
||||
function normalizeMessages(messages) {
|
||||
if (!Array.isArray(messages) || messages.length === 0) {
|
||||
@@ -50,18 +54,72 @@ function normalizeMessages(messages) {
|
||||
const out = [];
|
||||
for (const msg of messages) {
|
||||
if (!msg || typeof msg !== "object") continue;
|
||||
const text = extractText(msg.content);
|
||||
if (msg.role === "system") {
|
||||
const text = extractText(msg.content);
|
||||
if (text) systemParts.push(text);
|
||||
continue;
|
||||
}
|
||||
const cloned = { ...msg };
|
||||
cloned.content = text;
|
||||
cloned.content = normalizeContent(msg.content);
|
||||
out.push(cloned);
|
||||
}
|
||||
return { messages: out, systemText: systemParts.join("\n\n") };
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize one message's content for Qoder.
|
||||
*
|
||||
* Text-only content is flattened to a plain string (Qoder's historical
|
||||
* shape). When images are present the content stays an array and image
|
||||
* blocks are kept as OpenAI-style `image_url` parts — verified against the
|
||||
* upstream: it accepts both http(s) URLs and inline base64 data: URIs
|
||||
* directly, no pre-upload to the /image/upload OSS flow required (that is
|
||||
* a qodercli client-side choice, not a protocol requirement). The legacy
|
||||
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
|
||||
* qodercli leaves them null too.
|
||||
*
|
||||
* Claude-style `{type:"image", source:{...}}` blocks are converted to
|
||||
* `image_url` so claude-format clients also round-trip.
|
||||
*/
|
||||
function normalizeContent(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (content == null) return "";
|
||||
if (!Array.isArray(content)) return String(content);
|
||||
|
||||
const blocks = [];
|
||||
const textParts = [];
|
||||
let hasImage = false;
|
||||
for (const item of content) {
|
||||
if (!item || typeof item !== "object") continue;
|
||||
if (item.type === OPENAI_BLOCK.IMAGE_URL && typeof item.image_url?.url === "string" && item.image_url.url) {
|
||||
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: item.image_url.url } });
|
||||
hasImage = true;
|
||||
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
|
||||
// Claude base64/url image → OpenAI image_url equivalent.
|
||||
const src = item.source;
|
||||
const url = src.type === "base64" && src.data
|
||||
? encodeDataUri(src.media_type || "image/png", src.data)
|
||||
: typeof src.url === "string" && src.url ? src.url : null;
|
||||
if (url) {
|
||||
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
|
||||
hasImage = true;
|
||||
}
|
||||
} else if (typeof item.text === "string" && item.text) {
|
||||
if (hasImage || blocks.length) {
|
||||
// Keep ordering faithful once images are in play.
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: item.text });
|
||||
} else {
|
||||
textParts.push(item.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!hasImage) return textParts.join("\n");
|
||||
// Prepend any text collected before the first image block.
|
||||
if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") });
|
||||
return blocks;
|
||||
}
|
||||
|
||||
function extractText(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (content == null) return "";
|
||||
@@ -84,9 +142,9 @@ function extractText(content) {
|
||||
function lastUserText(messages) {
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const m = messages[i];
|
||||
if (m?.role === "user" && typeof m.content === "string") {
|
||||
return m.content;
|
||||
}
|
||||
if (m?.role !== "user") continue;
|
||||
if (typeof m.content === "string") return m.content;
|
||||
if (Array.isArray(m.content)) return extractText(m.content);
|
||||
}
|
||||
return "";
|
||||
}
|
||||
@@ -110,6 +168,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) {
|
||||
if (m.role) { h.update("\0"); h.update(m.role); }
|
||||
if (typeof m.content === "string" && m.content) {
|
||||
h.update("\0"); h.update(m.content);
|
||||
} else if (Array.isArray(m.content)) {
|
||||
// Include image refs so the same prompt with a different image gets
|
||||
// a distinct chat_record_id.
|
||||
h.update("\0");
|
||||
try { h.update(JSON.stringify(m.content)); } catch {}
|
||||
}
|
||||
}
|
||||
if (tools) {
|
||||
@@ -536,7 +599,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
};
|
||||
|
||||
// Abort if upstream doesn't return response headers within connect timeout.
|
||||
const timeoutMs = this.config?.timeoutMs || FETCH_CONNECT_TIMEOUT_MS;
|
||||
const timeoutMs = await resolveProviderTimeoutMs(this.provider, this.config?.timeoutMs, FETCH_CONNECT_TIMEOUT_MS);
|
||||
const connectCtrl = new AbortController();
|
||||
const connectTimer = setTimeout(() => connectCtrl.abort(new Error("fetch connect timeout")), timeoutMs);
|
||||
const mergedSignal = signal ? AbortSignal.any([signal, connectCtrl.signal]) : connectCtrl.signal;
|
||||
|
||||
@@ -11,7 +11,7 @@ import { PROVIDERS } from "../config/providers.js";
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js";
|
||||
import { handleBypassRequest } from "../utils/bypassHandler.js";
|
||||
import { trackPendingRequest, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { getExecutor } from "../executors/index.js";
|
||||
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
||||
import { buildRequestDetail, extractRequestConfig } from "./chatCore/requestDetail.js";
|
||||
@@ -28,7 +28,9 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
|
||||
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { maybeRejectEarlyStreamError } from "../utils/streamErrorPeek.js";
|
||||
|
||||
/**
|
||||
* Core chat handler - shared between SSE and Worker
|
||||
@@ -57,7 +59,7 @@ export function stripContinuityFields(body) {
|
||||
return body;
|
||||
}
|
||||
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking }) {
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, capsOverride = null, streamErrorPatterns = null }) {
|
||||
const { provider, model } = modelInfo;
|
||||
const requestStartTime = Date.now();
|
||||
// Stable per-session color so all lines of one CLI conversation share a tag
|
||||
@@ -90,7 +92,12 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
// differ — kimi/glm only do /chat/completions). Undeclared models keep the
|
||||
// upstream default (use the transport), preserving behavior for glm/deepseek/...
|
||||
const useTransport = (!modelSupportedFormats || modelSupportedFormats.includes(sourceFormat)) ? runtimeTransport : null;
|
||||
const targetFormat = modelTargetFormat || useTransport?.format || getTargetFormat(provider, credentials);
|
||||
// A source-format-matched endpoint keeps the request lossless. Prefer it
|
||||
// over a model-level targetFormat, which is only the fallback for clients
|
||||
// whose wire format has no supported transport (for example MiniMax-M3:
|
||||
// OpenAI clients should stay on /chat/completions; other clients can fall
|
||||
// back to its declared Claude target).
|
||||
const targetFormat = useTransport?.format || modelTargetFormat || getTargetFormat(provider, credentials);
|
||||
if (useTransport && credentials) credentials.runtimeTransport = useTransport;
|
||||
const stripList = getModelStrip(alias, model);
|
||||
const upstreamModel = getModelUpstreamId(alias, model);
|
||||
@@ -100,7 +107,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
if (providerThinking?.mode && providerThinking.mode !== "auto") {
|
||||
const mode = providerThinking.mode;
|
||||
if (mode === "on" && !body.thinking) {
|
||||
console.log("Injecting provider-level thinking config override: on");
|
||||
log?.debug?.("THINKING", `provider-level override: on`);
|
||||
body = { ...body, thinking: { type: "enabled", budget_tokens: 10000 } };
|
||||
} else if (mode === "off" && !body.thinking) {
|
||||
body = { ...body, thinking: { type: "disabled" } };
|
||||
@@ -149,8 +156,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
if (credentials) credentials.rawHeaders = clientRawRequest?.headers || {};
|
||||
|
||||
// Auto-strip media blocks the model can't read (vision/audio/pdf) before translation.
|
||||
// capsOverride lets the app layer assert per-model capabilities (e.g. user-registered
|
||||
// models) on top of the static tables.
|
||||
if (!passthrough) {
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
const caps = { ...getCapabilitiesForModel(provider, model), ...(capsOverride || {}) };
|
||||
if (stripUnsupportedModalities(body, sourceFormat, caps)) {
|
||||
log?.debug?.("MODALITY", `stripped unsupported media for ${provider}/${model}`);
|
||||
}
|
||||
@@ -235,17 +244,23 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
delete translatedBody.tools;
|
||||
}
|
||||
|
||||
// Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax)
|
||||
// reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing.
|
||||
if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) {
|
||||
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
|
||||
}
|
||||
|
||||
// Per-request opt-out: client can bypass all token savers via header
|
||||
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
|
||||
|
||||
// RTK: compress tool_result content
|
||||
const rtkStats = compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled);
|
||||
const rtkLine = formatRtkLog(rtkStats);
|
||||
if (rtkLine) console.log(rtkLine);
|
||||
if (rtkLine) log?.info?.("RTK", rtkLine.replace(/^\[RTK\] /, ""));
|
||||
|
||||
// Headroom: optional external proxy compression; fail open if proxy is absent.
|
||||
const headroomDiagnostics = {};
|
||||
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, diagnostics: headroomDiagnostics });
|
||||
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, timeoutMs: headroomTimeoutMs, diagnostics: headroomDiagnostics });
|
||||
const headroomLine = formatHeadroomLog(headroomStats);
|
||||
const headroomSizeLine = formatHeadroomSizeLog(headroomDiagnostics);
|
||||
if (headroomLine) {
|
||||
@@ -291,7 +306,6 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
|
||||
const executor = getExecutor(provider);
|
||||
trackPendingRequest(model, provider, connectionId, true);
|
||||
appendRequestLog({ model, provider, connectionId, status: "PENDING" }).catch(() => { });
|
||||
|
||||
const msgCount = translatedBody.messages?.length || translatedBody.input?.length || translatedBody.contents?.length || translatedBody.request?.contents?.length || 0;
|
||||
log?.debug?.("REQUEST", `${provider.toUpperCase()} | ${model} | ${msgCount} msgs`);
|
||||
@@ -344,7 +358,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
// exception: it is decoded by the executor into OpenAI-compatible output.
|
||||
let providerResponseFormat = targetFormat;
|
||||
try {
|
||||
const result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
|
||||
const result = await executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
providerSessionId: sessionSeed,
|
||||
clientTool,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
providerResponse = result.response;
|
||||
providerUrl = result.url;
|
||||
providerHeaders = result.headers;
|
||||
@@ -353,7 +377,6 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody);
|
||||
} catch (error) {
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
appendRequestLog({ model, provider, connectionId, status: `FAILED ${error.name === "AbortError" ? 499 : HTTP_STATUS.BAD_GATEWAY}` }).catch(() => { });
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
@@ -398,7 +421,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
|
||||
}
|
||||
try {
|
||||
const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
|
||||
const retryResult = await executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
providerSessionId: sessionSeed,
|
||||
clientTool,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
if (retryResult.response.ok) {
|
||||
providerResponse = retryResult.response;
|
||||
providerUrl = retryResult.url;
|
||||
@@ -413,11 +446,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Provider returned error
|
||||
if (!providerResponse.ok) {
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
const { statusCode, message, resetsAtMs } = await parseUpstreamError(providerResponse, executor);
|
||||
appendRequestLog({ model, provider, connectionId, status: `FAILED ${statusCode}` }).catch(() => { });
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
@@ -438,8 +471,31 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
return createErrorResult(statusCode, errMsg, resetsAtMs);
|
||||
}
|
||||
|
||||
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log };
|
||||
const appendLog = (extra) => appendRequestLog({ model, provider, connectionId, ...extra }).catch(() => { });
|
||||
const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers
|
||||
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log, streamErrorPatterns };
|
||||
|
||||
// Early-peek streaming responses for configured in-stream error patterns.
|
||||
// Some upstreams fail INSIDE a 200 SSE stream; without this the failure is
|
||||
// piped to the client verbatim and account/combo fallback never triggers
|
||||
// (see AGENTS.md "HTTP 200 in-stream errors"). Fail-open: no patterns → pass-through.
|
||||
if (providerResponse.ok && stream) {
|
||||
const peeked = await maybeRejectEarlyStreamError(
|
||||
providerResponse,
|
||||
streamErrorPatterns?.[provider],
|
||||
{ signal: streamController.signal },
|
||||
);
|
||||
if (!peeked.ok) {
|
||||
const { message } = await parseUpstreamError(peeked).catch(() => ({ message: "Stream error pattern matched" }));
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
|
||||
if (log?.errorLine) {
|
||||
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms (in-stream)\n ${message}`);
|
||||
}
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, message);
|
||||
}
|
||||
providerResponse = peeked;
|
||||
}
|
||||
|
||||
const trackDone = () => trackPendingRequest(model, provider, connectionId, false);
|
||||
|
||||
// Provider forced streaming but client wants JSON
|
||||
@@ -457,7 +513,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
|
||||
// Streaming response
|
||||
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials });
|
||||
}
|
||||
|
||||
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {
|
||||
|
||||
@@ -7,7 +7,8 @@ import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
@@ -281,7 +282,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
||||
/**
|
||||
* Handle non-streaming response from provider.
|
||||
*/
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log, streamErrorPatterns }) {
|
||||
trackDone();
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
let responseBody;
|
||||
@@ -316,6 +317,21 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Decloak tool_use names once on raw Claude body, before any translation (INPUT side)
|
||||
responseBody = decloakToolNames(responseBody, toolNameMap);
|
||||
|
||||
// Config-driven in-stream error detection: the HTTP call succeeded but the
|
||||
// assembled content signals an upstream failure — treat it as an error so
|
||||
// account/combo fallback and FAILED logging kick in (AGENTS.md hook #3).
|
||||
const matchedPattern = matchStreamErrorPatterns(
|
||||
streamErrorPatterns?.[provider],
|
||||
responseBody?.choices?.[0]?.message?.content || responseBody?.content || "",
|
||||
);
|
||||
if (matchedPattern) {
|
||||
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
|
||||
if (log?.errorLine) {
|
||||
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms (in-stream)\n Stream error pattern matched: ${matchedPattern}`);
|
||||
}
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Stream error pattern matched: ${matchedPattern}`);
|
||||
}
|
||||
|
||||
const usage = extractUsageFromResponse(responseBody);
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
@@ -372,7 +388,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
provider, model, connectionId, apiKey,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { saveRequestUsage, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { saveRequestUsage, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { COLORS } from "../../utils/stream.js";
|
||||
import { canonicalizeUsage } from "../../utils/usageTracking.js";
|
||||
|
||||
@@ -25,10 +25,16 @@ export function extractUsageFromResponse(responseBody) {
|
||||
if (!responseBody || typeof responseBody !== "object") return null;
|
||||
|
||||
// Claude format
|
||||
// Note: OpenAI Responses usage ({input_tokens, input_tokens_details:{cached_tokens}})
|
||||
// also matches this branch. Its prompt is cache-INCLUSIVE and its cache rides in
|
||||
// input_tokens_details, so emit it as cached_tokens — the convention
|
||||
// canonicalizeUsage() passes through without folding. Reading it here keeps
|
||||
// cache accounting correct for /v1/responses and codex traffic.
|
||||
if (responseBody.usage?.input_tokens !== undefined) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usage.input_tokens || 0,
|
||||
completion_tokens: responseBody.usage.output_tokens || 0,
|
||||
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.input_tokens_details?.cached_tokens,
|
||||
cache_read_input_tokens: responseBody.usage.cache_read_input_tokens,
|
||||
cache_creation_input_tokens: responseBody.usage.cache_creation_input_tokens
|
||||
};
|
||||
@@ -39,7 +45,7 @@ export function extractUsageFromResponse(responseBody) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usage.prompt_tokens || 0,
|
||||
completion_tokens: responseBody.usage.completion_tokens || 0,
|
||||
cached_tokens: responseBody.usage.prompt_tokens_details?.cached_tokens,
|
||||
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.prompt_tokens_details?.cached_tokens,
|
||||
reasoning_tokens: responseBody.usage.completion_tokens_details?.reasoning_tokens
|
||||
};
|
||||
}
|
||||
@@ -58,11 +64,21 @@ export function extractUsageFromResponse(responseBody) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Mask API keys before they reach the requestDetails data blob / DB column.
|
||||
// Only the prefix is kept — enough to distinguish keys without leaking them.
|
||||
export function maskApiKey(key) {
|
||||
if (!key || typeof key !== "string") return undefined;
|
||||
const trimmed = key.trim();
|
||||
if (trimmed.length <= 8) return trimmed.charAt(0) + "***";
|
||||
return trimmed.slice(0, 8) + "***";
|
||||
}
|
||||
|
||||
export function buildRequestDetail(base, overrides = {}) {
|
||||
return {
|
||||
provider: base.provider || "unknown",
|
||||
model: base.model || "unknown",
|
||||
connectionId: base.connectionId || undefined,
|
||||
apiKey: maskApiKey(base.apiKey),
|
||||
timestamp: new Date().toISOString(),
|
||||
latency: base.latency || { ttft: 0, total: 0 },
|
||||
tokens: base.tokens || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
|
||||
@@ -1,22 +1,24 @@
|
||||
import { convertResponsesStreamToJson } from "../../transformer/streamToJsonConverter.js";
|
||||
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
|
||||
const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
import { saveRequestDetail, appendRequestLog } from "@/lib/usageDb.js";
|
||||
const isResponsesProvider = (p) =>
|
||||
PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
|
||||
function textFromResponsesMessageItem(item) {
|
||||
if (!item?.content || !Array.isArray(item.content)) return "";
|
||||
const byType = item.content.find((c) => c.type === "output_text");
|
||||
if (typeof byType?.text === "string") return byType.text;
|
||||
const anyText = item.content.find((c) => typeof c.text === "string");
|
||||
if (typeof anyText?.text === "string") return anyText.text;
|
||||
return "";
|
||||
if (!item?.content || !Array.isArray(item.content)) return "";
|
||||
const byType = item.content.find((c) => c.type === "output_text");
|
||||
if (typeof byType?.text === "string") return byType.text;
|
||||
const anyText = item.content.find((c) => typeof c.text === "string");
|
||||
if (typeof anyText?.text === "string") return anyText.text;
|
||||
return "";
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -24,15 +26,15 @@ function textFromResponsesMessageItem(item) {
|
||||
* Early message blocks often have empty output_text; the user-visible answer is usually in the last non-empty message.
|
||||
*/
|
||||
function pickAssistantMessageForChatCompletion(output) {
|
||||
if (!Array.isArray(output)) return { msgItem: null, textContent: null };
|
||||
const messages = output.filter((item) => item?.type === "message");
|
||||
if (messages.length === 0) return { msgItem: null, textContent: null };
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const text = textFromResponsesMessageItem(messages[i]);
|
||||
if (text.length > 0) return { msgItem: messages[i], textContent: text };
|
||||
}
|
||||
const last = messages[messages.length - 1];
|
||||
return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
|
||||
if (!Array.isArray(output)) return { msgItem: null, textContent: null };
|
||||
const messages = output.filter((item) => item?.type === "message");
|
||||
if (messages.length === 0) return { msgItem: null, textContent: null };
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const text = textFromResponsesMessageItem(messages[i]);
|
||||
if (text.length > 0) return { msgItem: messages[i], textContent: text };
|
||||
}
|
||||
const last = messages[messages.length - 1];
|
||||
return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -110,250 +112,414 @@ function chatCompletionToResponses(responseBody, customToolNames = null) {
|
||||
* Used when provider forces streaming but client wants non-streaming.
|
||||
*/
|
||||
export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
const chunks = [];
|
||||
let streamError = null;
|
||||
const chunks = [];
|
||||
let streamError = null;
|
||||
|
||||
for (const line of String(rawSSE || "").split("\n")) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed.startsWith("data:")) continue;
|
||||
const payload = trimmed.slice(5).trim();
|
||||
if (!payload || payload === "[DONE]") continue;
|
||||
try {
|
||||
const chunk = JSON.parse(payload);
|
||||
if (chunk?.error) streamError = chunk.error;
|
||||
else chunks.push(chunk);
|
||||
} catch { /* ignore malformed lines */ }
|
||||
}
|
||||
for (const line of String(rawSSE || "").split("\n")) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed.startsWith("data:")) continue;
|
||||
const payload = trimmed.slice(5).trim();
|
||||
if (!payload || payload === "[DONE]") continue;
|
||||
try {
|
||||
const chunk = JSON.parse(payload);
|
||||
if (chunk?.error) streamError = chunk.error;
|
||||
else chunks.push(chunk);
|
||||
} catch {
|
||||
/* ignore malformed lines */
|
||||
}
|
||||
}
|
||||
|
||||
if (streamError) return { error: streamError };
|
||||
if (chunks.length === 0) return null;
|
||||
if (streamError) return { error: streamError };
|
||||
if (chunks.length === 0) return null;
|
||||
|
||||
const first = chunks[0];
|
||||
const contentParts = [];
|
||||
const reasoningParts = [];
|
||||
const toolCallMap = new Map(); // index -> { id, type, function: { name, arguments } }
|
||||
let finishReason = "stop";
|
||||
let usage = null;
|
||||
const first = chunks[0];
|
||||
const contentParts = [];
|
||||
const reasoningParts = [];
|
||||
const toolCallMap = new Map(); // index -> { id, type, function: { name, arguments } }
|
||||
let finishReason = "stop";
|
||||
let usage = null;
|
||||
|
||||
for (const chunk of chunks) {
|
||||
const choice = chunk?.choices?.[0];
|
||||
const delta = choice?.delta || {};
|
||||
if (typeof delta.content === "string" && delta.content.length > 0) contentParts.push(delta.content);
|
||||
if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0) reasoningParts.push(delta.reasoning_content);
|
||||
if (choice?.finish_reason) finishReason = choice.finish_reason;
|
||||
if (chunk?.usage && typeof chunk.usage === "object") usage = chunk.usage;
|
||||
for (const chunk of chunks) {
|
||||
const choice = chunk?.choices?.[0];
|
||||
const delta = choice?.delta || {};
|
||||
if (typeof delta.content === "string" && delta.content.length > 0)
|
||||
contentParts.push(delta.content);
|
||||
if (
|
||||
typeof delta.reasoning_content === "string" &&
|
||||
delta.reasoning_content.length > 0
|
||||
)
|
||||
reasoningParts.push(delta.reasoning_content);
|
||||
if (choice?.finish_reason) finishReason = choice.finish_reason;
|
||||
if (chunk?.usage && typeof chunk.usage === "object") usage = chunk.usage;
|
||||
|
||||
// Accumulate tool_calls from streaming deltas
|
||||
if (Array.isArray(delta.tool_calls)) {
|
||||
for (const tc of delta.tool_calls) {
|
||||
const idx = tc.index ?? 0;
|
||||
if (!toolCallMap.has(idx)) {
|
||||
toolCallMap.set(idx, { id: tc.id || "", type: "function", function: { name: "", arguments: "" } });
|
||||
}
|
||||
const existing = toolCallMap.get(idx);
|
||||
if (tc.id) existing.id = tc.id;
|
||||
if (tc.function?.name) existing.function.name += tc.function.name;
|
||||
if (tc.function?.arguments) existing.function.arguments += tc.function.arguments;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Accumulate tool_calls from streaming deltas
|
||||
if (Array.isArray(delta.tool_calls)) {
|
||||
for (const tc of delta.tool_calls) {
|
||||
const idx = tc.index ?? 0;
|
||||
if (!toolCallMap.has(idx)) {
|
||||
toolCallMap.set(idx, {
|
||||
id: tc.id || "",
|
||||
type: "function",
|
||||
function: { name: "", arguments: "" },
|
||||
});
|
||||
}
|
||||
const existing = toolCallMap.get(idx);
|
||||
if (tc.id) existing.id = tc.id;
|
||||
if (tc.function?.name) existing.function.name += tc.function.name;
|
||||
if (tc.function?.arguments)
|
||||
existing.function.arguments += tc.function.arguments;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const message = { role: "assistant", content: contentParts.join("") || (toolCallMap.size > 0 ? null : "") };
|
||||
if (reasoningParts.length > 0) message.reasoning_content = reasoningParts.join("");
|
||||
if (toolCallMap.size > 0) {
|
||||
message.tool_calls = [...toolCallMap.entries()].sort((a, b) => a[0] - b[0]).map(([, tc]) => tc);
|
||||
}
|
||||
const message = {
|
||||
role: "assistant",
|
||||
content: contentParts.join("") || (toolCallMap.size > 0 ? null : ""),
|
||||
};
|
||||
if (reasoningParts.length > 0)
|
||||
message.reasoning_content = reasoningParts.join("");
|
||||
if (toolCallMap.size > 0) {
|
||||
message.tool_calls = [...toolCallMap.entries()]
|
||||
.sort((a, b) => a[0] - b[0])
|
||||
.map(([, tc]) => tc);
|
||||
}
|
||||
|
||||
const result = {
|
||||
id: first.id || `chatcmpl-${Date.now()}`,
|
||||
object: "chat.completion",
|
||||
created: first.created || Math.floor(Date.now() / 1000),
|
||||
model: first.model || fallbackModel || "unknown",
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }]
|
||||
};
|
||||
if (usage) result.usage = usage;
|
||||
return result;
|
||||
const result = {
|
||||
id: first.id || `chatcmpl-${Date.now()}`,
|
||||
object: "chat.completion",
|
||||
created: first.created || Math.floor(Date.now() / 1000),
|
||||
model: first.model || fallbackModel || "unknown",
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
};
|
||||
if (usage) result.usage = usage;
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle case: provider forced streaming but client wants JSON.
|
||||
* Supports both Codex/Responses API SSE and standard Chat Completions SSE.
|
||||
*/
|
||||
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, targetFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, customToolNames, trackDone, appendLog, reqTag, log }) {
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
const isSSE = contentType.includes("text/event-stream") || (contentType === "" && isResponsesProvider(provider));
|
||||
if (!isSSE) return null; // not handled here
|
||||
export async function handleForcedSSEToJson({
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
provider,
|
||||
model,
|
||||
body,
|
||||
stream,
|
||||
translatedBody,
|
||||
finalBody,
|
||||
requestStartTime,
|
||||
connectionId,
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
customToolNames,
|
||||
trackDone,
|
||||
appendLog,
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
}) {
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
const isSSE =
|
||||
contentType.includes("text/event-stream") ||
|
||||
(contentType === "" && isResponsesProvider(provider));
|
||||
if (!isSSE) return null; // not handled here
|
||||
|
||||
trackDone();
|
||||
trackDone();
|
||||
|
||||
const ctx = {
|
||||
provider, model, connectionId,
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null
|
||||
};
|
||||
const ctx = {
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
};
|
||||
|
||||
// Codex/Responses API SSE path
|
||||
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
|
||||
// not the client format: a Responses-API client behind a chat-native forced-streaming
|
||||
// provider still receives chat SSE chunks, which must go through the standard path.
|
||||
const isCodexResponsesApi = isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
if (isCodexResponsesApi) {
|
||||
try {
|
||||
const jsonResponse = await convertResponsesStreamToJson(providerResponse.body);
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
// Codex/Responses API SSE path
|
||||
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
|
||||
// not the client format: a Responses-API client behind a chat-native forced-streaming
|
||||
// provider still receives chat SSE chunks, which must go through the standard path.
|
||||
const isCodexResponsesApi =
|
||||
isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
if (isCodexResponsesApi) {
|
||||
try {
|
||||
const jsonResponse = await convertResponsesStreamToJson(
|
||||
providerResponse.body,
|
||||
);
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
|
||||
const usage = jsonResponse.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
const usage = jsonResponse.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({
|
||||
provider,
|
||||
model,
|
||||
tokens: usage,
|
||||
connectionId,
|
||||
apiKey,
|
||||
endpoint: clientRawRequest?.endpoint,
|
||||
silent: true,
|
||||
});
|
||||
if (log?.line)
|
||||
log.line(
|
||||
reqTag,
|
||||
"📊",
|
||||
formatDoneLine({
|
||||
usage,
|
||||
latency: { total: Date.now() - requestStartTime },
|
||||
}),
|
||||
);
|
||||
|
||||
// Same cache-inclusive total for the recorded detail, so the DB and the
|
||||
// client-facing usage can never disagree.
|
||||
const inTokensForLog = (usage.input_tokens || 0)
|
||||
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0)
|
||||
+ (usage.cache_creation_input_tokens || 0);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(jsonResponse.output);
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
// Same cache-inclusive total for the recorded detail, so the DB and the
|
||||
// client-facing usage can never disagree.
|
||||
const inTokensForLog = (usage.input_tokens || 0)
|
||||
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0)
|
||||
+ (usage.cache_creation_input_tokens || 0);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(
|
||||
jsonResponse.output,
|
||||
);
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: { prompt_tokens: inTokensForLog, completion_tokens: usage.output_tokens || 0 },
|
||||
response: { content: textContent, thinking: null, finish_reason: jsonResponse.status || "unknown" },
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {});
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
apiKey,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: {
|
||||
prompt_tokens: inTokensForLog,
|
||||
completion_tokens: usage.output_tokens || 0,
|
||||
},
|
||||
response: {
|
||||
content: textContent,
|
||||
thinking: null,
|
||||
finish_reason: jsonResponse.status || "unknown",
|
||||
},
|
||||
status: "success",
|
||||
},
|
||||
{ endpoint: clientRawRequest?.endpoint || null },
|
||||
),
|
||||
).catch(() => {});
|
||||
|
||||
// Client is Responses API → return as-is
|
||||
if (sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return { success: true, response: new Response(JSON.stringify(jsonResponse), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
}
|
||||
// Client is Responses API → return as-is
|
||||
if (sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(jsonResponse), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
// Build client-format response.
|
||||
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
|
||||
// only input+output under-reports prompt_tokens — measured: 2012 reported
|
||||
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache
|
||||
// counters in, and keep them visible in prompt_tokens_details so a client can
|
||||
// tell a cache hit from a small prompt.
|
||||
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens || 0;
|
||||
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
|
||||
const outTokens = usage.output_tokens || 0;
|
||||
const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
|
||||
? { prompt_tokens_details: {
|
||||
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
|
||||
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
|
||||
: {};
|
||||
let finalResp;
|
||||
// Build client-format response.
|
||||
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
|
||||
// only input+output under-reports prompt_tokens — measured: 2012 reported
|
||||
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache
|
||||
// counters in, and keep them visible in prompt_tokens_details so a client can
|
||||
// tell a cache hit from a small prompt.
|
||||
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens || 0;
|
||||
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
|
||||
const outTokens = usage.output_tokens || 0;
|
||||
const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
|
||||
? {
|
||||
prompt_tokens_details: {
|
||||
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
|
||||
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
|
||||
: {};
|
||||
let finalResp;
|
||||
|
||||
// Extract tool calls from Responses API output (function_call items)
|
||||
const funcCallItems = (jsonResponse.output || []).filter(item => item.type === "function_call");
|
||||
const toolCalls = funcCallItems.map((item, idx) => ({
|
||||
id: item.call_id || `call_${item.name}_${Date.now()}_${idx}`,
|
||||
type: "function",
|
||||
function: {
|
||||
name: item.name,
|
||||
arguments: typeof item.arguments === "string" ? item.arguments : JSON.stringify(item.arguments || {})
|
||||
}
|
||||
}));
|
||||
const hasToolCalls = toolCalls.length > 0;
|
||||
// Extract tool calls from Responses API output (function_call items)
|
||||
const funcCallItems = (jsonResponse.output || []).filter(
|
||||
(item) => item.type === "function_call",
|
||||
);
|
||||
const toolCalls = funcCallItems.map((item, idx) => ({
|
||||
id: item.call_id || `call_${item.name}_${Date.now()}_${idx}`,
|
||||
type: "function",
|
||||
function: {
|
||||
name: item.name,
|
||||
arguments:
|
||||
typeof item.arguments === "string"
|
||||
? item.arguments
|
||||
: JSON.stringify(item.arguments || {}),
|
||||
},
|
||||
}));
|
||||
const hasToolCalls = toolCalls.length > 0;
|
||||
|
||||
if (sourceFormat === FORMATS.ANTIGRAVITY || sourceFormat === FORMATS.GEMINI || sourceFormat === FORMATS.GEMINI_CLI) {
|
||||
finalResp = {
|
||||
response: {
|
||||
candidates: [{ content: { role: "model", parts: [{ text: textContent || "" }] }, finishReason: "STOP", index: 0 }],
|
||||
usageMetadata: { promptTokenCount: inTokens, candidatesTokenCount: outTokens, totalTokenCount: inTokens + outTokens },
|
||||
modelVersion: model,
|
||||
responseId: jsonResponse.id || `resp_${Date.now()}`
|
||||
}
|
||||
};
|
||||
} else {
|
||||
const message = { role: "assistant", content: textContent || (hasToolCalls ? null : "") };
|
||||
if (hasToolCalls) message.tool_calls = toolCalls;
|
||||
const responseDone = jsonResponse.status === "completed" || jsonResponse.status === "done";
|
||||
const finishReason = hasToolCalls ? "tool_calls" : (responseDone ? "stop" : (jsonResponse.status || "stop"));
|
||||
finalResp = {
|
||||
id: jsonResponse.id || `chatcmpl-${Date.now()}`,
|
||||
object: "chat.completion",
|
||||
created: jsonResponse.created_at || Math.floor(Date.now() / 1000),
|
||||
model: jsonResponse.model || model,
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
usage: { prompt_tokens: inTokens, completion_tokens: outTokens, total_tokens: inTokens + outTokens, ...cacheDetails }
|
||||
};
|
||||
}
|
||||
if (
|
||||
sourceFormat === FORMATS.ANTIGRAVITY ||
|
||||
sourceFormat === FORMATS.GEMINI ||
|
||||
sourceFormat === FORMATS.GEMINI_CLI
|
||||
) {
|
||||
finalResp = {
|
||||
response: {
|
||||
candidates: [
|
||||
{
|
||||
content: {
|
||||
role: "model",
|
||||
parts: [{ text: textContent || "" }],
|
||||
},
|
||||
finishReason: "STOP",
|
||||
index: 0,
|
||||
},
|
||||
],
|
||||
usageMetadata: {
|
||||
promptTokenCount: inTokens,
|
||||
candidatesTokenCount: outTokens,
|
||||
totalTokenCount: inTokens + outTokens,
|
||||
},
|
||||
modelVersion: model,
|
||||
responseId: jsonResponse.id || `resp_${Date.now()}`,
|
||||
},
|
||||
};
|
||||
} else {
|
||||
const message = {
|
||||
role: "assistant",
|
||||
content: textContent || (hasToolCalls ? null : ""),
|
||||
};
|
||||
if (hasToolCalls) message.tool_calls = toolCalls;
|
||||
const responseDone =
|
||||
jsonResponse.status === "completed" || jsonResponse.status === "done";
|
||||
const finishReason = hasToolCalls
|
||||
? "tool_calls"
|
||||
: responseDone
|
||||
? "stop"
|
||||
: jsonResponse.status || "stop";
|
||||
finalResp = {
|
||||
id: jsonResponse.id || `chatcmpl-${Date.now()}`,
|
||||
object: "chat.completion",
|
||||
created: jsonResponse.created_at || Math.floor(Date.now() / 1000),
|
||||
model: jsonResponse.model || model,
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
usage: {
|
||||
prompt_tokens: inTokens,
|
||||
completion_tokens: outTokens,
|
||||
total_tokens: inTokens + outTokens,
|
||||
...cacheDetails,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
return { success: true, response: new Response(JSON.stringify(finalResp), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Responses API SSE→JSON failed:", err);
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON");
|
||||
}
|
||||
}
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(finalResp), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Responses API SSE→JSON failed:", err);
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
"Failed to convert streaming response to JSON",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Standard Chat Completions SSE path
|
||||
try {
|
||||
const sseText = await providerResponse.text();
|
||||
const parsed = parseSSEToOpenAIResponse(sseText, model);
|
||||
if (!parsed) return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Invalid SSE response for non-streaming request");
|
||||
if (parsed.error) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
parsed.error.message || "Upstream SSE stream failed"
|
||||
);
|
||||
}
|
||||
// Standard Chat Completions SSE path
|
||||
try {
|
||||
const sseText = await providerResponse.text();
|
||||
const parsed = parseSSEToOpenAIResponse(sseText, model);
|
||||
if (!parsed)
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
"Invalid SSE response for non-streaming request",
|
||||
);
|
||||
if (parsed.error) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
parsed.error.message || "Upstream SSE stream failed",
|
||||
);
|
||||
}
|
||||
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
// Config-driven in-stream error detection: the request "succeeded" at the
|
||||
// HTTP level, but the content signals an upstream failure — treat it as an
|
||||
// error so account/combo fallback and FAILED logging kick in.
|
||||
const matchedPattern = matchStreamErrorPatterns(
|
||||
streamErrorPatterns?.[provider],
|
||||
parsed.choices?.[0]?.message?.content || "",
|
||||
);
|
||||
if (matchedPattern) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
`Stream error pattern matched: ${matchedPattern}`,
|
||||
);
|
||||
}
|
||||
|
||||
const usage = parsed.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: usage,
|
||||
response: {
|
||||
content: parsed.choices?.[0]?.message?.content || null,
|
||||
thinking: parsed.choices?.[0]?.message?.reasoning_content || null,
|
||||
finish_reason: parsed.choices?.[0]?.finish_reason || "unknown"
|
||||
},
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {});
|
||||
const usage = parsed.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({
|
||||
provider,
|
||||
model,
|
||||
tokens: usage,
|
||||
connectionId,
|
||||
apiKey,
|
||||
endpoint: clientRawRequest?.endpoint,
|
||||
silent: true,
|
||||
});
|
||||
if (log?.line)
|
||||
log.line(
|
||||
reqTag,
|
||||
"📊",
|
||||
formatDoneLine({
|
||||
usage,
|
||||
latency: { total: Date.now() - requestStartTime },
|
||||
}),
|
||||
);
|
||||
|
||||
// Re-attach usage explicitly. This handler already HAS the correct usage — it is
|
||||
// the same object written to the usage DB, and for a cached Claude request that DB
|
||||
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
|
||||
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
|
||||
// on both sides). Whatever drops it between assembly and serialisation, the client
|
||||
// must not be left unable to account for its own token spend: a caller cannot tell
|
||||
// a 90%-cached request from a cheap one without this.
|
||||
if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
|
||||
// Re-attach usage explicitly. This handler already HAS the correct usage — it is
|
||||
// the same object written to the usage DB, and for a cached Claude request that DB
|
||||
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
|
||||
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
|
||||
// on both sides). Whatever drops it between assembly and serialisation, the client
|
||||
// must not be left unable to account for its own token spend: a caller cannot tell
|
||||
// a 90%-cached request from a cheap one without this.
|
||||
if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
|
||||
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
// Previously this was unconditional, which broke Qwen3.5, Claude extended thinking, etc.
|
||||
if (parsed?.choices) {
|
||||
for (const choice of parsed.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
// Previously this was unconditional, which broke Qwen3.5, Claude extended thinking, etc.
|
||||
if (parsed?.choices) {
|
||||
for (const choice of parsed.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A Responses-format client (e.g. Codex) forced this provider to stream,
|
||||
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
|
||||
// body; convert it to the Responses `output` shape so tool_calls are not
|
||||
// lost on the non-streaming return path. Inlined (not imported from
|
||||
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
|
||||
// already imports parseSSEToOpenAIResponse from this module.
|
||||
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
|
||||
? chatCompletionToResponses(parsed, customToolNames)
|
||||
: parsed;
|
||||
// A Responses-format client (e.g. Codex) forced this provider to stream,
|
||||
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
|
||||
// body; convert it to the Responses `output` shape so tool_calls are not
|
||||
// lost on the non-streaming return path. Inlined (not imported from
|
||||
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
|
||||
// already imports parseSSEToOpenAIResponse from this module.
|
||||
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
|
||||
? chatCompletionToResponses(parsed, customToolNames)
|
||||
: parsed;
|
||||
|
||||
return { success: true, response: new Response(JSON.stringify(finalBody), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Chat Completions SSE→JSON failed:", err);
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON");
|
||||
}
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(finalBody), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Chat Completions SSE→JSON failed:", err);
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
"Failed to convert streaming response to JSON",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,28 +1,37 @@
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { needsTranslation } from "../../translator/index.js";
|
||||
import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js";
|
||||
import {
|
||||
createSSETransformStreamWithLogger,
|
||||
createPassthroughStreamWithLogger,
|
||||
} from "../../utils/stream.js";
|
||||
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
saveUsageStats,
|
||||
formatDoneLine,
|
||||
} from "./requestDetail.js";
|
||||
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
|
||||
// Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat.
|
||||
// Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI.
|
||||
const CODEX_SOURCE_TO_TARGET = {
|
||||
[FORMATS.OPENAI_RESPONSES]: FORMATS.OPENAI_RESPONSES,
|
||||
[FORMATS.CLAUDE]: FORMATS.CLAUDE,
|
||||
[FORMATS.ANTIGRAVITY]: FORMATS.ANTIGRAVITY,
|
||||
[FORMATS.GEMINI]: FORMATS.ANTIGRAVITY,
|
||||
[FORMATS.GEMINI_CLI]: FORMATS.ANTIGRAVITY,
|
||||
[FORMATS.OPENAI_RESPONSES]: FORMATS.OPENAI_RESPONSES,
|
||||
[FORMATS.CLAUDE]: FORMATS.CLAUDE,
|
||||
[FORMATS.ANTIGRAVITY]: FORMATS.ANTIGRAVITY,
|
||||
[FORMATS.GEMINI]: FORMATS.ANTIGRAVITY,
|
||||
[FORMATS.GEMINI_CLI]: FORMATS.ANTIGRAVITY,
|
||||
};
|
||||
|
||||
/**
|
||||
* Determine which SSE transform stream to use based on provider/format.
|
||||
*/
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) {
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) {
|
||||
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
|
||||
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
|
||||
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
@@ -30,20 +39,28 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
|
||||
if (needsCodexTranslation) {
|
||||
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
|
||||
}
|
||||
|
||||
if (needsTranslation(targetFormat, sourceFormat)) {
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
|
||||
}
|
||||
|
||||
return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey);
|
||||
return createPassthroughStreamWithLogger(
|
||||
provider,
|
||||
reqLogger,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) {
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
@@ -52,93 +69,192 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
});
|
||||
}
|
||||
|
||||
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
|
||||
// page), piping it through the SSE transform stream causes Next.js
|
||||
// "failed to pipe response" and crashes the chat router. Read the body,
|
||||
// pull a short human-readable message from the <title>, sanitize it, and
|
||||
// return a clean JSON error instead. The message is stripped of HTML tags
|
||||
// and clamped so untrusted upstream text never reaches the client verbatim
|
||||
// (the UI may render error.message as HTML).
|
||||
const upstreamContentType = (providerResponse.headers.get('content-type') || '').toLowerCase();
|
||||
if (upstreamContentType && !upstreamContentType.includes('text/event-stream') && !upstreamContentType.includes('application/json')) {
|
||||
const bodyText = await providerResponse.text().catch(() => '');
|
||||
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
|
||||
const sanitizedTitle = (titleMatch?.[1] || '').replace(/<[^>]*>/g, '').replace(/[\r\n]+/g, ' ').trim().slice(0, 160);
|
||||
const shortMsg = sanitizedTitle
|
||||
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
if (log?.errorLine) log.errorLine(reqTag, "✗", `BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`);
|
||||
else console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
response: new Response(JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }), {
|
||||
status,
|
||||
headers: { 'Content-Type': 'application/json', 'Access-Control-Allow-Origin': '*' },
|
||||
}),
|
||||
};
|
||||
}
|
||||
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
|
||||
// page), piping it through the SSE transform stream causes Next.js
|
||||
// "failed to pipe response" and crashes the chat router. Read the body,
|
||||
// pull a short human-readable message from the <title>, sanitize it, and
|
||||
// return a clean JSON error instead. The message is stripped of HTML tags
|
||||
// and clamped so untrusted upstream text never reaches the client verbatim
|
||||
// (the UI may render error.message as HTML).
|
||||
const upstreamContentType = (
|
||||
providerResponse.headers.get("content-type") || ""
|
||||
).toLowerCase();
|
||||
if (
|
||||
upstreamContentType &&
|
||||
!upstreamContentType.includes("text/event-stream") &&
|
||||
!upstreamContentType.includes("application/json")
|
||||
) {
|
||||
const bodyText = await providerResponse.text().catch(() => "");
|
||||
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
|
||||
const sanitizedTitle = (titleMatch?.[1] || "")
|
||||
.replace(/<[^>]*>/g, "")
|
||||
.replace(/[\r\n]+/g, " ")
|
||||
.trim()
|
||||
.slice(0, 160);
|
||||
const shortMsg =
|
||||
sanitizedTitle ||
|
||||
(bodyText.length < 200
|
||||
? bodyText
|
||||
.replace(/<[^>]*>/g, "")
|
||||
.trim()
|
||||
.slice(0, 160)
|
||||
: `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
if (log?.errorLine)
|
||||
log.errorLine(
|
||||
reqTag,
|
||||
"✗",
|
||||
`BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`,
|
||||
);
|
||||
else
|
||||
console.warn(
|
||||
`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`,
|
||||
);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
response: new Response(
|
||||
JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }),
|
||||
{
|
||||
status,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
},
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey });
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
|
||||
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null;
|
||||
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
const isResponsesPassthrough =
|
||||
sourceFormat === FORMATS.OPENAI_RESPONSES &&
|
||||
targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough
|
||||
? buildAbortedResponsesTerminalBytes
|
||||
: null;
|
||||
const stallTimeoutMs =
|
||||
PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(
|
||||
providerResponse,
|
||||
transformStream,
|
||||
streamController,
|
||||
onAbortTerminal,
|
||||
stallTimeoutMs,
|
||||
);
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: "[Streaming - raw response not captured]",
|
||||
response: { content: "[Streaming in progress...]", thinking: null, type: "streaming" },
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save streaming request:", err.message);
|
||||
});
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
apiKey,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: "[Streaming - raw response not captured]",
|
||||
response: {
|
||||
content: "[Streaming in progress...]",
|
||||
thinking: null,
|
||||
type: "streaming",
|
||||
},
|
||||
pxpipe,
|
||||
status: "success",
|
||||
},
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to save streaming request:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(transformedBody, { headers: SSE_HEADERS })
|
||||
};
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(transformedBody, { headers: SSE_HEADERS }),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Build onStreamComplete callback for streaming usage tracking.
|
||||
*/
|
||||
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest, pxpipe, reqTag, log }) {
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
export function buildOnStreamComplete({
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
apiKey,
|
||||
requestStartTime,
|
||||
body,
|
||||
stream,
|
||||
finalBody,
|
||||
translatedBody,
|
||||
clientRawRequest,
|
||||
pxpipe,
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
}) {
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
|
||||
const onStreamComplete = (contentObj, usage, ttftAt) => {
|
||||
const latency = {
|
||||
ttft: ttftAt ? ttftAt - requestStartTime : Date.now() - requestStartTime,
|
||||
total: Date.now() - requestStartTime
|
||||
};
|
||||
const safeContent = contentObj?.content || "[Empty streaming response]";
|
||||
const safeThinking = contentObj?.thinking || null;
|
||||
const onStreamComplete = (contentObj, usage, ttftAt) => {
|
||||
const latency = {
|
||||
ttft: ttftAt ? ttftAt - requestStartTime : Date.now() - requestStartTime,
|
||||
total: Date.now() - requestStartTime,
|
||||
};
|
||||
const safeContent = contentObj?.content || "[Empty streaming response]";
|
||||
const safeThinking = contentObj?.thinking || null;
|
||||
const rawProviderText = typeof contentObj?.rawProviderText === "string" ? contentObj.rawProviderText : "";
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency,
|
||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: safeContent,
|
||||
response: { content: safeContent, thinking: safeThinking, type: "streaming" },
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to update streaming content:", err.message);
|
||||
});
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
apiKey,
|
||||
latency,
|
||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: rawProviderText || safeContent,
|
||||
response: {
|
||||
content: safeContent,
|
||||
thinking: safeThinking,
|
||||
type: "streaming",
|
||||
},
|
||||
pxpipe,
|
||||
status: streamStatusForContent(
|
||||
streamErrorPatterns?.[provider],
|
||||
safeContent,
|
||||
),
|
||||
},
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to update streaming content:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
|
||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, label: "STREAM USAGE", silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency }));
|
||||
};
|
||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||
saveUsageStats({
|
||||
provider,
|
||||
model,
|
||||
tokens: usage,
|
||||
connectionId,
|
||||
apiKey,
|
||||
endpoint: clientRawRequest?.endpoint,
|
||||
label: "STREAM USAGE",
|
||||
silent: true,
|
||||
});
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency }));
|
||||
};
|
||||
|
||||
return { onStreamComplete, streamDetailId };
|
||||
return { onStreamComplete, streamDetailId };
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa
|
||||
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama
|
||||
// Returns normalized shape across all providers
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 15000;
|
||||
@@ -56,8 +56,8 @@ function parseJinaTitle(text) {
|
||||
return m ? m[1].trim() : null;
|
||||
}
|
||||
|
||||
function buildData({ provider, url, title, format, text, costUsd, responseMs, upstreamMs }) {
|
||||
return {
|
||||
function buildData({ provider, url, title, format, text, links, costUsd, responseMs, upstreamMs }) {
|
||||
const data = {
|
||||
provider,
|
||||
url,
|
||||
title: title || null,
|
||||
@@ -66,6 +66,8 @@ function buildData({ provider, url, title, format, text, costUsd, responseMs, up
|
||||
usage: { fetch_cost_usd: costUsd ?? null },
|
||||
metrics: { response_time_ms: responseMs, upstream_latency_ms: upstreamMs }
|
||||
};
|
||||
if (Array.isArray(links)) data.links = links;
|
||||
return data;
|
||||
}
|
||||
|
||||
async function readJsonOrText(res) {
|
||||
@@ -115,6 +117,18 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
|
||||
if (provider === "exa") {
|
||||
return await runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt });
|
||||
}
|
||||
if (provider === "ollama") {
|
||||
return await runOllama({
|
||||
url,
|
||||
fmt,
|
||||
timeoutMs,
|
||||
apiKey,
|
||||
maxCharacters,
|
||||
costPerQuery,
|
||||
startedAt,
|
||||
baseUrl: providerConfig?.baseUrl,
|
||||
});
|
||||
}
|
||||
return { success: false, status: 400, error: `Unsupported provider: ${provider}` };
|
||||
} catch (err) {
|
||||
log?.("fetch handler error:", err?.message || err);
|
||||
@@ -241,3 +255,56 @@ async function runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery
|
||||
})
|
||||
};
|
||||
}
|
||||
|
||||
async function runOllama({
|
||||
url,
|
||||
fmt,
|
||||
timeoutMs,
|
||||
apiKey,
|
||||
maxCharacters,
|
||||
costPerQuery,
|
||||
startedAt,
|
||||
baseUrl,
|
||||
}) {
|
||||
const upstreamStart = Date.now();
|
||||
const r = await tryFetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"content-type": "application/json",
|
||||
...(apiKey ? { authorization: `Bearer ${apiKey}` } : {})
|
||||
},
|
||||
body: JSON.stringify({ url })
|
||||
}, timeoutMs);
|
||||
|
||||
if (!r.ok) {
|
||||
return { success: false, status: r.timeout ? 504 : 502, error: r.error };
|
||||
}
|
||||
const upstreamMs = Date.now() - upstreamStart;
|
||||
const { json, text: responseText } = await readJsonOrText(r.res);
|
||||
if (!r.res.ok) {
|
||||
const error = json?.error
|
||||
|| json?.message
|
||||
|| responseText?.slice(0, 500)
|
||||
|| `Ollama error: ${r.res.status}`;
|
||||
return { success: false, status: r.res.status, error };
|
||||
}
|
||||
if (!json || typeof json.content !== "string") {
|
||||
return { success: false, status: 502, error: "Ollama returned an empty or invalid web fetch response" };
|
||||
}
|
||||
|
||||
const text = truncate(json.content, maxCharacters);
|
||||
return {
|
||||
success: true,
|
||||
data: buildData({
|
||||
provider: "ollama",
|
||||
url,
|
||||
title: json.title || null,
|
||||
format: fmt,
|
||||
text,
|
||||
links: json.links,
|
||||
costUsd: costPerQuery,
|
||||
responseMs: Date.now() - startedAt,
|
||||
upstreamMs
|
||||
})
|
||||
};
|
||||
}
|
||||
|
||||
@@ -96,7 +96,7 @@ export async function handleImageGenerationCore({
|
||||
let requestBody;
|
||||
|
||||
try {
|
||||
url = adapter.buildUrl(model, credentials);
|
||||
url = adapter.buildUrl(model, credentials, body);
|
||||
requestBody = await adapter.buildBody(model, body);
|
||||
headers = adapter.buildHeaders(credentials, requestBody, model, body);
|
||||
} catch (error) {
|
||||
@@ -140,7 +140,7 @@ export async function handleImageGenerationCore({
|
||||
try {
|
||||
const retryBody = await adapter.buildBody(model, body);
|
||||
const retryHeaders = adapter.buildHeaders(credentials, retryBody, model, body);
|
||||
const retryUrl = adapter.buildUrl(model, credentials);
|
||||
const retryUrl = adapter.buildUrl(model, credentials, body);
|
||||
providerResponse = await fetch(retryUrl, {
|
||||
method: "POST",
|
||||
headers: retryHeaders,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// Antigravity image adapter - delegates to the executor for correct request
|
||||
// envelope (project, model, requestType, sessionId) and auth headers.
|
||||
import { nowSec } from "./_base.js";
|
||||
import { nowSec, sizeToAspectRatio } from "./_base.js";
|
||||
import { getExecutor } from "../../executors/index.js";
|
||||
|
||||
// Convert image input (data URI or raw base64) to Gemini inlineData part
|
||||
@@ -31,6 +31,19 @@ export default {
|
||||
const executor = getExecutor("antigravity");
|
||||
if (!executor) throw new Error("Antigravity executor not found");
|
||||
|
||||
// Ensure we use an image model for image generation
|
||||
const isImageModel = (m) => /image|imagen|image-generation/i.test(m || "");
|
||||
let targetModel = isImageModel(model) ? model : "gemini-3.1-flash-image";
|
||||
|
||||
// If body.size is provided, resolve aspect ratio and append to model
|
||||
if (body.size && typeof body.size === "string") {
|
||||
const ratio = sizeToAspectRatio(body.size);
|
||||
const suffix = ratio.replace(":", "x");
|
||||
if (!targetModel.includes(suffix)) {
|
||||
targetModel = `${targetModel}-${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
// Build parts: text prompt + optional input image for editing
|
||||
const parts = [{ text: body.prompt }];
|
||||
const imageInput = body.image || (Array.isArray(body.images) && body.images[0]);
|
||||
@@ -44,7 +57,7 @@ export default {
|
||||
};
|
||||
|
||||
const result = await executor.execute({
|
||||
model,
|
||||
model: targetModel,
|
||||
body: chatBody,
|
||||
stream: false,
|
||||
credentials,
|
||||
|
||||
@@ -12,6 +12,7 @@ import blackForestLabs from "./blackForestLabs.js";
|
||||
import runwayml from "./runwayml.js";
|
||||
import cloudflareAi from "./cloudflareAi.js";
|
||||
import antigravity from "./antigravity.js";
|
||||
import xai from "./xai.js";
|
||||
|
||||
const ADAPTERS = {
|
||||
openai: createOpenAIAdapter("openai"),
|
||||
@@ -19,7 +20,7 @@ const ADAPTERS = {
|
||||
openrouter: createOpenAIAdapter("openrouter"),
|
||||
recraft: createOpenAIAdapter("recraft"),
|
||||
"vercel-ai-gateway": createOpenAIAdapter("vercel-ai-gateway"),
|
||||
xai: createOpenAIAdapter("xai"),
|
||||
xai,
|
||||
gemini,
|
||||
codex,
|
||||
sdwebui,
|
||||
|
||||
137
open-sse/handlers/imageProviders/xai.js
Normal file
137
open-sse/handlers/imageProviders/xai.js
Normal file
@@ -0,0 +1,137 @@
|
||||
// xAI Grok Imagine — text-to-image + single/multi image editing
|
||||
// Docs:
|
||||
// https://docs.x.ai/developers/model-capabilities/images/generation
|
||||
// https://docs.x.ai/developers/model-capabilities/images/editing
|
||||
// https://docs.x.ai/developers/model-capabilities/images/multi-image-editing
|
||||
import { sizeToAspectRatio } from "./_base.js";
|
||||
import { PROVIDER_MEDIA } from "../../providers/index.js";
|
||||
|
||||
const IMG_CFG = PROVIDER_MEDIA["xai"]?.imageConfig || {};
|
||||
const GENERATIONS_URL = IMG_CFG.baseUrl || "https://api.x.ai/v1/images/generations";
|
||||
const EDITS_URL = IMG_CFG.editsUrl || "https://api.x.ai/v1/images/edits";
|
||||
|
||||
const ASPECT_RATIOS = new Set([
|
||||
"auto",
|
||||
"1:1",
|
||||
"16:9",
|
||||
"9:16",
|
||||
"4:3",
|
||||
"3:2",
|
||||
"2:3",
|
||||
"9:19.5",
|
||||
"20:9",
|
||||
]);
|
||||
|
||||
function hasEditInput(body) {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (body.image) return true;
|
||||
return Array.isArray(body.images) && body.images.some(Boolean);
|
||||
}
|
||||
|
||||
/** Normalize client image input → xAI image ref object */
|
||||
function toXaiImageRef(input) {
|
||||
if (!input) return null;
|
||||
|
||||
if (typeof input === "object") {
|
||||
// Already xAI-shaped or partial
|
||||
if (input.file_id) {
|
||||
return {
|
||||
type: input.type || "image_url",
|
||||
file_id: input.file_id,
|
||||
...(input.url ? { url: input.url } : {}),
|
||||
};
|
||||
}
|
||||
if (input.url) {
|
||||
return { type: input.type || "image_url", url: input.url };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
if (typeof input !== "string") return null;
|
||||
const trimmed = input.trim();
|
||||
if (!trimmed) return null;
|
||||
|
||||
// Public URL or data URI
|
||||
if (/^https?:\/\//i.test(trimmed) || /^data:image\//i.test(trimmed)) {
|
||||
return { type: "image_url", url: trimmed };
|
||||
}
|
||||
|
||||
// Raw base64 → data URI
|
||||
return { type: "image_url", url: `data:image/png;base64,${trimmed}` };
|
||||
}
|
||||
|
||||
function collectImageRefs(body) {
|
||||
const refs = [];
|
||||
if (Array.isArray(body.images)) {
|
||||
for (const item of body.images) {
|
||||
const ref = toXaiImageRef(item);
|
||||
if (ref) refs.push(ref);
|
||||
}
|
||||
}
|
||||
if (body.image) {
|
||||
const ref = toXaiImageRef(body.image);
|
||||
if (ref) refs.push(ref);
|
||||
}
|
||||
// xAI multi-edit supports up to 3 source images
|
||||
return refs.slice(0, 3);
|
||||
}
|
||||
|
||||
function resolveAspectRatio(body) {
|
||||
if (typeof body.aspect_ratio === "string" && body.aspect_ratio.trim()) {
|
||||
const ratio = body.aspect_ratio.trim();
|
||||
if (ASPECT_RATIOS.has(ratio)) return ratio;
|
||||
// Pass through unknown ratio strings (upstream will validate)
|
||||
return ratio;
|
||||
}
|
||||
// OpenAI-style size → aspect ratio (skip auto)
|
||||
if (body.size && body.size !== "auto") {
|
||||
return sizeToAspectRatio(body.size);
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveResolution(body) {
|
||||
if (typeof body.resolution !== "string") return undefined;
|
||||
const value = body.resolution.trim().toLowerCase();
|
||||
if (!value || value === "auto") return undefined;
|
||||
return value; // "1k" | "2k"
|
||||
}
|
||||
|
||||
export default {
|
||||
buildUrl: (_model, _credentials, body) => (hasEditInput(body) ? EDITS_URL : GENERATIONS_URL),
|
||||
|
||||
buildHeaders: (creds) => {
|
||||
const headers = { "Content-Type": "application/json", ...(IMG_CFG.headers || {}) };
|
||||
const key = creds?.apiKey || creds?.accessToken;
|
||||
if (key) headers["Authorization"] = `Bearer ${key}`;
|
||||
return headers;
|
||||
},
|
||||
|
||||
buildBody: (model, body) => {
|
||||
const req = {
|
||||
model,
|
||||
prompt: body.prompt,
|
||||
};
|
||||
|
||||
if (body.n != null) req.n = body.n;
|
||||
if (body.response_format) req.response_format = body.response_format;
|
||||
|
||||
const aspectRatio = resolveAspectRatio(body);
|
||||
if (aspectRatio) req.aspect_ratio = aspectRatio;
|
||||
|
||||
const resolution = resolveResolution(body);
|
||||
if (resolution) req.resolution = resolution;
|
||||
|
||||
const refs = collectImageRefs(body);
|
||||
if (refs.length === 1) {
|
||||
req.image = refs[0];
|
||||
} else if (refs.length > 1) {
|
||||
req.images = refs;
|
||||
}
|
||||
|
||||
return req;
|
||||
},
|
||||
|
||||
// xAI already returns OpenAI-compatible { created, data: [{ url | b64_json }] }
|
||||
normalize: (responseBody) => responseBody,
|
||||
};
|
||||
@@ -347,6 +347,81 @@ function buildSearxngRequest(config, params) {
|
||||
};
|
||||
}
|
||||
|
||||
function buildXquikRequest(config, params) {
|
||||
const apiKey = params.token;
|
||||
if (!apiKey) throw new Error("Xquik requires an API key");
|
||||
|
||||
const queryType = getProviderSetting(params, "queryType");
|
||||
if (queryType && !["Latest", "Top"].includes(queryType)) {
|
||||
throw new Error("Xquik queryType must be Latest or Top");
|
||||
}
|
||||
|
||||
const qp = new URLSearchParams({
|
||||
q: params.query,
|
||||
limit: String(params.maxResults),
|
||||
});
|
||||
const cursor = getProviderSetting(params, "cursor");
|
||||
if (cursor) qp.set("cursor", cursor);
|
||||
if (queryType) qp.set("queryType", queryType);
|
||||
if (params.language) qp.set("language", params.language);
|
||||
|
||||
return {
|
||||
url: `${resolveBaseUrl(config, params)}?${qp}`,
|
||||
init: {
|
||||
method: "GET",
|
||||
headers: { Accept: "application/json", "x-api-key": apiKey },
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── Ollama Cloud web_search ──────────────────────────────────────────────
|
||||
// POST https://ollama.com/api/web_search { query, max_results }
|
||||
// Response: { results: [{ title, url, content, published_at? }] }
|
||||
function buildOllamaSearchRequest(config, params) {
|
||||
const body = { query: params.query, max_results: params.maxResults };
|
||||
if (params.country) body.country = params.country;
|
||||
if (params.language) body.language = params.language;
|
||||
return {
|
||||
url: resolveBaseUrl(config, params),
|
||||
init: {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── GLM Coding plan MCP web_search_prime ──────────────────────────────────
|
||||
// POST https://api.z.ai/api/mcp/web_search_prime/mcp
|
||||
// JSON-RPC envelope: { jsonrpc, id, method: "tools/call",
|
||||
// params: { name: "web_search_prime", arguments: { search_query, count } } }
|
||||
// Response: { result: { content: [{ type: "text", text: "<json>" }] } }
|
||||
function buildGlmSearchRequest(config, params) {
|
||||
const body = {
|
||||
jsonrpc: "2.0",
|
||||
id: `9r-${Date.now()}`,
|
||||
method: "tools/call",
|
||||
params: {
|
||||
name: "web_search_prime",
|
||||
arguments: { search_query: params.query, count: params.maxResults },
|
||||
},
|
||||
};
|
||||
return {
|
||||
url: resolveBaseUrl(config, params),
|
||||
init: {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── Dispatcher ──────────────────────────────────────────────────────────
|
||||
|
||||
const BUILDERS = {
|
||||
@@ -360,6 +435,9 @@ const BUILDERS = {
|
||||
"searchapi": buildSearchApiRequest,
|
||||
"youcom": buildYouComRequest,
|
||||
"searxng": buildSearxngRequest,
|
||||
"xquik": buildXquikRequest,
|
||||
"ollama-search": buildOllamaSearchRequest,
|
||||
"glm": buildGlmSearchRequest,
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
/**
|
||||
* Wrap chat-completions endpoints (with built-in web search) into the unified
|
||||
* /v1/search response format. Supports gemini, openai, xai, kimi, minimax, perplexity.
|
||||
* /v1/search response format. Supports gemini, antigravity, openai, xai, kimi,
|
||||
* minimax, perplexity.
|
||||
*/
|
||||
import { PROVIDER_MEDIA } from "../../providers/index.js";
|
||||
import { ANTIGRAVITY_IDE_USER_AGENT } from "../../providers/shared.js";
|
||||
|
||||
// Default search model + endpoint derive from registry searchViaChat (single source)
|
||||
const searchModel = (id) => PROVIDER_MEDIA[id]?.searchViaChat?.defaultModel;
|
||||
@@ -28,13 +30,37 @@ function toResult(c, index, provider, retrievedAt) {
|
||||
score: null,
|
||||
published_at: null,
|
||||
favicon_url: null,
|
||||
content: null,
|
||||
content: c.content || null,
|
||||
metadata: {},
|
||||
citation: { provider, retrieved_at: retrievedAt, rank: index + 1 },
|
||||
provider_raw: null
|
||||
};
|
||||
}
|
||||
|
||||
// Antigravity search request envelope (mirrors the IDE client)
|
||||
const AG_CLIENT_NAME = "antigravity";
|
||||
const AG_SEARCH_GENERATION_CONFIG = { temperature: 1.0, maxOutputTokens: 8192 };
|
||||
const AG_CONTEXT_BEFORE = 150;
|
||||
const AG_CONTEXT_AFTER = 250;
|
||||
|
||||
/** Widen a grounded segment to its surrounding sentence(s) in the answer text. */
|
||||
function expandSegment(text, segment) {
|
||||
const { startIndex, endIndex } = segment || {};
|
||||
if (!text || !Number.isInteger(startIndex) || !Number.isInteger(endIndex)) return "";
|
||||
const start = Math.max(0, startIndex - AG_CONTEXT_BEFORE);
|
||||
const end = Math.min(text.length, endIndex + AG_CONTEXT_AFTER);
|
||||
let out = text.slice(start, end).trim();
|
||||
// Drop the partial words the window cut off at either edge
|
||||
if (start > 0) out = `...${out.replace(/^\S+/, "")}`;
|
||||
if (end < text.length) out = `${out.replace(/\S+$/, "")}...`;
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
/** Join deduped grounding pieces, skipping empties. */
|
||||
function joinPieces(set, sep) {
|
||||
return [...(set || [])].filter(Boolean).join(sep).trim();
|
||||
}
|
||||
|
||||
/** Coerce a citation that might be a raw URL string or an object. */
|
||||
function normalizeCitation(c) {
|
||||
if (!c) return null;
|
||||
@@ -46,6 +72,8 @@ function normalizeCitation(c) {
|
||||
/**
|
||||
* Provider-specific configuration map. All providers must implement:
|
||||
* { endpoint, defaultModel, buildBody, buildHeaders, extractAnswer }
|
||||
* Optional: requireCredentials(credentials) → error string when a provider needs
|
||||
* more than a token (returns null when satisfied).
|
||||
*/
|
||||
const CHAT_SEARCH_CONFIG = {
|
||||
gemini: {
|
||||
@@ -73,6 +101,71 @@ const CHAT_SEARCH_CONFIG = {
|
||||
}
|
||||
},
|
||||
|
||||
antigravity: {
|
||||
endpoint: () => searchEndpoint("antigravity"),
|
||||
// Upstream 403s on a missing or fabricated project — surface the real cause
|
||||
requireCredentials: (credentials) =>
|
||||
credentials?.projectId ? null : "Antigravity account has no projectId — reconnect the account",
|
||||
buildBody: (query, model, credentials) => ({
|
||||
project: credentials.projectId,
|
||||
model,
|
||||
userAgent: AG_CLIENT_NAME,
|
||||
requestType: "search",
|
||||
request: {
|
||||
contents: [{ role: "user", parts: [{ text: query }] }],
|
||||
tools: [{ googleSearch: {} }],
|
||||
generationConfig: AG_SEARCH_GENERATION_CONFIG
|
||||
}
|
||||
}),
|
||||
buildHeaders: (token) => ({
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${token}`,
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT
|
||||
}),
|
||||
extractAnswer: (data) => {
|
||||
// Antigravity wraps the Gemini payload in { response: {...} }
|
||||
const response = data?.response || data;
|
||||
const candidate = response?.candidates?.[0];
|
||||
const parts = candidate?.content?.parts || [];
|
||||
const text = parts.map((p) => p?.text || "").filter(Boolean).join("");
|
||||
const grounding = candidate?.groundingMetadata || {};
|
||||
const chunks = grounding.groundingChunks || [];
|
||||
const supports = grounding.groundingSupports || [];
|
||||
|
||||
// Upstream repeats the same source across chunks — key by URL so it stays one citation.
|
||||
// Map, not a plain object: both the index and the URL come from upstream.
|
||||
const sources = new Map();
|
||||
const byIndex = chunks.map((ch) => {
|
||||
const web = ch?.web;
|
||||
const url = web?.uri || web?.url || "";
|
||||
if (!url) return null;
|
||||
if (!sources.has(url)) sources.set(url, { title: web.title || "", snippets: new Set(), contexts: new Set() });
|
||||
return sources.get(url);
|
||||
});
|
||||
|
||||
// Each support ties a sentence of the answer back to the chunks that grounded it
|
||||
for (const s of supports) {
|
||||
const segment = s?.segment;
|
||||
const grounded = segment?.text || "";
|
||||
const expanded = expandSegment(text, segment) || grounded;
|
||||
for (const idx of s?.groundingChunkIndices || []) {
|
||||
const source = Number.isInteger(idx) ? byIndex[idx] : null;
|
||||
if (!source) continue;
|
||||
if (grounded) source.snippets.add(grounded);
|
||||
if (expanded) source.contexts.add(expanded);
|
||||
}
|
||||
}
|
||||
|
||||
const citations = [...sources].map(([url, src]) => {
|
||||
const snippet = joinPieces(src.snippets, " | ") || src.title;
|
||||
return { url, title: src.title, snippet, content: joinPieces(src.contexts, "\n\n") || snippet };
|
||||
});
|
||||
|
||||
const tokens = response?.usageMetadata?.totalTokenCount || 0;
|
||||
return { text, citations, tokens };
|
||||
}
|
||||
},
|
||||
|
||||
openai: {
|
||||
endpoint: () => searchEndpoint("openai"),
|
||||
buildBody: (query, model) => {
|
||||
@@ -366,13 +459,18 @@ export async function handleChatSearch({
|
||||
};
|
||||
}
|
||||
|
||||
const credentialError = cfg.requireCredentials?.(credentials);
|
||||
if (credentialError) {
|
||||
return { success: false, status: 401, error: credentialError };
|
||||
}
|
||||
|
||||
const limit =
|
||||
Number.isFinite(maxResults) && maxResults > 0
|
||||
? Math.floor(maxResults)
|
||||
: DEFAULT_MAX_RESULTS;
|
||||
const useModel = model || searchModel(provider);
|
||||
const url = cfg.endpoint(useModel);
|
||||
const body = cfg.buildBody(query, useModel);
|
||||
const body = cfg.buildBody(query, useModel, credentials);
|
||||
const headers = cfg.buildHeaders(token);
|
||||
|
||||
const controller = new AbortController();
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
import { buildSearchRequest } from "./callers.js";
|
||||
import { normalizeSearchResponse } from "./normalizers.js";
|
||||
import { handleChatSearch } from "./chatSearch.js";
|
||||
import { fetchPublic } from "../../../src/shared/utils/ssrfGuard.js";
|
||||
|
||||
const GLOBAL_TIMEOUT_MS = 15000;
|
||||
const NON_RETRIABLE = new Set([400, 401, 403, 404]);
|
||||
@@ -100,7 +101,7 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
|
||||
log?.info?.("SEARCH", `${provider.id} | "${params.query.slice(0, 80)}" | type=${params.searchType}`);
|
||||
|
||||
try {
|
||||
const resp = await fetch(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
|
||||
const resp = await fetchPublic(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
|
||||
clearTimeout(timer);
|
||||
if (!resp.ok) {
|
||||
const errText = await resp.text().catch(() => "");
|
||||
@@ -111,6 +112,13 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
|
||||
const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType);
|
||||
const results = normalized.results.slice(0, params.maxResults);
|
||||
const duration = Date.now() - startTime;
|
||||
const usage = {
|
||||
queries_used: 1,
|
||||
search_cost_usd: providerConfig.costPerQuery ?? null,
|
||||
};
|
||||
if (Number.isFinite(providerConfig.creditsPerResult)) {
|
||||
usage.provider_credits_used = results.length * providerConfig.creditsPerResult;
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
@@ -119,7 +127,8 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
|
||||
query: params.query,
|
||||
results,
|
||||
answer: null,
|
||||
usage: { queries_used: 1, search_cost_usd: providerConfig.costPerQuery || 0 },
|
||||
usage,
|
||||
...(normalized.pagination ? { pagination: normalized.pagination } : {}),
|
||||
metrics: { response_time_ms: duration, upstream_latency_ms: duration, total_results_available: normalized.totalResults },
|
||||
errors: []
|
||||
}
|
||||
|
||||
@@ -199,6 +199,89 @@ function normalizeSearxng(data, _query, _searchType) {
|
||||
return { results, totalResults: results.length };
|
||||
}
|
||||
|
||||
function normalizeXquik(data, _query, _searchType) {
|
||||
const now = new Date().toISOString();
|
||||
const items = Array.isArray(data.tweets) ? data.tweets : [];
|
||||
const results = items.map((item, idx) => {
|
||||
const username = typeof item?.author?.username === "string" ? item.author.username : "";
|
||||
const authorName = typeof item?.author?.name === "string" ? item.author.name : "";
|
||||
const tweetId = typeof item?.id === "string" ? item.id : String(item?.id || "");
|
||||
const url = username && tweetId
|
||||
? `https://x.com/${encodeURIComponent(username)}/status/${encodeURIComponent(tweetId)}`
|
||||
: tweetId
|
||||
? `https://x.com/i/web/status/${encodeURIComponent(tweetId)}`
|
||||
: "";
|
||||
const author = username ? `@${username}` : authorName || null;
|
||||
const title = author ? `${author} on X` : "X post";
|
||||
const imageUrl = Array.isArray(item?.media)
|
||||
? item.media.find((media) => typeof media?.mediaUrl === "string")?.mediaUrl
|
||||
: null;
|
||||
|
||||
return makeResult("xquik", {
|
||||
title,
|
||||
url,
|
||||
snippet: typeof item?.text === "string" ? item.text : "",
|
||||
published_at: typeof item?.createdAt === "string" ? item.createdAt : null,
|
||||
author,
|
||||
image_url: imageUrl || null,
|
||||
source_type: "x_post",
|
||||
full_text: typeof item?.text === "string" ? item.text : undefined,
|
||||
text_format: "text",
|
||||
}, idx, now);
|
||||
});
|
||||
const nextCursor = typeof data.next_cursor === "string" && data.next_cursor ? data.next_cursor : null;
|
||||
return {
|
||||
results,
|
||||
totalResults: null,
|
||||
pagination: {
|
||||
has_more: data.has_next_page === true,
|
||||
next_cursor: nextCursor,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function normalizeOllamaSearch(data, _query, _searchType) {
|
||||
const now = new Date().toISOString();
|
||||
const items = Array.isArray(data?.results) ? data.results : (Array.isArray(data) ? data : []);
|
||||
const results = items.map((item, idx) =>
|
||||
makeResult("ollama-search", {
|
||||
title: item.title,
|
||||
url: item.url,
|
||||
snippet: item.content || item.snippet || "",
|
||||
full_text: item.content,
|
||||
text_format: "text",
|
||||
published_at: item.published_at || null,
|
||||
source_type: item.source || null,
|
||||
}, idx, now)
|
||||
);
|
||||
return { results, totalResults: results.length };
|
||||
}
|
||||
|
||||
function normalizeGlmSearch(data, _query, _searchType) {
|
||||
const now = new Date().toISOString();
|
||||
// MCP envelope: { result: { content: [{ type: "text", text: "<json>" }] } }
|
||||
let payload = data;
|
||||
const textContent = data?.result?.content?.[0]?.text;
|
||||
if (typeof textContent === "string") {
|
||||
try { payload = JSON.parse(textContent); } catch { payload = {}; }
|
||||
}
|
||||
const items = Array.isArray(payload?.results) ? payload.results
|
||||
: Array.isArray(payload?.news) ? payload.news
|
||||
: Array.isArray(payload) ? payload
|
||||
: [];
|
||||
const results = items.map((item, idx) =>
|
||||
makeResult("glm", {
|
||||
title: item.title,
|
||||
url: item.link || item.url,
|
||||
snippet: item.content || "",
|
||||
published_at: item.publish_date || item.published_at || null,
|
||||
favicon_url: item.icon || null,
|
||||
source_type: item.media || null,
|
||||
}, idx, now)
|
||||
);
|
||||
return { results, totalResults: results.length };
|
||||
}
|
||||
|
||||
const NORMALIZERS = {
|
||||
"serper": normalizeSerper,
|
||||
"brave-search": normalizeBrave,
|
||||
@@ -210,11 +293,14 @@ const NORMALIZERS = {
|
||||
"searchapi": normalizeSearchApi,
|
||||
"youcom": normalizeYouCom,
|
||||
"searxng": normalizeSearxng,
|
||||
"xquik": normalizeXquik,
|
||||
"ollama-search": normalizeOllamaSearch,
|
||||
"glm": normalizeGlmSearch,
|
||||
};
|
||||
|
||||
/**
|
||||
* Dispatch to the appropriate normalizer based on providerId.
|
||||
* @returns {{results: Array, totalResults: number|null}}
|
||||
* @returns {{results: Array, totalResults: number|null, pagination?: object}}
|
||||
*/
|
||||
export function normalizeSearchResponse(providerId, data, query, searchType) {
|
||||
const fn = NORMALIZERS[providerId];
|
||||
|
||||
@@ -6,6 +6,16 @@
|
||||
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
|
||||
// 4. DEFAULT_CAPABILITIES — safe floor (always returned)
|
||||
//
|
||||
// Two extra layers then refine the result, and neither can override the hand
|
||||
// written tables above (steps 1-2 short-circuit before they are consulted):
|
||||
// • the synced catalog — modalities keyed by model, limits keyed by provider
|
||||
// + model, refreshed from models.dev in the background. It reads a file, so
|
||||
// the server installs it via setCatalogSource(); this module stays free of
|
||||
// node:fs because the dashboard bundles it into the browser too.
|
||||
// • visionPatterns.js — name-based vision detection, last resort so a model
|
||||
// nobody has catalogued yet still accepts images.
|
||||
// Both only ever turn a capability ON.
|
||||
//
|
||||
// ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
|
||||
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+
|
||||
// models, MIT). Each model exposes the exact fields we map below:
|
||||
@@ -23,6 +33,7 @@
|
||||
// 2.0+, Grok, Perplexity). Verify with: curl -s https://models.dev/api.json
|
||||
|
||||
import { matchPattern } from "./pricing.js";
|
||||
import { looksLikeVisionModel } from "./visionPatterns.js";
|
||||
|
||||
/**
|
||||
* Safe floor — every resolved result is merged over this so consumers
|
||||
@@ -46,6 +57,7 @@ export const DEFAULT_CAPABILITIES = {
|
||||
thinkingFormat: null,
|
||||
thinkingCanDisable: true, // false → model cannot turn thinking off (clamp to min instead of disable)
|
||||
thinkingRange: null, // { min, max } for budget formats; null = no clamp
|
||||
thinkingEffortSupported: false, // zai format only: model accepts a reasoning_effort level (GLM-5.2+; older GLM ignores it)
|
||||
// limits (tokens)
|
||||
contextWindow: 200000,
|
||||
maxOutput: 64000,
|
||||
@@ -71,7 +83,8 @@ export function capabilitiesFromServiceKind(kind) {
|
||||
* otherwise mis-match. Only declare deltas vs DEFAULT.
|
||||
*/
|
||||
export const MODEL_CAPABILITIES = {
|
||||
// Claude Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
|
||||
// Claude Fable 5.1, Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
|
||||
"claude-fable-5-1": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
@@ -94,8 +107,14 @@ export const MODEL_CAPABILITIES = {
|
||||
// Gemini image-gen / OpenAI image / xai image variants
|
||||
"gpt-image-1": { imageOutput: true, tools: false },
|
||||
|
||||
// GLM vision variant (text GLM has no vision)
|
||||
"glm-4.6v": { vision: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000 },
|
||||
// GLM vision variants (text GLM has no vision) — 5.3-Flash and 5V-Turbo are
|
||||
// natively multimodal per z.ai, and 5.3-Flash carries the full 1M window.
|
||||
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "zai", contextWindow: 1000000, maxOutput: 131072 },
|
||||
"glm-4.6v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000, maxOutput: 32768 },
|
||||
"glm-4.5v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 64000, maxOutput: 16384 },
|
||||
|
||||
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
|
||||
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
|
||||
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
|
||||
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
@@ -108,6 +127,10 @@ export const MODEL_CAPABILITIES = {
|
||||
"kimi-for-coding-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
"kimi-k2.7-code": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
"kimi-k2.7-code-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
// OpenCode Free Muse Spark — multimodal (text+image per models.dev meta/muse-spark)
|
||||
// via OpenAI Responses input_image; reasoning supports up to xhigh.
|
||||
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
};
|
||||
|
||||
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
@@ -131,6 +154,7 @@ export const PROVIDER_CAPABILITIES = {
|
||||
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
},
|
||||
"codex": {
|
||||
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
|
||||
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
|
||||
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
|
||||
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
|
||||
@@ -155,24 +179,69 @@ export const PROVIDER_CAPABILITIES = {
|
||||
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
|
||||
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
|
||||
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
|
||||
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking
|
||||
// off → thinkingCanDisable:false (clamped to minimal instead of disabled).
|
||||
// (see registry thinkingFormat). For thinkingCanDisable use the server's
|
||||
// reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
|
||||
// below; it is NOT the inverse of onlyReasoning.
|
||||
"codebuddy-cn": {
|
||||
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 },
|
||||
// maxOutput 64000 per both the plugin-baked fallback and the live server
|
||||
// table (the old 38000 had no source and truncated output).
|
||||
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
|
||||
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
|
||||
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 },
|
||||
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
|
||||
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
|
||||
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
|
||||
// Per-model values mirror the server's product-config payload (the plugin
|
||||
// fetches it from copilot.tencent.com; the `models[]` entries carry
|
||||
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
|
||||
// maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
|
||||
// plugin-baked fallback disagree, the server table wins.
|
||||
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
|
||||
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
|
||||
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
|
||||
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
|
||||
// their thinking is switchable; the hy* models are forced always-on.
|
||||
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
|
||||
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
|
||||
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
|
||||
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
|
||||
},
|
||||
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
|
||||
// registry `name` is display-only and capability lookup matches on the raw
|
||||
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
|
||||
// (200K) without this map. contextWindow follows the real model family's
|
||||
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
|
||||
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
|
||||
// max_output_tokens arrives as 0 for every model, so outputs are
|
||||
// best-guess from the real model family. Vision tags below follow the
|
||||
// upstream is_vl flag per explicit request, even though the executor
|
||||
// currently sends image_urls:null (image pass-through over the agent_chat
|
||||
// SSE protocol is unverified). reasoning:true on all of them — every model can
|
||||
// reason; the upstream is_reasoning flag only drives model_config selection.
|
||||
// thinkingFormat keeps the true-model family for documentation/UI, but
|
||||
// thinkingCanDisable:false everywhere: the executor only forwards
|
||||
// messages/tools/max_tokens, and thinking is fixed upstream via
|
||||
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
|
||||
// must never be offered as an option.
|
||||
"qoder": {
|
||||
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
|
||||
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
|
||||
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
|
||||
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
|
||||
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
|
||||
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
|
||||
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
|
||||
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
|
||||
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
|
||||
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
|
||||
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
|
||||
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
|
||||
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
|
||||
},
|
||||
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
|
||||
"poolside": {
|
||||
@@ -205,6 +274,7 @@ export const PATTERN_CAPABILITIES = [
|
||||
|
||||
// ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─
|
||||
{ pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } },
|
||||
{ pattern: "*gemini-3.8*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } },
|
||||
{ pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
@@ -214,6 +284,9 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
|
||||
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
|
||||
|
||||
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
|
||||
|
||||
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
@@ -234,6 +307,8 @@ export const PATTERN_CAPABILITIES = [
|
||||
// ── Grok (vision + Live Search) ──────────────────────────────────
|
||||
{ pattern: "*grok*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*grok-code*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
// Grok 4.6: 500k context, no text output limit (docs.x.ai/developers/grok-4-6)
|
||||
{ pattern: "*grok-4.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 500000 } },
|
||||
// Grok 4.5 (Grok CLI / Grok Build): 500k context per cli-chat-proxy /v1/models
|
||||
{ pattern: "*grok-4.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 64000 } },
|
||||
{ pattern: "*grok-4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
@@ -261,6 +336,10 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*kimi*", caps: { reasoning: true, thinkingFormat: "kimi", contextWindow: 262144 } },
|
||||
|
||||
// ── GLM / Z.ai (thinking.enabled; disable via enable_thinking:false) ─
|
||||
// reasoning_effort is only read by z.ai from GLM-5.2 onward (docs.z.ai/guides/capabilities/thinking) —
|
||||
// older GLM (4.x, 5.0, 5.1, 5-turbo, 5v-turbo) ignore it, so gate it per exact version, not the "*glm-5*" catch-all.
|
||||
{ pattern: "*glm-5.3*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-5.2*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-5*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-4.7*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-4*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
|
||||
@@ -309,6 +388,9 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*laguna-s-2.1*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 } },
|
||||
{ pattern: "*laguna*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 } },
|
||||
|
||||
|
||||
// ── OpenCode Free Muse Spark (multimodal text+image; OpenAI Responses reasoning supports up to xhigh) ─
|
||||
{ pattern: "*muse*spark*", caps: { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 } },
|
||||
// ── Others ───────────────────────────────────────────────────────
|
||||
{ pattern: "*hunyuan*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
|
||||
{ pattern: "hy3*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
|
||||
@@ -317,6 +399,11 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*ling-*", caps: { reasoning: true, contextWindow: 128000 } },
|
||||
];
|
||||
|
||||
// OpenRouter-style gateways validate modalities upstream — a text-only model
|
||||
// sent an image gets a clear upstream error instead of silent corruption. So for
|
||||
// unknown models on these providers, trust vision instead of stripping images.
|
||||
const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
|
||||
|
||||
/**
|
||||
* Resolve capabilities for a model using the 4-step fallback chain,
|
||||
* merged over DEFAULT_CAPABILITIES so the result is always complete.
|
||||
@@ -325,6 +412,46 @@ export const PATTERN_CAPABILITIES = [
|
||||
* @param {string} model
|
||||
* @returns {object} full capabilities object
|
||||
*/
|
||||
const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
|
||||
// Catalog lookups, installed by the server at startup. Left as no-ops in the
|
||||
// browser bundle, where there is no file to read.
|
||||
let catalogSource = null;
|
||||
|
||||
/**
|
||||
* Install the synced catalog reader (server only).
|
||||
* @param {{ getModalities: Function, getLimits: Function } | null} source
|
||||
*/
|
||||
export function setCatalogSource(source) {
|
||||
catalogSource = source;
|
||||
}
|
||||
|
||||
// Apply the synced catalog + name heuristic on top of a table-resolved result.
|
||||
// Strictly additive: a capability already true stays true, and a false one only
|
||||
// flips when an outside source positively declares support.
|
||||
function refine(base, provider, model) {
|
||||
const result = { ...DEFAULT_CAPABILITIES, ...base };
|
||||
|
||||
if (catalogSource) {
|
||||
const modalities = catalogSource.getModalities(model);
|
||||
if (modalities) {
|
||||
for (const key of MODALITY_KEYS) {
|
||||
if (modalities[key] === true) result[key] = true;
|
||||
}
|
||||
}
|
||||
|
||||
const limits = catalogSource.getLimits(provider, model);
|
||||
if (limits) {
|
||||
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
|
||||
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
|
||||
}
|
||||
}
|
||||
|
||||
if (!result.vision && looksLikeVisionModel(model)) result.vision = true;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
export function getCapabilitiesForModel(provider, model) {
|
||||
if (!model) return { ...DEFAULT_CAPABILITIES };
|
||||
|
||||
@@ -342,13 +469,13 @@ export function getCapabilitiesForModel(provider, model) {
|
||||
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
|
||||
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
|
||||
|
||||
// 3. Pattern match (first match wins)
|
||||
// 3. Pattern match (first match wins), refined by catalog + name heuristic
|
||||
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
|
||||
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
|
||||
return { ...DEFAULT_CAPABILITIES, ...caps };
|
||||
return refine(caps, provider, model);
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Floor
|
||||
return { ...DEFAULT_CAPABILITIES };
|
||||
return refine(null, provider, model);
|
||||
}
|
||||
|
||||
72
open-sse/providers/catalogOverride.js
Normal file
72
open-sse/providers/catalogOverride.js
Normal file
@@ -0,0 +1,72 @@
|
||||
// Read side of the model catalog synced from models.dev.
|
||||
//
|
||||
// The file is the source of truth; the only thing held in memory is a parsed
|
||||
// copy dropped as soon as the file's mtime changes. getCapabilitiesForModel is
|
||||
// synchronous and runs per request, so the hot path is one stat (~1us) and the
|
||||
// parse (~0.1ms on a ~18KB file) only reruns after a sync.
|
||||
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import { DATA_DIR } from "@/lib/dataDir.js";
|
||||
|
||||
export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
|
||||
// Trimmed upstream catalog, read by the add-models skill (not by the router).
|
||||
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
|
||||
|
||||
const EMPTY = { models: {}, providers: {} };
|
||||
let cache = EMPTY;
|
||||
let cachedMtime = -1;
|
||||
|
||||
// "zai-org/GLM-4.6V:free" -> "glm-4.6v"
|
||||
function baseId(model) {
|
||||
if (!model) return "";
|
||||
const withoutVendor = model.includes("/") ? model.split("/").pop() : model;
|
||||
return withoutVendor.toLowerCase().split(":")[0];
|
||||
}
|
||||
|
||||
function load() {
|
||||
let mtime;
|
||||
try {
|
||||
mtime = fs.statSync(CATALOG_FILE).mtimeMs;
|
||||
} catch {
|
||||
cache = EMPTY;
|
||||
cachedMtime = -1;
|
||||
return cache;
|
||||
}
|
||||
if (mtime === cachedMtime) return cache;
|
||||
|
||||
cachedMtime = mtime;
|
||||
try {
|
||||
const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8"));
|
||||
cache = { models: parsed?.models || {}, providers: parsed?.providers || {} };
|
||||
} catch {
|
||||
cache = EMPTY;
|
||||
}
|
||||
return cache;
|
||||
}
|
||||
|
||||
// Modality is a property of the model itself — any gateway serving it inherits
|
||||
// the same image/video/pdf support, so this is keyed by model id alone.
|
||||
export function getCatalogModalities(model) {
|
||||
return load().models[baseId(model)] || null;
|
||||
}
|
||||
|
||||
// Context and output limits are a property of the gateway, not the model: each
|
||||
// one truncates differently, so these stay keyed by provider + model.
|
||||
export function getCatalogLimits(provider, model) {
|
||||
const byProvider = provider && load().providers[provider];
|
||||
if (!byProvider) return null;
|
||||
return byProvider[model] || byProvider[baseId(model)] || null;
|
||||
}
|
||||
|
||||
// Force a re-read on the next lookup (called right after a sync writes the file).
|
||||
export function invalidateCatalog() {
|
||||
cachedMtime = -1;
|
||||
}
|
||||
|
||||
// Hand the reader to capabilities.js. That module is bundled into the browser
|
||||
// too, so it cannot import this file directly — the server pushes it in.
|
||||
export async function installCatalogSource() {
|
||||
const { setCatalogSource } = await import("./capabilities.js");
|
||||
setCatalogSource({ getModalities: getCatalogModalities, getLimits: getCatalogLimits });
|
||||
}
|
||||
@@ -18,3 +18,10 @@ export function withCodexReviewModels(models) {
|
||||
];
|
||||
});
|
||||
}
|
||||
|
||||
export function isMuseSparkModel(modelId) {
|
||||
if (!modelId || typeof modelId !== "string") return false;
|
||||
const clean = modelId.replace(/\([^()]+\)\s*$/, "").trim();
|
||||
const base = clean.includes("/") ? clean.split("/").pop() : clean;
|
||||
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
|
||||
}
|
||||
|
||||
@@ -53,10 +53,15 @@ export const MODEL_PRICING = {
|
||||
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
|
||||
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
|
||||
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
|
||||
// === Gemini ===
|
||||
"gemini-3.8-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.8-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.8-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.8-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
@@ -260,6 +265,7 @@ export const PROVIDER_PRICING = {
|
||||
"z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 },
|
||||
"z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 },
|
||||
"z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 },
|
||||
"z-ai/glm-5.3-free": { input: 0, output: 0, cached: 0, reasoning: 0 },
|
||||
},
|
||||
};
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ export default {
|
||||
deprecationNotice: "RISK_NOTICE",
|
||||
},
|
||||
category: "oauth",
|
||||
serviceKinds: ["llm", "image"],
|
||||
serviceKinds: ["llm", "image", "webSearch"],
|
||||
transport: {
|
||||
baseUrls: [ANTIGRAVITY_IDE_BASE_URL],
|
||||
format: "antigravity",
|
||||
@@ -36,8 +36,7 @@ export default {
|
||||
},
|
||||
},
|
||||
usage: {
|
||||
// Discovery (quota/project) on PROD; daily host rejects these.
|
||||
quotaApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels",
|
||||
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
|
||||
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
|
||||
tokenUrl: "https://oauth2.googleapis.com/token",
|
||||
},
|
||||
@@ -45,6 +44,10 @@ export default {
|
||||
clientSecret: "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",
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.8-flash-high", name: "Gemini 3.8 Flash (High)", upstreamModelId: "gemini-3.8-flash-high(high)" },
|
||||
{ id: "gemini-3.8-flash-medium", name: "Gemini 3.8 Flash (Medium)", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
|
||||
{ id: "gemini-3.8-flash-low", name: "Gemini 3.8 Flash (Low)", upstreamModelId: "gemini-3.8-flash-low(low)" },
|
||||
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
|
||||
{ id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" },
|
||||
{ id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" },
|
||||
@@ -82,6 +85,11 @@ export default {
|
||||
loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT,
|
||||
refreshLeadMs: 300000,
|
||||
},
|
||||
searchViaChat: {
|
||||
defaultModel: "gemini-2.5-flash",
|
||||
endpoint: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:generateContent`,
|
||||
freeTier: "Free — Google Search grounding through an Antigravity OAuth account.",
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
},
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { CLAUDE_CLI_SPOOF_HEADERS } from "../shared.js";
|
||||
import { CLAUDE_CLI_VERSION } from "../shared.js";
|
||||
|
||||
export default {
|
||||
id: "claude",
|
||||
@@ -25,7 +25,7 @@ export default {
|
||||
"Anthropic-Version": "2023-06-01",
|
||||
"Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28",
|
||||
"Anthropic-Dangerous-Direct-Browser-Access": "true",
|
||||
"User-Agent": "claude-cli/2.1.92 (external, sdk-cli)",
|
||||
"User-Agent": `claude-cli/${CLAUDE_CLI_VERSION} (external, sdk-cli)`,
|
||||
"X-App": "cli",
|
||||
"X-Stainless-Helper-Method": "stream",
|
||||
"X-Stainless-Retry-Count": "0",
|
||||
@@ -58,6 +58,7 @@ export default {
|
||||
},
|
||||
models: [
|
||||
{ id: "claude-opus-5", name: "Claude Opus 5" },
|
||||
{ id: "claude-fable-5-1", name: "Claude Fable 5.1" },
|
||||
{ id: "claude-fable-5", name: "Claude Fable 5" },
|
||||
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" },
|
||||
|
||||
@@ -47,19 +47,26 @@ export default {
|
||||
models: [
|
||||
{ id: "glm-5.2", name: "GLM-5.2" },
|
||||
{ id: "glm-5.1", name: "GLM-5.1" },
|
||||
{ id: "glm-5.0", name: "GLM-5.0" },
|
||||
{ id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" },
|
||||
{ id: "glm-5v-turbo", name: "GLM-5v-Turbo" },
|
||||
{ id: "glm-4.7", name: "GLM-4.7" },
|
||||
{ id: "minimax-m3", name: "MiniMax-M3" },
|
||||
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
|
||||
{ id: "kimi-k2.7", name: "Kimi-K2.7-Code" },
|
||||
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
|
||||
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
|
||||
{ id: "hy3-preview", name: "Hy3 Preview" },
|
||||
// Catalog mirrors the server's product-config payload (the plugin fetches
|
||||
// it from copilot.tencent.com). Models the server no longer publishes are
|
||||
// removed even when the chat endpoint still answers them — the published
|
||||
// list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x
|
||||
// (endpoint returns 11102 "model service info not found"), plus
|
||||
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
|
||||
// deepseek-v3-2-volc (absent from the server list, though still answering
|
||||
// 200) and hy3-x (paid tier, not used here).
|
||||
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
|
||||
{ id: "hy3", name: "Hy3" },
|
||||
{ id: "hy4-preview", name: "Hy4-Preview" },
|
||||
{ id: "glm-5.3", name: "GLM-5.3" },
|
||||
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
|
||||
{ id: "kimi-k3-1", name: "Kimi-K3" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
|
||||
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
|
||||
],
|
||||
oauth: {
|
||||
baseUrl: "https://copilot.tencent.com",
|
||||
|
||||
@@ -45,6 +45,7 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
|
||||
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
|
||||
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
|
||||
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
|
||||
@@ -59,6 +60,9 @@ export default {
|
||||
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
|
||||
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
|
||||
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
|
||||
@@ -23,21 +23,55 @@ export default {
|
||||
format: "commandcode",
|
||||
forceStream: true,
|
||||
headers: {
|
||||
"x-command-code-version": "0.25.7",
|
||||
"x-command-code-version": "0.45.0",
|
||||
"x-cli-environment": "cli",
|
||||
"User-Agent": "cli",
|
||||
},
|
||||
// Quota/billing endpoints (same alpha API the official CLI /usage calls).
|
||||
// whoami resolves orgId; credits+subscription+usage/summary then report the
|
||||
// 5-hour/weekly windows, plan, and period credits. See services/usage/commandcode.js.
|
||||
usage: {
|
||||
baseUrl: "https://api.commandcode.ai",
|
||||
whoamiUrl: "/alpha/whoami",
|
||||
creditsUrl: "/alpha/billing/credits",
|
||||
subscriptionsUrl: "/alpha/billing/subscriptions",
|
||||
summaryUrl: "/alpha/usage/summary",
|
||||
},
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
models: [
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "moonshotai/Kimi-K2.7-Code", name: "Kimi K2.7 Code" },
|
||||
{ id: "moonshotai/Kimi-K2.7-Code-Highspeed", name: "Kimi K2.7 Code Highspeed" },
|
||||
{ id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6" },
|
||||
{ id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5" },
|
||||
{ id: "zai-org/GLM-5.2", name: "GLM 5.2" },
|
||||
{ id: "zai-org/GLM-5.2-Fast", name: "GLM 5.2 Fast" },
|
||||
{ id: "zai-org/GLM-5.1", name: "GLM 5.1" },
|
||||
{ id: "zai-org/GLM-5", name: "GLM 5" },
|
||||
{ id: "MiniMaxAI/MiniMax-M3", name: "MiniMax M3" },
|
||||
{ id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7" },
|
||||
{ id: "MiniMaxAI/MiniMax-M2.5", name: "MiniMax M2.5" },
|
||||
{ id: "xiaomi/mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
|
||||
{ id: "xiaomi/mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "Qwen/Qwen3.7-Max", name: "Qwen 3.7 Max" },
|
||||
{ id: "Qwen/Qwen3.7-Plus", name: "Qwen 3.7 Plus" },
|
||||
{ id: "Qwen/Qwen3.6-Max-Preview", name: "Qwen 3.6 Max Preview" },
|
||||
{ id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" },
|
||||
{ id: "stepfun/Step-3.7-Flash", name: "Step 3.7 Flash" },
|
||||
{ id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" },
|
||||
{ id: "tencent/Hy3", name: "Tencent Hy3" },
|
||||
{ id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B A55B" },
|
||||
{ id: "thinkingmachines/inkling", name: "Inkling" },
|
||||
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "claude-fable-5", name: "Claude Fable 5" },
|
||||
{ id: "claude-opus-4-8", name: "Claude Opus 4.8" },
|
||||
{ id: "claude-opus-4-7", name: "Claude Opus 4.7" },
|
||||
{ id: "claude-haiku-4-5", name: "Claude Haiku 4.5" },
|
||||
],
|
||||
};
|
||||
|
||||
@@ -45,6 +45,7 @@ export default {
|
||||
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
|
||||
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
|
||||
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
|
||||
{ id: "deepseek-reasoner", name: "DeepSeek V3.2 Reasoner" },
|
||||
],
|
||||
|
||||
@@ -36,6 +36,7 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash" },
|
||||
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" },
|
||||
{ id: "gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
|
||||
|
||||
@@ -22,10 +22,13 @@ export default {
|
||||
},
|
||||
models: [
|
||||
{ id: "glm-5.3", name: "GLM 5.3" },
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
{ id: "glm-4.7", name: "GLM-4.7" },
|
||||
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
|
||||
{ id: "glm-4.6", name: "GLM-4.6" },
|
||||
{ id: "glm-4.5-air", name: "GLM-4.5-Air" },
|
||||
],
|
||||
|
||||
@@ -46,12 +46,28 @@ export default {
|
||||
],
|
||||
models: [
|
||||
{ id: "glm-5.3", name: "GLM 5.3" },
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
{ id: "glm-4.7", name: "GLM 4.7" },
|
||||
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
|
||||
],
|
||||
serviceKinds: ["llm", "webSearch"],
|
||||
// Coding plan bundles web search on the same API key as chat.
|
||||
searchConfig: {
|
||||
baseUrl: "https://api.z.ai/api/mcp/web_search_prime/mcp",
|
||||
method: "POST",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
costPerQuery: 0,
|
||||
searchTypes: ["web"],
|
||||
defaultMaxResults: 5,
|
||||
maxMaxResults: 50,
|
||||
timeoutMs: 10000,
|
||||
cacheTTLMs: 300000,
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
|
||||
@@ -17,6 +17,12 @@ export default {
|
||||
transport: {
|
||||
baseUrl: "https://api.groq.com/openai/v1/chat/completions",
|
||||
validateUrl: "https://api.groq.com/openai/v1/models",
|
||||
// No dedicated quota endpoint; rate-limit info rides on x-ratelimit-*
|
||||
// response headers, always included. Reuse the models list (already
|
||||
// used as validateUrl) so reading usage never costs tokens.
|
||||
usage: {
|
||||
url: "https://api.groq.com/openai/v1/models",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "llama-3.3-70b-versatile", name: "Llama 3.3 70B" },
|
||||
@@ -34,4 +40,8 @@ export default {
|
||||
authHeader: "bearer",
|
||||
format: "openai",
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -66,6 +66,7 @@ import p63 from "./nebius.js";
|
||||
import p64 from "./nvidia.js";
|
||||
import p65 from "./ollama-local.js";
|
||||
import p66 from "./ollama.js";
|
||||
import p123 from "./ollama-search.js";
|
||||
import p67 from "./openai.js";
|
||||
import p68 from "./opencode-go.js";
|
||||
import p69 from "./opencode.js";
|
||||
@@ -121,6 +122,7 @@ import p118 from "./selfhosted-tts.js";
|
||||
import p119 from "./selfhosted-embedding.js";
|
||||
import p120 from "./fish-audio.js";
|
||||
import p121 from "./alitp-intl.js";
|
||||
import p122 from "./xquik.js";
|
||||
|
||||
export default [
|
||||
p0,
|
||||
@@ -190,6 +192,7 @@ export default [
|
||||
p64,
|
||||
p65,
|
||||
p66,
|
||||
p123,
|
||||
p67,
|
||||
p68,
|
||||
p69,
|
||||
@@ -243,4 +246,5 @@ export default [
|
||||
p119,
|
||||
p120,
|
||||
p121,
|
||||
p122,
|
||||
];
|
||||
|
||||
35
open-sse/providers/registry/ollama-search.js
Normal file
35
open-sse/providers/registry/ollama-search.js
Normal file
@@ -0,0 +1,35 @@
|
||||
export default {
|
||||
id: "ollama-search",
|
||||
alias: "ollama-search",
|
||||
display: {
|
||||
name: "Ollama Search",
|
||||
icon: "cloud",
|
||||
color: "#ffffff",
|
||||
textIcon: "OL",
|
||||
website: "https://ollama.com",
|
||||
notice: {
|
||||
text: "Web search via Ollama Cloud subscription. Reuses the API key from the Ollama (chat) provider.",
|
||||
apiKeyUrl: "https://ollama.com/settings/keys",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
authModes: ["apikey"],
|
||||
serviceKinds: ["webSearch"],
|
||||
// Credential fallback: reuses the API key registered under the `ollama`
|
||||
// chat provider — one key, chat + search.
|
||||
credentialFallback: "ollama",
|
||||
searchConfig: {
|
||||
baseUrl: "https://ollama.com/api/web_search",
|
||||
method: "POST",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
costPerQuery: 0,
|
||||
freeMonthlyQuota: 1000,
|
||||
searchTypes: ["web"],
|
||||
defaultMaxResults: 5,
|
||||
maxMaxResults: 10,
|
||||
timeoutMs: 10000,
|
||||
cacheTTLMs: 300000,
|
||||
},
|
||||
};
|
||||
@@ -31,7 +31,16 @@ export default {
|
||||
{ id: "qwen3.5", name: "Qwen3.5" },
|
||||
{ id: "minimax-m3", name: "MiniMax M3" },
|
||||
],
|
||||
serviceKinds: ["llm"],
|
||||
serviceKinds: ["llm", "webFetch"],
|
||||
fetchConfig: {
|
||||
baseUrl: "https://ollama.com/api/web_fetch",
|
||||
method: "POST",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
formats: ["markdown"],
|
||||
maxCharacters: 200000,
|
||||
timeoutMs: 30000,
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
|
||||
@@ -21,6 +21,9 @@ export default {
|
||||
transport: {
|
||||
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
|
||||
headers: {},
|
||||
usage: {
|
||||
url: "https://opencode.ai/zen/go/v1/usage",
|
||||
},
|
||||
},
|
||||
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
|
||||
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
|
||||
@@ -31,12 +34,14 @@ export default {
|
||||
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
|
||||
],
|
||||
models: [
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
|
||||
@@ -45,5 +50,13 @@ export default {
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
|
||||
// Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces
|
||||
// chatCore past the sourceFormat-matched transports into translation (see chatCore guard).
|
||||
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
],
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -19,7 +19,12 @@ export default {
|
||||
},
|
||||
noAuth: true,
|
||||
},
|
||||
models: [],
|
||||
models: [
|
||||
// Muse Spark models are served by /zen/v1/responses; the rest stay on
|
||||
// /chat/completions, so the format is declared per-model, not per-provider.
|
||||
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
|
||||
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
|
||||
],
|
||||
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
|
||||
@@ -30,12 +30,15 @@ export default {
|
||||
{ id: "auto", name: "Auto" },
|
||||
{ id: "performance", name: "Performance" },
|
||||
{ id: "efficient", name: "Efficient" },
|
||||
{ id: "qmodel_preview", name: "Qwen3.8-Max-Preview" },
|
||||
{ id: "lite", name: "Lite" },
|
||||
{ id: "qmodel_38max", name: "Qwen3.8-Max" },
|
||||
{ id: "qmodel_latest", name: "Qwen3.7-Max" },
|
||||
{ id: "qmodel", name: "Qwen3.7-Plus" },
|
||||
{ id: "qfmodel", name: "Qwen3.8-Flash" },
|
||||
{ id: "kmodel_latest", name: "Kimi-K3" },
|
||||
{ id: "kmodel", name: "Kimi-K2.7-Code" },
|
||||
{ id: "gm51model", name: "GLM-5.2" },
|
||||
{ id: "gmodel", name: "GLM-5.3" },
|
||||
{ id: "gfmodel", name: "GLM-5.3-Flash" },
|
||||
{ id: "dmodel", name: "DeepSeek-V4-Pro" },
|
||||
{ id: "dfmodel", name: "DeepSeek-V4-Flash" },
|
||||
{ id: "mmodel", name: "MiniMax-M3" },
|
||||
|
||||
@@ -24,129 +24,31 @@ export default {
|
||||
validateUrl: "https://api.tokenrouter.com/v1/models",
|
||||
thinkingFormat: "tokenrouter",
|
||||
},
|
||||
// Seed snapshot from live /v1/models (120 entries). Latest catalogue is
|
||||
// Seed snapshot from live /v1/models. Latest catalogue is
|
||||
// fetched via modelsFetcher; other ids still accepted via passthroughModels.
|
||||
models: [
|
||||
{ id: "MiniMax-Hailuo-2.3", name: "Minimax Hailuo 2.3", kind: "video" },
|
||||
{ id: "MiniMax-M3", name: "Minimax M3" },
|
||||
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5" },
|
||||
{ id: "anthropic/claude-haiku-4.5", name: "Claude Haiku 4.5" },
|
||||
{ id: "anthropic/claude-opus-4.5", name: "Claude Opus 4.5" },
|
||||
{ id: "anthropic/claude-opus-4.6", name: "Claude Opus 4.6" },
|
||||
{ id: "anthropic/claude-opus-4.7", name: "Claude Opus 4.7" },
|
||||
{ id: "anthropic/claude-opus-4.7-fast", name: "Claude Opus 4.7 Fast" },
|
||||
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "anthropic/claude-opus-4.8", name: "Claude Opus 4.8" },
|
||||
{ id: "anthropic/claude-opus-4.8-fast", name: "Claude Opus 4.8 Fast" },
|
||||
{ id: "anthropic/claude-opus-5", name: "Claude Opus 5" },
|
||||
{ id: "anthropic/claude-opus-5-fast", name: "Claude Opus 5 Fast" },
|
||||
{ id: "anthropic/claude-sonnet-4", name: "Claude Sonnet 4" },
|
||||
{ id: "anthropic/claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
|
||||
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "bytedance-seed/seedream-4.5", name: "Seedream 4.5", kind: "image" },
|
||||
{ id: "bytedance-seed/seedream-5.0-lite", name: "Seedream 5.0 Lite", kind: "image" },
|
||||
{ id: "bytedance-seed/seedream-5.0-pro", name: "Seedream 5.0 Pro", kind: "image" },
|
||||
{ id: "claude-haiku-4-5", name: "Claude Haiku 4 5" },
|
||||
{ id: "claude-opus-4-8-m-aws", name: "Claude Opus 4 8 M Aws" },
|
||||
{ id: "deepseek/deepseek-v3.2", name: "Deepseek V3.2" },
|
||||
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-flash-0731", name: "Deepseek V4 Flash 0731" },
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
|
||||
{ id: "ex/gpt-5.4", name: "Gpt 5.4" },
|
||||
{ id: "google/gemini-2.5-flash-image", name: "Gemini 2.5 Flash Image" },
|
||||
{ id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
|
||||
{ id: "google/gemini-3-pro-image-preview", name: "Gemini 3 Pro Image Preview" },
|
||||
{ id: "google/gemini-3.1-flash-image-preview", name: "Gemini 3.1 Flash Image Preview" },
|
||||
{ id: "google/gemini-3.1-flash-lite-image", name: "Gemini 3.1 Flash Lite Image" },
|
||||
{ id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
|
||||
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
|
||||
{ id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
|
||||
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "google/gemini-embedding-2", name: "Gemini Embedding 2" },
|
||||
{ id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B It" },
|
||||
{ id: "happyhorse-1.0-t2v", name: "Happyhorse 1.0 T2V", kind: "video" },
|
||||
{ id: "kling-3.0-turbo", name: "Kling 3.0 Turbo", kind: "video" },
|
||||
{ id: "kling-v2-6", name: "Kling V2 6", kind: "video" },
|
||||
{ id: "kling-v3", name: "Kling V3", kind: "video" },
|
||||
{ id: "kling-v3-omni", name: "Kling V3 Omni", kind: "video" },
|
||||
{ id: "microsoft/mai-image-2.5", name: "Mai Image 2.5" },
|
||||
{ id: "minimax/minimax-m2-her", name: "Minimax M2 Her" },
|
||||
{ id: "minimax/minimax-m2.1", name: "Minimax M2.1" },
|
||||
{ id: "minimax/minimax-m2.1-highspeed", name: "Minimax M2.1 Highspeed" },
|
||||
{ id: "minimax/minimax-m2.5", name: "Minimax M2.5" },
|
||||
{ id: "minimax/minimax-m2.7", name: "Minimax M2.7" },
|
||||
{ id: "minimax/minimax-m2.7-highspeed", name: "Minimax M2.7 Highspeed" },
|
||||
{ id: "miromind/mirothinker-1-7-deepresearch", name: "Mirothinker 1 7 Deepresearch" },
|
||||
{ id: "miromind/mirothinker-1-7-deepresearch-mini", name: "Mirothinker 1 7 Deepresearch Mini" },
|
||||
{ id: "mistralai/devstral-2512", name: "Devstral 2512" },
|
||||
{ id: "mistralai/mistral-medium-3-5", name: "Mistral Medium 3 5" },
|
||||
{ id: "mistralai/mistral-small-2603", name: "Mistral Small 2603" },
|
||||
{ id: "mistralai/voxtral-small-24b-2507", name: "Voxtral Small 24B 2507" },
|
||||
{ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5" },
|
||||
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
|
||||
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "moonshotai/kimi-k3", name: "Kimi K3" },
|
||||
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
|
||||
{ id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", name: "Nemotron 3 Nano Omni 30B A3B Reasoning:Free" },
|
||||
{ id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B" },
|
||||
{ id: "openai/gpt-4o-mini", name: "Gpt 4O Mini" },
|
||||
{ id: "openai/gpt-5", name: "Gpt 5" },
|
||||
{ id: "openai/gpt-5-image", name: "Gpt 5 Image" },
|
||||
{ id: "openai/gpt-5-image-mini", name: "Gpt 5 Image Mini" },
|
||||
{ id: "openai/gpt-5-mini", name: "Gpt 5 Mini" },
|
||||
{ id: "openai/gpt-5.2", name: "Gpt 5.2" },
|
||||
{ id: "openai/gpt-5.4", name: "Gpt 5.4" },
|
||||
{ id: "openai/gpt-5.4-image-2", name: "Gpt 5.4 Image 2", kind: "image" },
|
||||
{ id: "openai/gpt-5.4-mini", name: "Gpt 5.4 Mini" },
|
||||
{ id: "openai/gpt-5.4-nano", name: "Gpt 5.4 Nano" },
|
||||
{ id: "openai/gpt-5.4-pro", name: "Gpt 5.4 Pro" },
|
||||
{ id: "openai/gpt-5.5", name: "Gpt 5.5" },
|
||||
{ id: "openai/gpt-5.5-pro", name: "Gpt 5.5 Pro" },
|
||||
{ id: "openai/gpt-5.6-luna", name: "Gpt 5.6 Luna" },
|
||||
{ id: "openai/gpt-5.6-sol", name: "Gpt 5.6 Sol" },
|
||||
{ id: "openai/gpt-5.6-terra", name: "Gpt 5.6 Terra" },
|
||||
{ id: "openai/gpt-audio", name: "Gpt Audio", kind: "audio" },
|
||||
{ id: "openai/gpt-audio-mini", name: "Gpt Audio Mini", kind: "audio" },
|
||||
{ id: "openai/gpt-oss-120b", name: "Gpt Oss 120B" },
|
||||
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
|
||||
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
|
||||
{ id: "qwen/qwen3-coder-next", name: "Qwen3 Coder Next" },
|
||||
{ id: "qwen/qwen3.5-122b-a10b", name: "Qwen3.5 122B A10B" },
|
||||
{ id: "qwen/qwen3.5-35b-a3b", name: "Qwen3.5 35B A3B" },
|
||||
{ id: "qwen/qwen3.5-397b-a17b", name: "Qwen3.5 397B A17B" },
|
||||
{ id: "qwen/qwen3.5-9b", name: "Qwen3.5 9B" },
|
||||
{ id: "qwen/qwen3.5-flash", name: "Qwen3.5 Flash" },
|
||||
{ id: "qwen/qwen3.5-plus-02-15", name: "Qwen3.5 Plus 02 15" },
|
||||
{ id: "qwen/qwen3.6-plus", name: "Qwen3.6 Plus" },
|
||||
{ id: "qwen/qwen3.7-max", name: "Qwen3.7 Max" },
|
||||
{ id: "qwen/qwen3.7-plus", name: "Qwen3.7 Plus" },
|
||||
{ id: "qwen/qwen3.8-max", name: "Qwen3.8 Max" },
|
||||
{ id: "qwen3.5-omni-plus", name: "Qwen3.5 Omni Plus" },
|
||||
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
|
||||
{ id: "sakana/fugu-ultra", name: "Fugu Ultra" },
|
||||
{ id: "seed-2-0-code-preview-260328", name: "Seed 2 0 Code Preview 260328" },
|
||||
{ id: "seed-2-0-lite-260428", name: "Seed 2 0 Lite 260428" },
|
||||
{ id: "seed-2-0-mini-260428", name: "Seed 2 0 Mini 260428" },
|
||||
{ id: "seed-2-0-pro-260328", name: "Seed 2 0 Pro 260328" },
|
||||
{ id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash" },
|
||||
{ id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash" },
|
||||
{ id: "tencent/hy3-preview", name: "Hy3 Preview" },
|
||||
{ id: "x-ai/grok-4.1-fast", name: "Grok 4.1 Fast" },
|
||||
{ id: "x-ai/grok-4.20-beta", name: "Grok 4.20 Beta" },
|
||||
{ id: "x-ai/grok-4.3", name: "Grok 4.3" },
|
||||
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
|
||||
{ id: "x-ai/grok-build-0.1", name: "Grok Build 0.1" },
|
||||
{ id: "xiaomi/mimo-v2-flash", name: "Mimo V2 Flash" },
|
||||
{ id: "xiaomi/mimo-v2-omni", name: "Mimo V2 Omni" },
|
||||
{ id: "xiaomi/mimo-v2-pro", name: "Mimo V2 Pro" },
|
||||
{ id: "xiaomi/mimo-v2.5", name: "Mimo V2.5" },
|
||||
{ id: "xiaomi/mimo-v2.5-pro", name: "Mimo V2.5 Pro" },
|
||||
{ id: "z-ai/glm-4.5-air", name: "Glm 4.5 Air" },
|
||||
{ id: "z-ai/glm-4.6", name: "Glm 4.6" },
|
||||
{ id: "z-ai/glm-4.6v", name: "Glm 4.6V" },
|
||||
{ id: "z-ai/glm-4.7", name: "Glm 4.7" },
|
||||
{ id: "z-ai/glm-5", name: "Glm 5" },
|
||||
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
|
||||
{ id: "z-ai/glm-5.1", name: "Glm 5.1" },
|
||||
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
|
||||
{ id: "z-ai/glm-5.3-free", name: "Glm 5.3 Free" },
|
||||
{ id: "z-ai/glm-5.2", name: "Glm 5.2" },
|
||||
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
|
||||
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
|
||||
],
|
||||
serviceKinds: ["llm", "embedding", "image"],
|
||||
embeddingConfig: {
|
||||
|
||||
@@ -25,17 +25,50 @@ export default {
|
||||
clientId: "b1a00492-073a-47ea-816f-4c329264a828",
|
||||
tokenUrl: "https://auth.x.ai/oauth2/token",
|
||||
refreshUrl: "https://auth.x.ai/oauth2/token",
|
||||
// OAuth-only SuperGrok quota surfaces:
|
||||
// - url: monthly API usage allotment (JSON)
|
||||
// - creditsUrl: weekly SuperGrok limit (grpc-web)
|
||||
// - settingsUrl: plan label (subscription_tier_display)
|
||||
usage: {
|
||||
url: "https://cli-chat-proxy.grok.com/v1/billing",
|
||||
creditsUrl: "https://grok.com/grok_api_v2.GrokBuildBilling/GetGrokCreditsConfig",
|
||||
settingsUrl: "https://cli-chat-proxy.grok.com/v1/settings",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "grok-4.6", name: "Grok 4.6" },
|
||||
{ id: "grok-4.5", name: "Grok 4.5" },
|
||||
{ id: "grok-4", name: "Grok 4" },
|
||||
{ id: "grok-4-fast-reasoning", name: "Grok 4 Fast Reasoning" },
|
||||
{ id: "grok-code-fast-1", name: "Grok Code Fast" },
|
||||
{ id: "grok-3", name: "Grok 3" },
|
||||
{ id: "grok-2-image-1212", name: "Grok 2 Image", params: ["n","response_format"], kind: "image" },
|
||||
{ id: "grok-imagine-video", name: "Grok Imagine Video", params: ["duration","aspect_ratio","resolution"], kind: "video" },
|
||||
{
|
||||
id: "grok-imagine-image-quality",
|
||||
name: "Grok Imagine Image Quality",
|
||||
capabilities: ["text2img", "edit"],
|
||||
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
|
||||
kind: "image",
|
||||
},
|
||||
{
|
||||
id: "grok-2-image-1212",
|
||||
name: "Grok 2 Image",
|
||||
capabilities: ["text2img", "edit"],
|
||||
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
|
||||
kind: "image",
|
||||
},
|
||||
{
|
||||
id: "grok-imagine-video",
|
||||
name: "Grok Imagine Video",
|
||||
params: ["duration", "aspect_ratio", "resolution"],
|
||||
kind: "video",
|
||||
},
|
||||
],
|
||||
serviceKinds: ["llm","imageToText","webSearch","image","video"],
|
||||
imageConfig: { baseUrl: "https://api.x.ai/v1/images/generations", bodyFields: ["model","prompt","n","response_format"] },
|
||||
serviceKinds: ["llm", "imageToText", "webSearch", "image", "video"],
|
||||
imageConfig: {
|
||||
baseUrl: "https://api.x.ai/v1/images/generations",
|
||||
editsUrl: "https://api.x.ai/v1/images/edits",
|
||||
bodyFields: ["model", "prompt", "n", "response_format", "aspect_ratio", "resolution", "image", "images"],
|
||||
},
|
||||
// Async video jobs (POST returns { request_id }, GET polls until done/failed).
|
||||
// Docs: https://docs.x.ai/developers/rest-api-reference/inference/videos
|
||||
videoConfig: { baseUrl: "https://api.x.ai/v1/videos" },
|
||||
@@ -44,4 +77,7 @@ export default {
|
||||
endpoint: "https://api.x.ai/v1/responses",
|
||||
pricingUrl: "https://x.ai/api#pricing",
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
},
|
||||
};
|
||||
|
||||
35
open-sse/providers/registry/xquik.js
Normal file
35
open-sse/providers/registry/xquik.js
Normal file
@@ -0,0 +1,35 @@
|
||||
export default {
|
||||
id: "xquik",
|
||||
alias: "xquik",
|
||||
display: {
|
||||
name: "Xquik",
|
||||
icon: "tag",
|
||||
color: "#5C3327",
|
||||
textIcon: "XQ",
|
||||
website: "https://docs.xquik.com/api-reference/x/search-tweets",
|
||||
notice: {
|
||||
apiKeyUrl: "https://xquik.com",
|
||||
text: "Searches public X posts. Billing uses 1 Xquik credit per returned post."
|
||||
}
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
serviceKinds: [
|
||||
"webSearch"
|
||||
],
|
||||
searchConfig: {
|
||||
baseUrl: "https://xquik.com/api/v1/x/tweets/search",
|
||||
validateUrl: "https://xquik.com/api/v1/credits",
|
||||
method: "GET",
|
||||
authType: "apikey",
|
||||
authHeader: "x-api-key",
|
||||
searchTypes: [
|
||||
"x"
|
||||
],
|
||||
defaultMaxResults: 5,
|
||||
maxMaxResults: 100,
|
||||
timeoutMs: 10000,
|
||||
cacheTTLMs: 60000,
|
||||
creditsPerResult: 1
|
||||
}
|
||||
};
|
||||
@@ -22,6 +22,7 @@ export function mapStainlessArch() {
|
||||
|
||||
// Anthropic API version (single source — reused across claude-format providers/executors)
|
||||
export const ANTHROPIC_API_VERSION = "2023-06-01";
|
||||
export const CLAUDE_CLI_VERSION = "2.1.258";
|
||||
|
||||
// Shared Claude-compatible API headers (reused across claude-format providers)
|
||||
export const CLAUDE_API_HEADERS = {
|
||||
@@ -34,7 +35,7 @@ export const CLAUDE_CLI_SPOOF_HEADERS = {
|
||||
"Anthropic-Version": ANTHROPIC_API_VERSION,
|
||||
"Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28",
|
||||
"Anthropic-Dangerous-Direct-Browser-Access": "true",
|
||||
"User-Agent": "claude-cli/2.1.92 (external, sdk-cli)",
|
||||
"User-Agent": `claude-cli/${CLAUDE_CLI_VERSION} (external, sdk-cli)`,
|
||||
"X-App": "cli",
|
||||
"X-Stainless-Helper-Method": "stream",
|
||||
"X-Stainless-Retry-Count": "0",
|
||||
@@ -74,10 +75,10 @@ export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
|
||||
export const OPENAI_COMPAT_BASE = "https://api.openai.com/v1";
|
||||
export const ANTHROPIC_COMPAT_BASE = "https://api.anthropic.com/v1";
|
||||
|
||||
// Official Antigravity IDE Desktop 2.1.1 fingerprint captured from macOS arm64.
|
||||
// Official Antigravity IDE Desktop 2.11.0 fingerprint captured from macOS arm64.
|
||||
// Keep this static even when 9router runs on Linux: the provider profile is
|
||||
// intentionally matching the IDE client, not the server host.
|
||||
export const ANTIGRAVITY_IDE_VERSION = "2.1.1";
|
||||
export const ANTIGRAVITY_IDE_VERSION = "2.11.0";
|
||||
export const ANTIGRAVITY_IDE_BASE_URL = "https://daily-cloudcode-pa.googleapis.com";
|
||||
export const ANTIGRAVITY_IDE_USER_AGENT = `antigravity/ide/${ANTIGRAVITY_IDE_VERSION} darwin/arm64`;
|
||||
|
||||
|
||||
@@ -35,10 +35,23 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh
|
||||
|
||||
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
|
||||
const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
// codebuddy-cn per-model effort sets — the server's product-config payload
|
||||
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
|
||||
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
|
||||
// → all 200), but values outside a model's supportedEfforts are silently
|
||||
// clamped, so the declared set stays authoritative for the picker. Models
|
||||
// that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x /
|
||||
// kimi-k3-1 / minimax-m3) fall through to the openai format default.
|
||||
{ provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] },
|
||||
{ provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] },
|
||||
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
|
||||
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
|
||||
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
|
||||
];
|
||||
|
||||
// Returns valid thinking levels for a model, or null when the model has no reasoning.
|
||||
|
||||
42
open-sse/providers/visionPatterns.js
Normal file
42
open-sse/providers/visionPatterns.js
Normal file
@@ -0,0 +1,42 @@
|
||||
// Name-based vision detection — last resort when neither the catalog file nor
|
||||
// the capability tables know a model. Vendors put the modality in the id
|
||||
// ("qwen3-vl-plus", "glm-4.6v", "deepseek-v4-flash-vision-exp"), so a custom or
|
||||
// freshly released model still gets image input instead of silently dropping it.
|
||||
//
|
||||
// Only ever turns vision ON. Never used to turn a declared capability off.
|
||||
|
||||
const SEP = "[-_/:.]";
|
||||
|
||||
// Image GENERATION, video generation, and non-chat models also carry these
|
||||
// words but take no image input — checked first so they can never match.
|
||||
const NOT_VISION = new RegExp(
|
||||
[
|
||||
`(^|${SEP})(image|img)(${SEP}|$)`,
|
||||
"stable-image", "gen[0-9]_image", "nanobanana", "imagine",
|
||||
"t2v", "i2v", "flux", "dall", "sdxl", "diffusion",
|
||||
"embed", "rerank", "guard", "moderation",
|
||||
"tts", "stt", "whisper", "voice", "speech", "audio",
|
||||
].join("|"),
|
||||
"i"
|
||||
);
|
||||
|
||||
// Explicit modality words, plus the "<digit>v" suffix vendors use for vision
|
||||
// variants (glm-4.6v, glm-5v-turbo). The digit-v branch requires a dotted
|
||||
// version so the never-shipped `gpt-4v` cannot match.
|
||||
const VISION_NAME = new RegExp(
|
||||
[
|
||||
`(^|${SEP})(vision|vl|vlm|multimodal|omni|visual)(${SEP}|$)`,
|
||||
`[0-9]\\.[0-9]+v(${SEP}|$)`,
|
||||
`(^|${SEP})glm-[0-9]+v(${SEP}|$)`,
|
||||
"(^|[-_/:.])(llava|pixtral|internvl|cogvlm|minicpm-v|moondream|idefics|fuyu)",
|
||||
].join("|"),
|
||||
"i"
|
||||
);
|
||||
|
||||
// Does this model id look like a vision model? Name signal only.
|
||||
export function looksLikeVisionModel(modelId) {
|
||||
if (!modelId) return false;
|
||||
const id = String(modelId).toLowerCase();
|
||||
if (NOT_VISION.test(id)) return false;
|
||||
return VISION_NAME.test(id);
|
||||
}
|
||||
@@ -7,6 +7,12 @@ import {
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 3000;
|
||||
|
||||
function normalizeTimeout(value) {
|
||||
return typeof value === "number" && Number.isFinite(value) && value > 0
|
||||
? value
|
||||
: DEFAULT_TIMEOUT_MS;
|
||||
}
|
||||
|
||||
function jsonBytes(value) {
|
||||
try {
|
||||
return new TextEncoder().encode(JSON.stringify(value) || "").length;
|
||||
@@ -240,6 +246,7 @@ async function callCompress(url, messages, model, timeoutMs, compressUserMessage
|
||||
// /v1/compress only understands OpenAI shape, so Claude bodies are translated
|
||||
// to OpenAI, compressed, then translated back using 9Router's own translators.
|
||||
export async function compressWithHeadroom(body, { enabled, url, model, format, compressUserMessages, timeoutMs = DEFAULT_TIMEOUT_MS, diagnostics = null } = {}) {
|
||||
timeoutMs = normalizeTimeout(timeoutMs);
|
||||
if (!enabled) {
|
||||
setDiagnostic(diagnostics, "disabled");
|
||||
return null;
|
||||
@@ -281,7 +288,10 @@ export async function compressWithHeadroom(body, { enabled, url, model, format,
|
||||
return null;
|
||||
}
|
||||
const oai = openaiResponsesToOpenAIRequest(model, body, false);
|
||||
if (!Array.isArray(oai?.messages)) return null;
|
||||
if (!Array.isArray(oai?.messages)) {
|
||||
setDiagnostic(diagnostics, "openai-responses request did not translate to messages[]");
|
||||
return null;
|
||||
}
|
||||
const data = await callCompress(url, oai.messages, model, timeoutMs, compressUserMessages, diagnostics || {});
|
||||
if (!data) return null;
|
||||
// input: undefined so the translator rebuilds input from the compressed
|
||||
|
||||
@@ -3,96 +3,335 @@
|
||||
// native-passthrough flows. Used by caveman.js and ponytail.js.
|
||||
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { OPENAI_BLOCK, CLAUDE_BLOCK, RESPONSES_ITEM } from "../translator/schema/blocks.js";
|
||||
import { ROLE } from "../translator/schema/roles.js";
|
||||
|
||||
const SEP = "\n\n";
|
||||
|
||||
export function injectSystemPrompt(body, format, prompt) {
|
||||
if (!body || !prompt) return;
|
||||
try {
|
||||
if (!body || !prompt) return;
|
||||
if (typeof body !== "object") return;
|
||||
|
||||
switch (format) {
|
||||
case FORMATS.CLAUDE:
|
||||
// Kiro wire shape is unique (conversationState/systemPrompt) — handle directly.
|
||||
if (isKiroBody(body) || format === FORMATS.KIRO) {
|
||||
injectKiroSystem(body, prompt);
|
||||
return;
|
||||
}
|
||||
|
||||
// Claude/Gemini own a dedicated system field, yet their bodies also carry
|
||||
// messages[]/contents[] — decide by format label before the shape sniff below.
|
||||
// Anthropic rejects a "system" role inside messages[] (no such input role).
|
||||
if (format === FORMATS.CLAUDE) {
|
||||
injectClaudeSystem(body, prompt);
|
||||
return;
|
||||
case FORMATS.GEMINI:
|
||||
case FORMATS.GEMINI_CLI:
|
||||
case FORMATS.VERTEX:
|
||||
case FORMATS.ANTIGRAVITY:
|
||||
}
|
||||
if (format === FORMATS.GEMINI || format === FORMATS.GEMINI_CLI
|
||||
|| format === FORMATS.VERTEX || format === FORMATS.ANTIGRAVITY) {
|
||||
// Antigravity wraps Gemini shape in body.request → injectGeminiSystem handles it
|
||||
injectGeminiSystem(body, prompt);
|
||||
return;
|
||||
default:
|
||||
// OpenAI and OpenAI-shaped formats (responses/codex/cursor/kiro/ollama)
|
||||
injectMessagesSystem(body, prompt);
|
||||
}
|
||||
}
|
||||
|
||||
// OpenAI-shaped: messages[] (chat) or input[] (responses) or instructions (responses string)
|
||||
function injectMessagesSystem(body, prompt) {
|
||||
// OpenAI Responses API: top-level string field
|
||||
if (typeof body.instructions === "string") {
|
||||
body.instructions = body.instructions
|
||||
? `${body.instructions}${SEP}${prompt}`
|
||||
: prompt;
|
||||
return;
|
||||
}
|
||||
|
||||
const arr = Array.isArray(body.messages) ? body.messages
|
||||
: Array.isArray(body.input) ? body.input
|
||||
: null;
|
||||
if (!arr) return;
|
||||
|
||||
const idx = arr.findIndex(m => m && (m.role === "system" || m.role === "developer"));
|
||||
if (idx >= 0) {
|
||||
appendToOpenAIMessage(arr[idx], prompt);
|
||||
} else {
|
||||
arr.unshift({ role: "system", content: prompt });
|
||||
}
|
||||
}
|
||||
|
||||
function appendToOpenAIMessage(msg, prompt) {
|
||||
if (typeof msg.content === "string") {
|
||||
msg.content = `${msg.content}${SEP}${prompt}`;
|
||||
} else if (Array.isArray(msg.content)) {
|
||||
// Responses-style array of parts {type:"input_text"|"text", text}
|
||||
msg.content.push({ type: "input_text", text: prompt });
|
||||
} else {
|
||||
msg.content = prompt;
|
||||
}
|
||||
}
|
||||
|
||||
// Claude shape: body.system as string | array of {type:"text", text}
|
||||
// Insert before the last cache_control block to keep injection inside the cached prefix.
|
||||
function injectClaudeSystem(body, prompt) {
|
||||
if (typeof body.system === "string" && body.system.length > 0) {
|
||||
body.system = `${body.system}${SEP}${prompt}`;
|
||||
return;
|
||||
}
|
||||
if (Array.isArray(body.system)) {
|
||||
const block = { type: "text", text: prompt };
|
||||
let lastCacheIdx = -1;
|
||||
for (let i = body.system.length - 1; i >= 0; i--) {
|
||||
if (body.system[i]?.cache_control) { lastCacheIdx = i; break; }
|
||||
}
|
||||
if (lastCacheIdx >= 0) {
|
||||
body.system.splice(lastCacheIdx, 0, block);
|
||||
|
||||
// Dispatch by actual wire shape for OpenAI-shaped formats.
|
||||
// instructions string takes precedence; messages[] means Chat; input[] means Responses.
|
||||
if (typeof body.instructions === "string") {
|
||||
injectInstructionsSystem(body, prompt);
|
||||
return;
|
||||
}
|
||||
if (Array.isArray(body.messages)) {
|
||||
injectChatSystem(body, prompt);
|
||||
return;
|
||||
}
|
||||
if (Array.isArray(body.input)) {
|
||||
// Responses input[]: empty array already normalized elsewhere; string stays untouched here
|
||||
injectResponsesInputSystem(body, prompt);
|
||||
return;
|
||||
}
|
||||
if (typeof body.input === "string") {
|
||||
// string input must stay untouched
|
||||
return;
|
||||
}
|
||||
|
||||
// OpenAI-shaped but no array (e.g. empty body) — no-op
|
||||
} catch (_) {
|
||||
// fail-open
|
||||
}
|
||||
}
|
||||
|
||||
function isKiroBody(body) {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (typeof body.systemPrompt !== "string") return false;
|
||||
const cs = body.conversationState;
|
||||
if (!cs || typeof cs !== "object") return false;
|
||||
return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object");
|
||||
}
|
||||
|
||||
// Exact idempotency: prompt present as its own SEP-delimited segment (or the
|
||||
// whole string), not as a substring of unrelated text.
|
||||
function hasPrompt(haystack, prompt) {
|
||||
if (!haystack || typeof haystack !== "string") return false;
|
||||
if (haystack === prompt) return true;
|
||||
return haystack.split(SEP).includes(prompt);
|
||||
}
|
||||
|
||||
function dedupStringAppend(curr, prompt) {
|
||||
if (!curr) return prompt;
|
||||
if (hasPrompt(curr, prompt)) return curr;
|
||||
return `${curr}${SEP}${prompt}`;
|
||||
}
|
||||
|
||||
// ---- OpenAI instructions string ----
|
||||
function injectInstructionsSystem(body, prompt) {
|
||||
try {
|
||||
const curr = body.instructions;
|
||||
if (typeof curr !== "string") return;
|
||||
if (hasPrompt(curr, prompt)) return;
|
||||
const next = curr ? `${curr}${SEP}${prompt}` : prompt;
|
||||
try { body.instructions = next; } catch (_) { /* frozen/proxy fail-open */ }
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
// ---- Chat messages[] ----
|
||||
function injectChatSystem(body, prompt) {
|
||||
try {
|
||||
const arr = body.messages;
|
||||
if (!Array.isArray(arr)) return;
|
||||
// Exact idempotency: scan existing system/developer content for full prompt
|
||||
if (containsPromptInMessages(arr, prompt)) return;
|
||||
let idx = -1;
|
||||
try { idx = arr.findIndex(m => m && (m.role === ROLE.SYSTEM || m.role === ROLE.DEVELOPER)); } catch (_) { return; }
|
||||
if (idx >= 0) {
|
||||
appendToChatMessage(arr[idx], prompt);
|
||||
} else {
|
||||
body.system.push(block);
|
||||
// create typed system message at index 0; fail-open on frozen/proxy
|
||||
try { arr.unshift({ role: ROLE.SYSTEM, content: prompt }); } catch (_) {}
|
||||
}
|
||||
return;
|
||||
}
|
||||
body.system = prompt;
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
// Gemini shape: body.system_instruction | body.systemInstruction | body.request.systemInstruction
|
||||
// Each shape: { parts: [{ text }] }
|
||||
function injectGeminiSystem(body, prompt) {
|
||||
const target = body.request && typeof body.request === "object" ? body.request : body;
|
||||
const useSnake = Object.prototype.hasOwnProperty.call(target, "system_instruction");
|
||||
const key = useSnake ? "system_instruction" : "systemInstruction";
|
||||
const sys = target[key];
|
||||
if (sys && Array.isArray(sys.parts)) {
|
||||
sys.parts.push({ text: prompt });
|
||||
return;
|
||||
}
|
||||
target[key] = { parts: [{ text: prompt }] };
|
||||
function containsPromptInMessages(arr, prompt) {
|
||||
try {
|
||||
for (const m of arr) {
|
||||
if (!m || (m.role !== ROLE.SYSTEM && m.role !== ROLE.DEVELOPER)) continue;
|
||||
const c = m.content;
|
||||
if (typeof c === "string" && hasPrompt(c, prompt)) return true;
|
||||
if (Array.isArray(c)) {
|
||||
for (const part of c) {
|
||||
if (part && typeof part.text === "string" && hasPrompt(part.text, prompt)) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (_) {}
|
||||
return false;
|
||||
}
|
||||
|
||||
function appendToChatMessage(msg, prompt) {
|
||||
try {
|
||||
if (!msg || typeof msg !== "object") return;
|
||||
const c = msg.content;
|
||||
if (typeof c === "string") {
|
||||
const next = dedupStringAppend(c, prompt);
|
||||
if (next === c) return;
|
||||
// avoid partial mutation: try assignment, bail if setter throws
|
||||
try { msg.content = next; } catch (_) {}
|
||||
return;
|
||||
}
|
||||
if (Array.isArray(c)) {
|
||||
// already deduped at message level; but guard block-level too
|
||||
try {
|
||||
if (c.some(b => b && b.text === prompt)) return;
|
||||
} catch (_) {}
|
||||
try { c.push({ type: OPENAI_BLOCK.TEXT, text: prompt }); } catch (_) {}
|
||||
return;
|
||||
}
|
||||
try { msg.content = prompt; } catch (_) {}
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
// ---- Responses input[] ----
|
||||
function injectResponsesInputSystem(body, prompt) {
|
||||
try {
|
||||
const arr = body.input;
|
||||
if (!Array.isArray(arr)) return;
|
||||
// instructions already handled above
|
||||
if (containsPromptInResponsesInput(arr, prompt)) return;
|
||||
// find system/developer message items only (type === message)
|
||||
let idx = -1;
|
||||
try {
|
||||
idx = arr.findIndex(m => m && m.type === RESPONSES_ITEM.MESSAGE && (m.role === ROLE.SYSTEM || m.role === ROLE.DEVELOPER));
|
||||
} catch (_) { return; }
|
||||
if (idx >= 0) {
|
||||
appendToResponsesMessage(arr[idx], prompt);
|
||||
} else {
|
||||
const msg = { type: RESPONSES_ITEM.MESSAGE, role: ROLE.SYSTEM, content: [{ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }] };
|
||||
try { arr.unshift(msg); } catch (_) {}
|
||||
}
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
function containsPromptInResponsesInput(arr, prompt) {
|
||||
try {
|
||||
for (const item of arr) {
|
||||
if (!item || item.type !== RESPONSES_ITEM.MESSAGE) continue;
|
||||
if (item.role !== ROLE.SYSTEM && item.role !== ROLE.DEVELOPER) continue;
|
||||
const c = item.content;
|
||||
if (typeof c === "string" && hasPrompt(c, prompt)) return true;
|
||||
if (Array.isArray(c)) {
|
||||
for (const part of c) {
|
||||
if (part && typeof part.text === "string" && hasPrompt(part.text, prompt)) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (_) {}
|
||||
return false;
|
||||
}
|
||||
|
||||
function appendToResponsesMessage(msg, prompt) {
|
||||
try {
|
||||
if (!msg || typeof msg !== "object") return;
|
||||
const c = msg.content;
|
||||
if (typeof c === "string") {
|
||||
const next = dedupStringAppend(c, prompt);
|
||||
if (next === c) return;
|
||||
try { msg.content = next; } catch (_) {}
|
||||
return;
|
||||
}
|
||||
if (Array.isArray(c)) {
|
||||
try { if (c.some(b => b && b.text === prompt)) return; } catch (_) {}
|
||||
try { c.push({ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }); } catch (_) {}
|
||||
return;
|
||||
}
|
||||
try { msg.content = [{ type: RESPONSES_ITEM.INPUT_TEXT, text: prompt }]; } catch (_) {}
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
// ---- Claude ----
|
||||
function injectClaudeSystem(body, prompt) {
|
||||
try {
|
||||
const sys = body.system;
|
||||
if (typeof sys === "string") {
|
||||
if (hasPrompt(sys, prompt)) return;
|
||||
const next = sys.length > 0 ? `${sys}${SEP}${prompt}` : prompt;
|
||||
try { body.system = next; } catch (_) {}
|
||||
return;
|
||||
}
|
||||
if (Array.isArray(sys)) {
|
||||
try { if (sys.some(b => b && b.text === prompt)) return; } catch (_) {}
|
||||
const block = { type: CLAUDE_BLOCK.TEXT, text: prompt };
|
||||
let lastCacheIdx = -1;
|
||||
try {
|
||||
for (let i = sys.length - 1; i >= 0; i--) {
|
||||
if (sys[i]?.cache_control) { lastCacheIdx = i; break; }
|
||||
}
|
||||
} catch (_) {}
|
||||
try {
|
||||
if (lastCacheIdx >= 0) sys.splice(lastCacheIdx, 0, block);
|
||||
else sys.push(block);
|
||||
} catch (_) {}
|
||||
return;
|
||||
}
|
||||
// absent/null
|
||||
try { body.system = prompt; } catch (_) {}
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
// ---- Gemini ----
|
||||
function injectGeminiSystem(body, prompt) {
|
||||
try {
|
||||
let target = body;
|
||||
try {
|
||||
if (body.request && typeof body.request === "object") target = body.request;
|
||||
} catch (_) {}
|
||||
let useSnake = false;
|
||||
try { useSnake = Object.prototype.hasOwnProperty.call(target, "system_instruction"); } catch (_) {}
|
||||
const key = useSnake ? "system_instruction" : "systemInstruction";
|
||||
let sys;
|
||||
try { sys = target[key]; } catch (_) { sys = undefined; }
|
||||
if (sys && Array.isArray(sys.parts)) {
|
||||
try { if (sys.parts.some(p => p && p.text === prompt)) return; } catch (_) {}
|
||||
try { sys.parts.push({ text: prompt }); } catch (_) {}
|
||||
return;
|
||||
}
|
||||
try { target[key] = { parts: [{ text: prompt }] }; } catch (_) {}
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
// ---- Kiro ----
|
||||
// Updates top-level systemPrompt and only the mirrored leading prefix of the
|
||||
// first user history turn, else current user. next = old + SEP + prompt.
|
||||
// Replace old leading prefix only; preserve time context and user tail.
|
||||
function injectKiroSystem(body, prompt) {
|
||||
try {
|
||||
let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : "";
|
||||
// Repair path: a previous partial write left systemPrompt updated but user
|
||||
// content still mirroring the pre-write prefix. Re-derive the effective old
|
||||
// prefix from content so this pass converges instead of early-returning.
|
||||
const cs0 = body.conversationState;
|
||||
let firstUser0 = cs0 && Array.isArray(cs0.history)
|
||||
? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null)
|
||||
: null;
|
||||
if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage;
|
||||
|
||||
if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) {
|
||||
const c0 = firstUser0.content;
|
||||
if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) {
|
||||
// systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored
|
||||
oldPrompt = "";
|
||||
}
|
||||
}
|
||||
if (oldPrompt && hasPrompt(oldPrompt, prompt)) return;
|
||||
const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt;
|
||||
|
||||
// Atomicity: write user content first, then systemPrompt only if content
|
||||
// write succeeded (or was a no-op). If systemPrompt write then fails, the
|
||||
// repair heuristic above re-derives from content on retry — no permanent
|
||||
// half-applied state.
|
||||
const cs = body.conversationState;
|
||||
let targetMsg = null;
|
||||
try {
|
||||
const hist = Array.isArray(cs?.history) ? cs.history : null;
|
||||
if (hist) {
|
||||
for (const item of hist) {
|
||||
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
|
||||
}
|
||||
}
|
||||
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
|
||||
targetMsg = cs.currentMessage.userInputMessage;
|
||||
}
|
||||
} catch (_) { targetMsg = null; }
|
||||
|
||||
let sysWritten = false;
|
||||
try { body.systemPrompt = next; sysWritten = true; } catch (_) {}
|
||||
|
||||
const applyContent = () => {
|
||||
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
|
||||
if (oldPrompt === "") {
|
||||
// Empty old prompt: prepend unless already at head (exact, not substring)
|
||||
if (content.startsWith(prompt) || content.startsWith(next)) return;
|
||||
const newContent = content ? `${next}${SEP}${content}` : next;
|
||||
try { targetMsg.content = newContent; } catch (_) {}
|
||||
return;
|
||||
}
|
||||
if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone
|
||||
if (content.startsWith(next)) return; // already applied → idempotent
|
||||
const tail = content.slice(oldPrompt.length);
|
||||
try { targetMsg.content = `${next}${tail}`; } catch (_) {}
|
||||
};
|
||||
|
||||
try {
|
||||
if (targetMsg) applyContent();
|
||||
} catch (_) {}
|
||||
if (sysWritten && targetMsg) {
|
||||
// verify convergence: content should now start with next (or be un-mirrored)
|
||||
let ok = false;
|
||||
try {
|
||||
const c = targetMsg.content;
|
||||
ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt);
|
||||
} catch (_) {}
|
||||
if (!ok) {
|
||||
try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback
|
||||
}
|
||||
}
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
@@ -26,8 +26,14 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) {
|
||||
: "";
|
||||
|
||||
for (const rule of ERROR_RULES) {
|
||||
// Request-scoped rule: the request body itself is at fault — no cooldown,
|
||||
// no account lock. Caller must stop rotating and surface the error.
|
||||
if (rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
|
||||
return { shouldFallback: false, requestScoped: true, cooldownMs: 0 };
|
||||
}
|
||||
|
||||
// Text-based rule: match substring in error message
|
||||
if (rule.text && lowerError && lowerError.includes(rule.text)) {
|
||||
if (rule.text && !rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
|
||||
if (rule.backoff) {
|
||||
const newLevel = Math.min(backoffLevel + 1, BACKOFF_CONFIG.maxLevel);
|
||||
return { shouldFallback: true, cooldownMs: getQuotaCooldown(newLevel), newBackoffLevel: newLevel };
|
||||
|
||||
@@ -1,71 +0,0 @@
|
||||
/**
|
||||
* Shared combo (model combo) handling with fallback support
|
||||
*/
|
||||
|
||||
/**
|
||||
* Get combo models from combos data
|
||||
* @param {string} modelStr - Model string to check
|
||||
* @param {Array|Object} combosData - Array of combos or object with combos
|
||||
* @returns {string[]|null} Array of models or null if not a combo
|
||||
*/
|
||||
export function getComboModelsFromData(modelStr, combosData) {
|
||||
// Don't check if it's in provider/model format
|
||||
if (modelStr.includes("/")) return null;
|
||||
|
||||
// Handle both array and object formats
|
||||
const combos = Array.isArray(combosData) ? combosData : (combosData?.combos || []);
|
||||
|
||||
const combo = combos.find(c => c.name === modelStr);
|
||||
if (combo && combo.models && combo.models.length > 0) {
|
||||
return combo.models;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle combo chat with fallback
|
||||
* @param {Object} options
|
||||
* @param {Object} options.body - Request body
|
||||
* @param {string[]} options.models - Array of model strings to try
|
||||
* @param {Function} options.handleSingleModel - Function to handle single model: (body, modelStr) => Promise<Response>
|
||||
* @param {Object} options.log - Logger object
|
||||
* @returns {Promise<Response>}
|
||||
*/
|
||||
export async function handleComboChat({ body, models, handleSingleModel, log }) {
|
||||
let lastError = null;
|
||||
|
||||
for (let i = 0; i < models.length; i++) {
|
||||
const modelStr = models[i];
|
||||
log.info("COMBO", `Trying model ${i + 1}/${models.length}: ${modelStr}`);
|
||||
|
||||
let result;
|
||||
try {
|
||||
result = await handleSingleModel(body, modelStr);
|
||||
} catch (e) {
|
||||
lastError = `${modelStr}: ${e.message}`;
|
||||
log.warn("COMBO", `Model threw exception, trying next`, { model: modelStr, error: e.message });
|
||||
continue;
|
||||
}
|
||||
|
||||
// Success or client error - return response
|
||||
if (result.ok || result.status < 500) {
|
||||
return result;
|
||||
}
|
||||
|
||||
// 5xx error - try next model
|
||||
lastError = `${modelStr}: ${result.statusText || result.status}`;
|
||||
log.warn("COMBO", `Model failed, trying next`, { model: modelStr, status: result.status });
|
||||
}
|
||||
|
||||
log.warn("COMBO", "All models failed");
|
||||
|
||||
// Return 503 with last error
|
||||
return new Response(
|
||||
JSON.stringify({ error: lastError || "All combo models unavailable" }),
|
||||
{
|
||||
status: 503,
|
||||
headers: { "Content-Type": "application/json" }
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
@@ -203,7 +203,8 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
|
||||
|
||||
const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA };
|
||||
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
|
||||
const MAX_ATTEMPTS = 5;
|
||||
const MAX_ATTEMPTS = Number(process.env.ONBOARD_MAX_ATTEMPTS) || 2;
|
||||
const BASE_RETRY_DELAY_MS = Number(process.env.ONBOARD_RETRY_DELAY_MS) || 12_000;
|
||||
|
||||
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
|
||||
// Bail out immediately if the connection was removed
|
||||
@@ -241,9 +242,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
|
||||
throw new Error("onboardUser done but no project_id in response");
|
||||
}
|
||||
|
||||
// Server not done yet – wait and retry
|
||||
// Server not done yet – wait and retry with jitter
|
||||
const jitter = Math.floor(Math.random() * 5000);
|
||||
console.log(`[ProjectId] Onboard attempt ${attempt}/${MAX_ATTEMPTS}: not done yet, waiting...`);
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter));
|
||||
|
||||
} catch (error) {
|
||||
clearTimeout(timeoutId);
|
||||
@@ -256,9 +258,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi
|
||||
console.warn(`[ProjectId] onboardUser failed after ${MAX_ATTEMPTS} attempts: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
// Continue to next attempt instead of throwing (which would skip remaining retries)
|
||||
// Wait with jitter before retrying
|
||||
const jitter = Math.floor(Math.random() * 5000);
|
||||
console.warn(`[ProjectId] onboardUser attempt ${attempt} failed: ${error.message}, retrying...`);
|
||||
await new Promise(resolve => setTimeout(resolve, 2000));
|
||||
await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter));
|
||||
} finally {
|
||||
clearTimeout(timeoutId);
|
||||
externalSignal?.removeEventListener("abort", forwardAbort);
|
||||
|
||||
56
open-sse/services/providerTimeout.js
Normal file
56
open-sse/services/providerTimeout.js
Normal file
@@ -0,0 +1,56 @@
|
||||
/**
|
||||
* Per-provider connect timeout overrides from user settings.
|
||||
* Settings are read from the DB lazily and cached with a short TTL
|
||||
* so UI changes take effect without requiring a restart.
|
||||
*/
|
||||
|
||||
let cached = {};
|
||||
let cacheTs = 0;
|
||||
const CACHE_TTL_MS = 10_000; // 10s — responsive enough for dashboard changes
|
||||
|
||||
async function refreshCache() {
|
||||
const now = Date.now();
|
||||
if (now - cacheTs < CACHE_TTL_MS && Object.keys(cached).length > 0) return cached;
|
||||
|
||||
try {
|
||||
const { getSettings } = await import("@/lib/localDb");
|
||||
// Return full settings so we can read providerTimeouts + globalTimeoutMs
|
||||
cached = await getSettings();
|
||||
cacheTs = now;
|
||||
} catch {
|
||||
// If DB is unavailable, keep stale cache — don't throw on hot path
|
||||
}
|
||||
return cached;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the effective connect timeout for a provider.
|
||||
* Priority: per-provider override > global default timeout (settings) > registry config > env default.
|
||||
* @param {string} providerId
|
||||
* @param {number} configTimeoutMs - timeoutMs from the static provider registry config
|
||||
* @param {number} envDefaultMs - global default from env (FETCH_CONNECT_TIMEOUT_MS)
|
||||
* @returns {number} timeout in milliseconds
|
||||
*/
|
||||
export async function resolveProviderTimeoutMs(providerId, configTimeoutMs, envDefaultMs) {
|
||||
const overrides = await refreshCache();
|
||||
|
||||
// 1. Per-provider override (set in provider detail page)
|
||||
const providerOverride = overrides.providerTimeouts?.[providerId];
|
||||
if (providerOverride?.timeoutMs && Number.isFinite(providerOverride.timeoutMs) && providerOverride.timeoutMs > 0) {
|
||||
return providerOverride.timeoutMs;
|
||||
}
|
||||
|
||||
// 2. Global default timeout (set in Profile / Settings page)
|
||||
const globalDefault = overrides.defaultTimeoutMs;
|
||||
if (globalDefault && Number.isFinite(globalDefault) && globalDefault > 0) {
|
||||
return globalDefault;
|
||||
}
|
||||
|
||||
// 3. Registry per-provider config
|
||||
if (configTimeoutMs && Number.isFinite(configTimeoutMs) && configTimeoutMs > 0) {
|
||||
return configTimeoutMs;
|
||||
}
|
||||
|
||||
// 4. Env default
|
||||
return envDefaultMs;
|
||||
}
|
||||
170
open-sse/services/thoughtSignatureStore.js
Normal file
170
open-sse/services/thoughtSignatureStore.js
Normal file
@@ -0,0 +1,170 @@
|
||||
import { makeKv } from "../../src/lib/db/helpers/kvStore.js";
|
||||
|
||||
const MAX_SIGNATURES = 2000;
|
||||
const MAX_PERSISTED_SIGNATURES = 10_000;
|
||||
const MEMORY_TTL_MS = 1000 * 60 * 60; // 1 hour
|
||||
const PERSISTED_TTL_MS = 1000 * 60 * 60 * 24 * 7; // 7 days
|
||||
const SCOPE = "gemini_thought_signatures";
|
||||
|
||||
const signatureKv = makeKv(SCOPE);
|
||||
const memorySignatures = new Map();
|
||||
let pruneCounter = 0;
|
||||
|
||||
function pruneMemoryExpired() {
|
||||
const now = Date.now();
|
||||
for (const [key, value] of memorySignatures.entries()) {
|
||||
if (value.expiresAt <= now) {
|
||||
memorySignatures.delete(key);
|
||||
}
|
||||
}
|
||||
|
||||
while (memorySignatures.size > MAX_SIGNATURES) {
|
||||
const oldestKey = memorySignatures.keys().next().value;
|
||||
if (!oldestKey) break;
|
||||
memorySignatures.delete(oldestKey);
|
||||
}
|
||||
}
|
||||
|
||||
async function maybePrunePersisted() {
|
||||
pruneCounter++;
|
||||
if (pruneCounter % 100 !== 0) return;
|
||||
|
||||
try {
|
||||
const all = await signatureKv.getAll();
|
||||
const keys = Object.keys(all);
|
||||
const now = Date.now();
|
||||
const expiredKeys = [];
|
||||
const valid = [];
|
||||
|
||||
for (const k of keys) {
|
||||
const entry = all[k];
|
||||
if (!entry || typeof entry.signature !== "string" || (entry.expiresAt && entry.expiresAt <= now)) {
|
||||
expiredKeys.push(k);
|
||||
} else {
|
||||
valid.push({ key: k, createdAt: entry.createdAt || 0 });
|
||||
}
|
||||
}
|
||||
|
||||
for (const k of expiredKeys) {
|
||||
await signatureKv.remove(k).catch(() => {});
|
||||
}
|
||||
|
||||
if (valid.length > MAX_PERSISTED_SIGNATURES) {
|
||||
valid.sort((a, b) => b.createdAt - a.createdAt);
|
||||
const toRemove = valid.slice(MAX_PERSISTED_SIGNATURES);
|
||||
for (const item of toRemove) {
|
||||
await signatureKv.remove(item.key).catch(() => {});
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Fail-open
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async)
|
||||
*/
|
||||
export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null) {
|
||||
if (typeof toolCallId !== "string" || !toolCallId) return;
|
||||
if (typeof signature !== "string" || !signature) return;
|
||||
|
||||
const now = Date.now();
|
||||
pruneMemoryExpired();
|
||||
|
||||
const keys = [];
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
keys.push(`${sessionId}:${toolCallId}`);
|
||||
}
|
||||
keys.push(toolCallId);
|
||||
|
||||
for (const k of keys) {
|
||||
memorySignatures.set(k, {
|
||||
signature,
|
||||
expiresAt: now + MEMORY_TTL_MS,
|
||||
});
|
||||
|
||||
// Async persist to SQLite kv table without blocking
|
||||
signatureKv.set(k, {
|
||||
signature,
|
||||
createdAt: now,
|
||||
expiresAt: now + PERSISTED_TTL_MS,
|
||||
}).catch(() => {});
|
||||
}
|
||||
|
||||
maybePrunePersisted().catch(() => {});
|
||||
}
|
||||
|
||||
/**
|
||||
* Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback)
|
||||
*/
|
||||
export async function getGeminiThoughtSignature(toolCallId, sessionId = null) {
|
||||
if (typeof toolCallId !== "string" || !toolCallId) return null;
|
||||
|
||||
pruneMemoryExpired();
|
||||
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
const sessionKey = `${sessionId}:${toolCallId}`;
|
||||
const sessionEntry = memorySignatures.get(sessionKey);
|
||||
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
|
||||
return sessionEntry.signature;
|
||||
}
|
||||
}
|
||||
|
||||
const entry = memorySignatures.get(toolCallId);
|
||||
if (entry && entry.expiresAt > Date.now()) {
|
||||
return entry.signature;
|
||||
}
|
||||
|
||||
try {
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
const sessionKey = `${sessionId}:${toolCallId}`;
|
||||
const sessionRow = await signatureKv.get(sessionKey);
|
||||
if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now())) {
|
||||
memorySignatures.set(sessionKey, {
|
||||
signature: sessionRow.signature,
|
||||
expiresAt: Date.now() + MEMORY_TTL_MS,
|
||||
});
|
||||
return sessionRow.signature;
|
||||
}
|
||||
}
|
||||
|
||||
const row = await signatureKv.get(toolCallId);
|
||||
if (row && typeof row.signature === "string") {
|
||||
if (row.expiresAt && row.expiresAt <= Date.now()) {
|
||||
signatureKv.remove(toolCallId).catch(() => {});
|
||||
return null;
|
||||
}
|
||||
memorySignatures.set(toolCallId, {
|
||||
signature: row.signature,
|
||||
expiresAt: Date.now() + MEMORY_TTL_MS,
|
||||
});
|
||||
return row.signature;
|
||||
}
|
||||
} catch {
|
||||
// Fail-open
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Synchronous get from RAM cache only (for sync translators)
|
||||
*/
|
||||
export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null) {
|
||||
if (typeof toolCallId !== "string" || !toolCallId) return null;
|
||||
pruneMemoryExpired();
|
||||
|
||||
if (sessionId && typeof sessionId === "string") {
|
||||
const sessionKey = `${sessionId}:${toolCallId}`;
|
||||
const sessionEntry = memorySignatures.get(sessionKey);
|
||||
if (sessionEntry && sessionEntry.expiresAt > Date.now()) {
|
||||
return sessionEntry.signature;
|
||||
}
|
||||
}
|
||||
|
||||
const entry = memorySignatures.get(toolCallId);
|
||||
if (entry && entry.expiresAt > Date.now()) {
|
||||
return entry.signature;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import {
|
||||
refreshXaiToken,
|
||||
refreshAccessToken,
|
||||
refreshKimiToken,
|
||||
refreshClineToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshCodexToken,
|
||||
@@ -23,6 +24,7 @@ import {
|
||||
export {
|
||||
refreshAccessToken,
|
||||
refreshKimiToken,
|
||||
refreshClineToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshCodexToken,
|
||||
@@ -145,6 +147,7 @@ const REFRESH_HANDLERS = {
|
||||
"codebuddy-cn": (c, log) => refreshCodebuddyToken(c.refreshToken, log),
|
||||
"codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log),
|
||||
trae: (c, log) => refreshTraeToken(c.refreshToken, c, log),
|
||||
cline: (c, log) => refreshClineToken(c.refreshToken, log),
|
||||
zed: () => refreshZedToken(),
|
||||
windsurf: (c, log) => refreshWindsurfToken(c, log),
|
||||
// Kimi Code OAuth (merged into id `kimi`); legacy id still routes here
|
||||
|
||||
@@ -147,6 +147,53 @@ export async function refreshKimiToken(refreshToken, credentials, log) {
|
||||
return refreshAccessToken("kimi", refreshToken, credentials, log);
|
||||
}
|
||||
|
||||
export async function refreshClineToken(refreshToken, log) {
|
||||
if (!refreshToken) return null;
|
||||
|
||||
return dedupRefresh("cline", refreshToken, async () => {
|
||||
try {
|
||||
const response = await fetch(PROVIDERS.cline?.refreshUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
refreshToken,
|
||||
grantType: "refresh_token",
|
||||
clientType: "extension",
|
||||
}),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
log?.error?.("TOKEN_REFRESH", "Failed to refresh Cline token", {
|
||||
status: response.status,
|
||||
error: errorText,
|
||||
});
|
||||
return null;
|
||||
}
|
||||
|
||||
const body = await response.json();
|
||||
const tokens = body?.data || body;
|
||||
if (!tokens?.accessToken) return null;
|
||||
|
||||
const expiresIn = tokens.expiresAt
|
||||
? Math.max(1, Math.floor((new Date(tokens.expiresAt).getTime() - Date.now()) / 1000))
|
||||
: (tokens.expiresIn || tokens.expires_in || 3600);
|
||||
|
||||
return {
|
||||
accessToken: tokens.accessToken,
|
||||
refreshToken: tokens.refreshToken || refreshToken,
|
||||
expiresIn,
|
||||
};
|
||||
} catch (error) {
|
||||
log?.error?.("TOKEN_REFRESH", `Error refreshing Cline token: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
}, log);
|
||||
}
|
||||
|
||||
// Claude OAuth: JSON body, client_id only. Delegate to refreshAccessToken("claude", ...).
|
||||
export async function refreshClaudeOAuthToken(refreshToken, log) {
|
||||
return refreshAccessToken("claude", refreshToken, {}, log);
|
||||
|
||||
@@ -11,14 +11,18 @@ export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
|
||||
import { getKiroUsage } from "./usage/kiro.js";
|
||||
import { getMiniMaxUsage } from "./usage/minimax.js";
|
||||
import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js";
|
||||
import { getXaiUsage } from "./usage/xai.js";
|
||||
import { getGrokCliUsage } from "./usage/grok-cli.js";
|
||||
import { getKimiUsage } from "./usage/kimi.js";
|
||||
import { getDeepseekUsage } from "./usage/deepseek.js";
|
||||
import { getOpenCodeGoUsage } from "./usage/opencode-go.js";
|
||||
import { getGroqUsage } from "./usage/groq.js";
|
||||
import { getZedUsage } from "./usage/zed.js";
|
||||
import { resolveQoderCredentials } from "./qoderModels.js";
|
||||
import { getGlmUsage } from "./usage/glm.js";
|
||||
import {
|
||||
getIflowUsage,
|
||||
getOllamaUsage,
|
||||
getGlmUsage,
|
||||
getVercelAiGatewayUsage,
|
||||
getQoderUsage,
|
||||
} from "./usage/misc.js";
|
||||
@@ -50,10 +54,14 @@ const USAGE_HANDLERS = {
|
||||
"minimax-cn": (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
"vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions),
|
||||
"codebuddy-cn": (c) => getCodeBuddyCnUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
xai: (c) => getXaiUsage(c.accessToken, c.proxyOptions),
|
||||
"codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData),
|
||||
"opencode-go": (c) => getOpenCodeGoUsage(c.apiKey, c.proxyOptions),
|
||||
deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions),
|
||||
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
|
||||
zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
};
|
||||
|
||||
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {
|
||||
|
||||
@@ -102,14 +102,34 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
|
||||
quotas["weekly (7d)"] = createQuotaObject(data.seven_day);
|
||||
}
|
||||
|
||||
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus)
|
||||
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus, seven_day_fable)
|
||||
const MODEL_DISPLAY_NAMES = {
|
||||
fable_5_1: "fable",
|
||||
fable_5: "fable",
|
||||
};
|
||||
|
||||
for (const [key, value] of Object.entries(data)) {
|
||||
if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) {
|
||||
const modelName = key.replace("seven_day_", "");
|
||||
const rawName = key.replace("seven_day_", "");
|
||||
const modelName = MODEL_DISPLAY_NAMES[rawName] || rawName;
|
||||
quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value);
|
||||
} else if ((key === "fable" || key === "fable_5" || key === "fable_5_1") && hasUtilization(value)) {
|
||||
quotas["weekly fable (7d)"] = createQuotaObject(value);
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: surface Fable quota row if weekly window exists but Fable was not returned yet
|
||||
if (!quotas["weekly fable (7d)"] && hasUtilization(data.seven_day)) {
|
||||
quotas["weekly fable (7d)"] = {
|
||||
used: 0,
|
||||
total: 100,
|
||||
remaining: 100,
|
||||
remainingPercentage: 100,
|
||||
resetAt: parseResetTime(data.seven_day.resets_at),
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
plan: "Claude Code",
|
||||
extraUsage: data.extra_usage ?? null,
|
||||
|
||||
@@ -21,6 +21,13 @@ function toIsoDate(value) {
|
||||
return Number.isFinite(time) ? date.toISOString() : null;
|
||||
}
|
||||
|
||||
function errorMessage(value, fallback) {
|
||||
if (!value) return fallback;
|
||||
if (typeof value === "string") return value;
|
||||
if (typeof value.message === "string") return value.message;
|
||||
return JSON.stringify(value);
|
||||
}
|
||||
|
||||
function getCodexAccountId(providerSpecificData) {
|
||||
return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null;
|
||||
}
|
||||
@@ -80,6 +87,23 @@ function getCodexReviewRateLimit(data) {
|
||||
}) || null;
|
||||
}
|
||||
|
||||
function getCodexSparkRateLimit(data) {
|
||||
if (data.spark_rate_limit || data.gpt_5_3_codex_spark_rate_limit) {
|
||||
return data.spark_rate_limit || data.gpt_5_3_codex_spark_rate_limit;
|
||||
}
|
||||
|
||||
const byLimitId = data.rate_limits_by_limit_id;
|
||||
if (byLimitId && typeof byLimitId === "object" && !Array.isArray(byLimitId)) {
|
||||
return byLimitId["gpt-5.3-codex-spark"] || byLimitId.gpt_5_3_codex_spark || byLimitId.spark || null;
|
||||
}
|
||||
|
||||
const additional = Array.isArray(data.additional_rate_limits) ? data.additional_rate_limits : [];
|
||||
return additional.find((entry) => {
|
||||
const id = String(entry?.limit_name || entry?.metered_feature || entry?.id || "").toLowerCase();
|
||||
return id.includes("spark") || id.includes("5.3-codex-spark");
|
||||
}) || null;
|
||||
}
|
||||
|
||||
export async function getCodexUsage(accessToken, proxyOptions = null) {
|
||||
try {
|
||||
const response = await proxyAwareFetch(CODEX_CONFIG.usageUrl, {
|
||||
@@ -97,16 +121,19 @@ export async function getCodexUsage(accessToken, proxyOptions = null) {
|
||||
const data = await response.json();
|
||||
const normalRateLimit = data.rate_limit || data.rate_limits || data.rate_limits_by_limit_id?.codex || {};
|
||||
const reviewRateLimit = getCodexReviewRateLimit(data);
|
||||
const sparkRateLimit = getCodexSparkRateLimit(data);
|
||||
const availableResetCredits = Math.max(0, toFiniteNumber(data.rate_limit_reset_credits?.available_count, 0));
|
||||
const quotas = {};
|
||||
|
||||
appendCodexQuotaWindows(quotas, "", normalRateLimit);
|
||||
appendCodexQuotaWindows(quotas, "review", reviewRateLimit);
|
||||
appendCodexQuotaWindows(quotas, "spark", sparkRateLimit);
|
||||
|
||||
return {
|
||||
plan: data.plan_type || data.summary?.plan || "unknown",
|
||||
limitReached: getCodexRateLimitBody(normalRateLimit)?.limit_reached || false,
|
||||
reviewLimitReached: getCodexRateLimitBody(reviewRateLimit)?.limit_reached || false,
|
||||
sparkLimitReached: getCodexRateLimitBody(sparkRateLimit)?.limit_reached || false,
|
||||
resetCredits: { availableCount: availableResetCredits },
|
||||
quotas,
|
||||
};
|
||||
@@ -142,7 +169,7 @@ export async function getCodexRateLimitResetCredits(accessToken, proxyOptions =
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`;
|
||||
const message = errorMessage(data?.message || data?.error || data?.detail, `Codex reset credits API unavailable (${response.status}).`);
|
||||
throw new Error(message);
|
||||
}
|
||||
|
||||
|
||||
207
open-sse/services/usage/commandcode.js
Normal file
207
open-sse/services/usage/commandcode.js
Normal file
@@ -0,0 +1,207 @@
|
||||
/**
|
||||
* CommandCode usage handler
|
||||
*
|
||||
* Mirrors the official command-code CLI /usage command: it calls the alpha API
|
||||
* to surface the 5-hour + weekly usage windows, the subscription plan, and the
|
||||
* credits consumed in the current billing period.
|
||||
*
|
||||
* GET /alpha/whoami → org.id (org-scoped billing; null for personal)
|
||||
* GET /alpha/billing/credits → { credits: { monthlyCredits, purchasedCredits,
|
||||
* freeCredits }, windowLimits: { fiveHour, weekly } }
|
||||
* GET /alpha/billing/subscriptions → { data: { planId, currentPeriodStart, ... } }
|
||||
* GET /alpha/usage/summary?since= → period token/cost totals
|
||||
*
|
||||
* The CLI fetches whoami first (for orgId), then credits + subscription in
|
||||
* parallel, then the summary with since = currentPeriodStart. We keep the same
|
||||
* order/dependencies: window limits live on credits, and the plan period start
|
||||
* determines the summary window.
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U, parseResetTime } from "./shared.js";
|
||||
|
||||
const USAGE = U("commandcode");
|
||||
const BASE = USAGE.baseUrl || "https://api.commandcode.ai";
|
||||
const WHOAMI_URL = BASE + (USAGE.whoamiUrl || "/alpha/whoami");
|
||||
const CREDITS_URL = BASE + (USAGE.creditsUrl || "/alpha/billing/credits");
|
||||
const SUBSCRIPTIONS_URL =
|
||||
BASE + (USAGE.subscriptionsUrl || "/alpha/billing/subscriptions");
|
||||
const SUMMARY_URL = BASE + (USAGE.summaryUrl || "/alpha/usage/summary");
|
||||
|
||||
function buildHeaders(token) {
|
||||
return {
|
||||
Authorization: `Bearer ${token}`,
|
||||
Accept: "application/json",
|
||||
};
|
||||
}
|
||||
|
||||
/** Build a normalized quota row. `unit` is "$" — the API reports currency credits. */
|
||||
function makeQuota({ used, total, resetAt, unlimited = false, unit = "$" }) {
|
||||
const safeTotal = Math.max(0, Number(total) || 0);
|
||||
const safeUsed = Math.max(0, Number(used) || 0);
|
||||
if (unlimited || safeTotal === 0) {
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: 0,
|
||||
remainingPercentage: unlimited ? 100 : 0,
|
||||
resetAt: resetAt || null,
|
||||
unit,
|
||||
unlimited: true,
|
||||
};
|
||||
}
|
||||
const remaining = Math.max(0, safeTotal - safeUsed);
|
||||
const remainingPercentage = (remaining / safeTotal) * 100;
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: safeTotal,
|
||||
remainingPercentage,
|
||||
resetAt: resetAt || null,
|
||||
unit,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} apiKey - commandcode API key (user_...)
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getCommandCodeUsage(apiKey, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "CommandCode credential not available." };
|
||||
}
|
||||
|
||||
const headers = buildHeaders(apiKey);
|
||||
|
||||
try {
|
||||
// whoami resolves the org id (billing is org-scoped; null for personal).
|
||||
const whoamiRes = await proxyAwareFetch(
|
||||
WHOAMI_URL,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
if (whoamiRes.status === 401 || whoamiRes.status === 403) {
|
||||
return { message: "CommandCode credential invalid or expired." };
|
||||
}
|
||||
if (!whoamiRes.ok) {
|
||||
return { message: `CommandCode whoami API error (${whoamiRes.status}).` };
|
||||
}
|
||||
const whoami = await whoamiRes.json().catch(() => null);
|
||||
const orgId = whoami?.org?.id ?? null;
|
||||
|
||||
const orgQuery = orgId ? `?orgId=${encodeURIComponent(orgId)}` : "";
|
||||
|
||||
const [creditsRes, subsRes] = await Promise.all([
|
||||
proxyAwareFetch(
|
||||
CREDITS_URL + orgQuery,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
proxyAwareFetch(
|
||||
SUBSCRIPTIONS_URL + orgQuery,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
]);
|
||||
|
||||
if (
|
||||
creditsRes.status === 401 ||
|
||||
creditsRes.status === 403 ||
|
||||
subsRes.status === 401 ||
|
||||
subsRes.status === 403
|
||||
) {
|
||||
return { message: "CommandCode credential invalid or expired." };
|
||||
}
|
||||
if (!creditsRes.ok) {
|
||||
return {
|
||||
message: `CommandCode credits API error (${creditsRes.status}).`,
|
||||
};
|
||||
}
|
||||
|
||||
const credits = await creditsRes.json().catch(() => null);
|
||||
const subs = await subsRes.json().catch(() => null);
|
||||
|
||||
const subData = subs?.data;
|
||||
const planId = subData?.planId ?? null;
|
||||
const periodStart = subData?.currentPeriodStart ?? null;
|
||||
|
||||
// Summary needs `since`; the CLI falls back to first-of-month when the
|
||||
// subscription period start is unavailable.
|
||||
const since = periodStart || firstOfMonth();
|
||||
const summaryRes = await proxyAwareFetch(
|
||||
`${SUMMARY_URL}?since=${encodeURIComponent(since)}`,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
);
|
||||
const summary = summaryRes.ok
|
||||
? await summaryRes.json().catch(() => null)
|
||||
: null;
|
||||
|
||||
const quotas = {};
|
||||
const windowLimits = credits?.windowLimits || {};
|
||||
|
||||
const fiveHour = windowLimits.fiveHour;
|
||||
if (fiveHour && Number(fiveHour.cap) > 0) {
|
||||
quotas["5-hour window"] = makeQuota({
|
||||
used: fiveHour.used,
|
||||
total: fiveHour.cap,
|
||||
resetAt: parseResetTime(fiveHour.resetAt),
|
||||
});
|
||||
}
|
||||
|
||||
const weekly = windowLimits.weekly;
|
||||
if (weekly && Number(weekly.cap) > 0) {
|
||||
quotas["Weekly window"] = makeQuota({
|
||||
used: weekly.used,
|
||||
total: weekly.cap,
|
||||
resetAt: parseResetTime(weekly.resetAt),
|
||||
});
|
||||
}
|
||||
|
||||
// The credits API reports remaining balances (monthly/purchased/free),
|
||||
// not a total. The official CLI renders the monthly line as
|
||||
// `used = summary.totalCost`, `total = totalCost + remaining` — i.e.
|
||||
// the plan ceiling is the sum of what was consumed and what is left.
|
||||
const monthlyUsed =
|
||||
typeof summary?.totalCredits === "number"
|
||||
? summary.totalCredits
|
||||
: typeof summary?.totalCost === "number"
|
||||
? summary.totalCost
|
||||
: 0;
|
||||
|
||||
const creditsObj = credits?.credits || {};
|
||||
const remaining =
|
||||
Math.max(0, Number(creditsObj.monthlyCredits) || 0) +
|
||||
Math.max(0, Number(creditsObj.purchasedCredits) || 0) +
|
||||
Math.max(0, Number(creditsObj.freeCredits) || 0);
|
||||
const monthlyTotal = monthlyUsed + remaining;
|
||||
|
||||
if (monthlyTotal > 0 || monthlyUsed > 0) {
|
||||
quotas["Monthly credits"] = makeQuota({
|
||||
used: monthlyUsed,
|
||||
total: monthlyTotal,
|
||||
resetAt: periodStart ? undefined : null,
|
||||
});
|
||||
}
|
||||
|
||||
if (Object.keys(quotas).length === 0) {
|
||||
return {
|
||||
plan: planId || "CommandCode",
|
||||
message: "CommandCode connected, but no quota was reported.",
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
plan: planId || "CommandCode",
|
||||
quotas,
|
||||
periodBasis: summary?.periodBasis || "billing-period",
|
||||
};
|
||||
} catch (error) {
|
||||
return { message: `CommandCode usage error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
function firstOfMonth() {
|
||||
const now = new Date();
|
||||
return new Date(now.getFullYear(), now.getMonth(), 1).toISOString();
|
||||
}
|
||||
88
open-sse/services/usage/glm.js
Normal file
88
open-sse/services/usage/glm.js
Normal file
@@ -0,0 +1,88 @@
|
||||
/**
|
||||
* GLM Coding Plan usage (international + China regions)
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U } from "./shared.js";
|
||||
|
||||
// GLM quota endpoints (region-aware) — url from registry transport.usage
|
||||
const GLM_QUOTA_URLS = {
|
||||
international: U("glm").url,
|
||||
china: U("glm-cn").url,
|
||||
};
|
||||
|
||||
/**
|
||||
* GLM Coding Plan usage (international + China regions)
|
||||
* Supports both TOKENS_LIMIT and CREDIT_LIMIT and dynamic intervals (e.g. session 5h, weekly 7d).
|
||||
*/
|
||||
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "GLM API key not available." };
|
||||
}
|
||||
|
||||
const region = provider === "glm-cn" ? "china" : "international";
|
||||
const quotaUrl = GLM_QUOTA_URLS[region];
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(
|
||||
quotaUrl,
|
||||
{
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
Accept: "application/json",
|
||||
},
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
|
||||
if (!response.ok) {
|
||||
if (response.status === 401) {
|
||||
return { message: "GLM API key invalid or expired." };
|
||||
}
|
||||
return { message: `GLM quota API error (${response.status}).` };
|
||||
}
|
||||
|
||||
const json = await response.json();
|
||||
const data = json?.data && typeof json.data === "object" ? json.data : {};
|
||||
const limits = Array.isArray(data.limits) ? data.limits : [];
|
||||
const quotas = {};
|
||||
|
||||
for (const limit of limits) {
|
||||
// 1. Accept both TOKENS_LIMIT and CREDIT_LIMIT from GLM API
|
||||
if (!limit || (limit.type !== "TOKENS_LIMIT" && limit.type !== "CREDIT_LIMIT")) continue;
|
||||
const usedPercent = Number(limit.percentage) || 0;
|
||||
const resetMs = Number(limit.nextResetTime) || 0;
|
||||
const remaining = Math.max(0, 100 - usedPercent);
|
||||
|
||||
// 2. Map key dynamically based on type and period (unit) to avoid overwriting
|
||||
let key = "session";
|
||||
if (limit.unit === 3) {
|
||||
key = `Session (${limit.number}h)`;
|
||||
} else if (limit.unit === 6) {
|
||||
key = "Weekly (7d)";
|
||||
} else if (limit.type === "TOKENS_LIMIT") {
|
||||
key = "Tokens";
|
||||
} else {
|
||||
key = `Limit (${limit.number})`;
|
||||
}
|
||||
|
||||
quotas[key] = {
|
||||
used: usedPercent,
|
||||
total: 100,
|
||||
remaining,
|
||||
remainingPercentage: remaining,
|
||||
resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
const levelRaw = typeof data.level === "string" ? data.level : "";
|
||||
const plan = levelRaw
|
||||
? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase()
|
||||
: "Unknown";
|
||||
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: `GLM error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
@@ -161,6 +161,9 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
if (data.models) {
|
||||
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
|
||||
const importantModels = [
|
||||
'gemini-3.8-flash-high',
|
||||
'gemini-3.8-flash-medium',
|
||||
'gemini-3.8-flash-low',
|
||||
'gemini-3.7-flash-high',
|
||||
'gemini-3.7-flash-medium',
|
||||
'gemini-3.7-flash-low',
|
||||
|
||||
133
open-sse/services/usage/groq.js
Normal file
133
open-sse/services/usage/groq.js
Normal file
@@ -0,0 +1,133 @@
|
||||
/**
|
||||
* Groq usage — no dedicated quota endpoint. Rate-limit info instead rides on
|
||||
* every API response as x-ratelimit-* headers (requests + tokens, always
|
||||
* included). We piggyback on the models list (already used as
|
||||
* transport.validateUrl) so reading usage never costs tokens.
|
||||
*
|
||||
* Headers:
|
||||
* x-ratelimit-limit-requests / x-ratelimit-remaining-requests
|
||||
* x-ratelimit-limit-tokens / x-ratelimit-remaining-tokens
|
||||
* x-ratelimit-reset-requests / x-ratelimit-reset-tokens (duration strings, e.g. "2m59.56s")
|
||||
*
|
||||
* Docs: https://console.groq.com/docs/rate-limits
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U } from "./shared.js";
|
||||
|
||||
const MODELS_URL = U("groq").url;
|
||||
|
||||
// Groq reset headers are Go-style duration strings ("2m59.56s", "7.66s"), not
|
||||
// timestamps — parse the h/m/s/ms components and add them to now().
|
||||
function parseGroqDurationMs(value) {
|
||||
if (typeof value !== "string" || !value.trim()) return null;
|
||||
|
||||
const re = /(\d+(?:\.\d+)?)(ms|s|m|h)/g;
|
||||
let match;
|
||||
let totalMs = 0;
|
||||
let matched = false;
|
||||
while ((match = re.exec(value))) {
|
||||
matched = true;
|
||||
const amount = Number(match[1]);
|
||||
const unit = match[2];
|
||||
const unitMs = unit === "h" ? 3600000 : unit === "m" ? 60000 : unit === "ms" ? 1 : 1000;
|
||||
totalMs += amount * unitMs;
|
||||
}
|
||||
return matched ? totalMs : null;
|
||||
}
|
||||
|
||||
function resetAtFromDuration(value) {
|
||||
const ms = parseGroqDurationMs(value);
|
||||
return ms === null ? null : new Date(Date.now() + ms).toISOString();
|
||||
}
|
||||
|
||||
function buildRateLimitQuota(headers, limitKey, remainingKey, resetKey) {
|
||||
// headers.get() returns null when absent, and Number(null) is 0 (a finite
|
||||
// number) — check presence explicitly so a missing header can't masquerade
|
||||
// as a real "0 remaining" quota.
|
||||
const limitRaw = headers.get(limitKey);
|
||||
const remainingRaw = headers.get(remainingKey);
|
||||
if (limitRaw === null || remainingRaw === null) return null;
|
||||
|
||||
const limit = Number(limitRaw);
|
||||
const remaining = Number(remainingRaw);
|
||||
if (!Number.isFinite(limit) || !Number.isFinite(remaining)) return null;
|
||||
|
||||
return {
|
||||
used: Math.max(0, limit - remaining),
|
||||
total: limit,
|
||||
resetAt: resetAtFromDuration(headers.get(resetKey)),
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string|null|undefined} apiKey
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getGroqUsage(apiKey, proxyOptions = null) {
|
||||
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
|
||||
return { message: "Groq API key not available. Add a key to view usage." };
|
||||
}
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(
|
||||
MODELS_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey.trim()}`,
|
||||
Accept: "application/json",
|
||||
},
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
return { plan: "Groq", message: "Groq authentication failed. Check the API key." };
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
const errText = await response.text().catch(() => "");
|
||||
return {
|
||||
plan: "Groq",
|
||||
message: `Groq usage API error (${response.status})${errText ? `: ${errText.slice(0, 120)}` : ""}`,
|
||||
};
|
||||
}
|
||||
|
||||
// The quota data lives in headers, not the body — drain it so the
|
||||
// connection can be released without needing the payload.
|
||||
await response.text().catch(() => {});
|
||||
|
||||
const requests = buildRateLimitQuota(
|
||||
response.headers,
|
||||
"x-ratelimit-limit-requests",
|
||||
"x-ratelimit-remaining-requests",
|
||||
"x-ratelimit-reset-requests",
|
||||
);
|
||||
const tokens = buildRateLimitQuota(
|
||||
response.headers,
|
||||
"x-ratelimit-limit-tokens",
|
||||
"x-ratelimit-remaining-tokens",
|
||||
"x-ratelimit-reset-tokens",
|
||||
);
|
||||
|
||||
if (!requests && !tokens) {
|
||||
// Key is valid (request succeeded) but no rate-limit bucket reported —
|
||||
// distinguish "not tracked yet" from an auth/error state.
|
||||
return {
|
||||
plan: "Groq",
|
||||
message: "Groq connected. No rate-limit data reported for this key yet.",
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
const quotas = {};
|
||||
if (requests) quotas["Requests"] = requests;
|
||||
if (tokens) quotas["Tokens"] = tokens;
|
||||
|
||||
return { plan: "Groq", quotas };
|
||||
} catch (error) {
|
||||
return { message: `Groq error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
@@ -5,11 +5,8 @@
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U } from "./shared.js";
|
||||
|
||||
// GLM quota endpoints (region-aware) — url from registry transport.usage
|
||||
const GLM_QUOTA_URLS = {
|
||||
international: U("glm").url,
|
||||
china: U("glm-cn").url,
|
||||
};
|
||||
export { getGlmUsage } from "./glm.js";
|
||||
|
||||
|
||||
// Vercel AI Gateway credits endpoint
|
||||
// Returns { balance: "95.50", total_used: "4.50" } (USD as decimal strings).
|
||||
@@ -112,63 +109,7 @@ export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* GLM Coding Plan usage (international + China regions)
|
||||
*/
|
||||
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "GLM API key not available." };
|
||||
}
|
||||
|
||||
const region = provider === "glm-cn" ? "china" : "international";
|
||||
const quotaUrl = GLM_QUOTA_URLS[region];
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(quotaUrl, {
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
Accept: "application/json",
|
||||
},
|
||||
}, proxyOptions);
|
||||
|
||||
if (!response.ok) {
|
||||
if (response.status === 401) {
|
||||
return { message: "GLM API key invalid or expired." };
|
||||
}
|
||||
return { message: `GLM quota API error (${response.status}).` };
|
||||
}
|
||||
|
||||
const json = await response.json();
|
||||
const data = json?.data && typeof json.data === "object" ? json.data : {};
|
||||
const limits = Array.isArray(data.limits) ? data.limits : [];
|
||||
const quotas = {};
|
||||
|
||||
for (const limit of limits) {
|
||||
if (!limit || limit.type !== "TOKENS_LIMIT") continue;
|
||||
const usedPercent = Number(limit.percentage) || 0;
|
||||
const resetMs = Number(limit.nextResetTime) || 0;
|
||||
const remaining = Math.max(0, 100 - usedPercent);
|
||||
|
||||
quotas["session"] = {
|
||||
used: usedPercent,
|
||||
total: 100,
|
||||
remaining,
|
||||
remainingPercentage: remaining,
|
||||
resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
const levelRaw = typeof data.level === "string" ? data.level : "";
|
||||
const plan = levelRaw
|
||||
? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase()
|
||||
: "Unknown";
|
||||
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: `GLM error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Vercel AI Gateway usage — credit balance for the API key
|
||||
|
||||
107
open-sse/services/usage/opencode-go.js
Normal file
107
open-sse/services/usage/opencode-go.js
Normal file
@@ -0,0 +1,107 @@
|
||||
/**
|
||||
* OpenCode Go usage — GET https://opencode.ai/zen/go/v1/usage
|
||||
* Auth: Bearer <apiKey>
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { parseResetTime, toFiniteNumber, U } from "./shared.js";
|
||||
|
||||
const USAGE_URL = U("opencode-go").url;
|
||||
const QUOTA_NAMES = {
|
||||
rolling: "Rolling",
|
||||
weekly: "Weekly",
|
||||
monthly: "Monthly",
|
||||
};
|
||||
|
||||
function parsePercent(value) {
|
||||
if (typeof value === "number" && Number.isFinite(value)) return value;
|
||||
if (typeof value === "string" && value.trim()) {
|
||||
const parsed = Number(value);
|
||||
if (Number.isFinite(parsed)) return parsed;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
export async function getOpenCodeGoUsage(apiKey = null, proxyOptions = null) {
|
||||
if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) {
|
||||
return {
|
||||
message: "OpenCode Go API key not available. Add a key to view usage.",
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(
|
||||
USAGE_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey.trim()}`,
|
||||
Accept: "application/json",
|
||||
},
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
|
||||
if (response.status === 401) {
|
||||
return {
|
||||
plan: "OpenCode Go",
|
||||
message: "OpenCode Go authentication failed. Check the API key.",
|
||||
};
|
||||
}
|
||||
|
||||
if (response.status === 403) {
|
||||
const error = await response.json().catch(() => null);
|
||||
const subscriptionRequired = error?.error?.type === "EntitlementError";
|
||||
return {
|
||||
plan: "OpenCode Go",
|
||||
message: subscriptionRequired
|
||||
? "OpenCode Go subscription required for this API key."
|
||||
: "OpenCode Go access forbidden for this API key.",
|
||||
};
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
return {
|
||||
plan: "OpenCode Go",
|
||||
message: `OpenCode Go usage API error (${response.status}).`,
|
||||
};
|
||||
}
|
||||
|
||||
const data = await response.json().catch(() => null);
|
||||
if (!data?.usage || typeof data.usage !== "object") {
|
||||
return {
|
||||
plan: "OpenCode Go",
|
||||
message: "OpenCode Go usage response did not contain quota data.",
|
||||
};
|
||||
}
|
||||
|
||||
const quotas = {};
|
||||
for (const [period, name] of Object.entries(QUOTA_NAMES)) {
|
||||
const quota = data.usage[period];
|
||||
if (!quota || typeof quota !== "object") continue;
|
||||
const percent = parsePercent(quota.percent);
|
||||
if (percent === null) continue;
|
||||
const used = Math.max(0, Math.min(100, toFiniteNumber(percent, 0)));
|
||||
quotas[name] = {
|
||||
used,
|
||||
total: 100,
|
||||
remaining: 100 - used,
|
||||
remainingPercentage: 100 - used,
|
||||
resetAt: parseResetTime(quota.resetsAt),
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
if (Object.keys(quotas).length === 0) {
|
||||
return {
|
||||
plan: "OpenCode Go",
|
||||
message: "OpenCode Go usage response did not contain valid quota data.",
|
||||
};
|
||||
}
|
||||
|
||||
return { plan: "OpenCode Go", quotas };
|
||||
} catch (error) {
|
||||
return { message: `OpenCode Go error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
326
open-sse/services/usage/xai.js
Normal file
326
open-sse/services/usage/xai.js
Normal file
@@ -0,0 +1,326 @@
|
||||
/**
|
||||
* xAI (Grok) OAuth usage handler
|
||||
*
|
||||
* SuperGrok quota is split across two upstream surfaces (OAuth only):
|
||||
*
|
||||
* 1) Monthly / API usage allotment (JSON)
|
||||
* GET https://cli-chat-proxy.grok.com/v1/billing
|
||||
* {
|
||||
* "config": {
|
||||
* "monthlyLimit": { "val": 15000 },
|
||||
* "used": { "val": 733 },
|
||||
* "onDemandCap": { "val": 0 },
|
||||
* "billingPeriodStart": "...",
|
||||
* "billingPeriodEnd": "..."
|
||||
* }
|
||||
* }
|
||||
*
|
||||
* 2) Weekly SuperGrok limit (grpc-web protobuf)
|
||||
* POST https://grok.com/grok_api_v2.GrokBuildBilling/GetGrokCreditsConfig
|
||||
* Empty request frame; response message1 contains:
|
||||
* usedPercent (float32), window start/end timestamps, nested windows.
|
||||
* This is what the grok.com usage page labels "Weekly limit" / "Resets …".
|
||||
*
|
||||
* Plan label comes from cli-chat-proxy settings:
|
||||
* GET https://cli-chat-proxy.grok.com/v1/settings → subscription_tier_display
|
||||
*
|
||||
* Note: grok.com/rest/rate-limits is a short chat-window (e.g. 2h query count)
|
||||
* behind Cloudflare browser cookies — not usable with pure OAuth bearer.
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U, parseResetTime, toFiniteNumber } from "./shared.js";
|
||||
|
||||
// Empty grpc-web request frame: flag(0) + length(0) + no payload.
|
||||
const GRPC_WEB_EMPTY_FRAME = Buffer.from([0x00, 0x00, 0x00, 0x00, 0x00]);
|
||||
|
||||
function moneyVal(wrapper) {
|
||||
if (wrapper == null) return null;
|
||||
if (typeof wrapper === "number") return toFiniteNumber(wrapper, null);
|
||||
if (typeof wrapper === "object" && wrapper.val != null) {
|
||||
return toFiniteNumber(wrapper.val, null);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function authHeaders(accessToken, extra = {}) {
|
||||
return {
|
||||
Authorization: `Bearer ${accessToken}`,
|
||||
...extra,
|
||||
};
|
||||
}
|
||||
|
||||
function readVarint(buf, offset) {
|
||||
let val = 0;
|
||||
let shift = 0;
|
||||
let i = offset;
|
||||
while (i < buf.length) {
|
||||
const b = buf[i++];
|
||||
val |= (b & 0x7f) << shift;
|
||||
if ((b & 0x80) === 0) return { value: val >>> 0, offset: i };
|
||||
shift += 7;
|
||||
if (shift > 35) break;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal protobuf decoder for GetGrokCreditsConfig.
|
||||
* Only understands varint / fixed32 / fixed64 / length-delimited.
|
||||
*/
|
||||
function parseProtobufFields(buf) {
|
||||
const fields = [];
|
||||
let i = 0;
|
||||
while (i < buf.length) {
|
||||
const key = readVarint(buf, i);
|
||||
if (!key) break;
|
||||
i = key.offset;
|
||||
const field = key.value >>> 3;
|
||||
const wt = key.value & 7;
|
||||
|
||||
if (wt === 0) {
|
||||
const v = readVarint(buf, i);
|
||||
if (!v) break;
|
||||
i = v.offset;
|
||||
fields.push({ field, type: "varint", value: v.value });
|
||||
} else if (wt === 1) {
|
||||
if (i + 8 > buf.length) break;
|
||||
fields.push({ field, type: "fixed64", value: buf.subarray(i, i + 8) });
|
||||
i += 8;
|
||||
} else if (wt === 5) {
|
||||
if (i + 4 > buf.length) break;
|
||||
fields.push({
|
||||
field,
|
||||
type: "fixed32",
|
||||
value: buf.readFloatLE(i),
|
||||
});
|
||||
i += 4;
|
||||
} else if (wt === 2) {
|
||||
const ln = readVarint(buf, i);
|
||||
if (!ln) break;
|
||||
i = ln.offset;
|
||||
if (i + ln.value > buf.length) break;
|
||||
fields.push({
|
||||
field,
|
||||
type: "bytes",
|
||||
value: buf.subarray(i, i + ln.value),
|
||||
});
|
||||
i += ln.value;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return fields;
|
||||
}
|
||||
|
||||
function parseTimestamp(bytes) {
|
||||
if (!bytes || !bytes.length) return null;
|
||||
const fields = parseProtobufFields(bytes);
|
||||
const seconds = fields.find((f) => f.field === 1 && f.type === "varint")?.value;
|
||||
if (!Number.isFinite(seconds) || seconds <= 0) return null;
|
||||
return new Date(seconds * 1000).toISOString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse grpc-web response bytes from GetGrokCreditsConfig.
|
||||
* Returns { usedPercent, resetAt, periodStart } or null.
|
||||
*/
|
||||
export function parseGrokCreditsConfig(raw) {
|
||||
if (!raw || !raw.length) return null;
|
||||
const buf = Buffer.isBuffer(raw) ? raw : Buffer.from(raw);
|
||||
|
||||
// grpc-web data frame: 1-byte flag + 4-byte big-endian length + message
|
||||
if (buf.length < 5) return null;
|
||||
const flag = buf[0];
|
||||
// Data frames have flag 0; ignore trailer frames (flag 0x80).
|
||||
if (flag !== 0) return null;
|
||||
const msgLen = buf.readUInt32BE(1);
|
||||
if (msgLen <= 0 || 5 + msgLen > buf.length) return null;
|
||||
const msg = buf.subarray(5, 5 + msgLen);
|
||||
|
||||
// Response is typically { 1: CreditsConfig }
|
||||
const top = parseProtobufFields(msg);
|
||||
const configBytes = top.find((f) => f.field === 1 && f.type === "bytes")?.value || msg;
|
||||
const fields = parseProtobufFields(configBytes);
|
||||
|
||||
const usedPercentRaw = fields.find((f) => f.field === 1 && f.type === "fixed32")?.value;
|
||||
const periodStart = parseTimestamp(fields.find((f) => f.field === 4 && f.type === "bytes")?.value);
|
||||
const periodEnd = parseTimestamp(fields.find((f) => f.field === 5 && f.type === "bytes")?.value);
|
||||
|
||||
if (usedPercentRaw == null || !Number.isFinite(usedPercentRaw)) return null;
|
||||
|
||||
const usedPercent = Math.max(0, Math.min(100, usedPercentRaw));
|
||||
return {
|
||||
usedPercent,
|
||||
remainingPercent: Math.max(0, 100 - usedPercent),
|
||||
periodStart,
|
||||
resetAt: periodEnd,
|
||||
};
|
||||
}
|
||||
|
||||
async function fetchBilling(accessToken, billingUrl, proxyOptions) {
|
||||
const response = await proxyAwareFetch(billingUrl, {
|
||||
method: "GET",
|
||||
headers: authHeaders(accessToken, { Accept: "application/json" }),
|
||||
}, proxyOptions);
|
||||
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
return { error: "auth", status: response.status };
|
||||
}
|
||||
if (!response.ok) {
|
||||
return { error: "http", status: response.status };
|
||||
}
|
||||
|
||||
const data = await response.json().catch(() => null);
|
||||
if (!data || typeof data !== "object") {
|
||||
return { error: "json" };
|
||||
}
|
||||
return { data };
|
||||
}
|
||||
|
||||
async function fetchWeeklyCredits(accessToken, creditsUrl, proxyOptions) {
|
||||
if (!creditsUrl) return null;
|
||||
try {
|
||||
const response = await proxyAwareFetch(creditsUrl, {
|
||||
method: "POST",
|
||||
headers: authHeaders(accessToken, {
|
||||
"Content-Type": "application/grpc-web+proto",
|
||||
"x-grpc-web": "1",
|
||||
"x-user-agent": "connect-es/2.1.1",
|
||||
Accept: "*/*",
|
||||
Origin: "https://grok.com",
|
||||
Referer: "https://grok.com/?_s=usage",
|
||||
}),
|
||||
body: GRPC_WEB_EMPTY_FRAME,
|
||||
}, proxyOptions);
|
||||
|
||||
if (!response.ok) return null;
|
||||
const ab = await response.arrayBuffer();
|
||||
return parseGrokCreditsConfig(Buffer.from(ab));
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchPlanLabel(accessToken, settingsUrl, proxyOptions) {
|
||||
if (!settingsUrl) return null;
|
||||
try {
|
||||
const response = await proxyAwareFetch(settingsUrl, {
|
||||
method: "GET",
|
||||
headers: authHeaders(accessToken, { Accept: "application/json" }),
|
||||
}, proxyOptions);
|
||||
if (!response.ok) return null;
|
||||
const data = await response.json().catch(() => null);
|
||||
const label = data?.subscription_tier_display;
|
||||
return typeof label === "string" && label.trim() ? label.trim() : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} accessToken - xAI OAuth access token
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getXaiUsage(accessToken, proxyOptions = null) {
|
||||
if (!accessToken) {
|
||||
return { message: "xAI usage unavailable: no access token. Re-authorize the connection." };
|
||||
}
|
||||
|
||||
const cfg = U("xai") || {};
|
||||
const billingUrl = cfg.url;
|
||||
if (!billingUrl) {
|
||||
return { message: "xAI usage endpoint is not configured." };
|
||||
}
|
||||
|
||||
try {
|
||||
const [billingResult, weekly, planLabel] = await Promise.all([
|
||||
fetchBilling(accessToken, billingUrl, proxyOptions),
|
||||
fetchWeeklyCredits(accessToken, cfg.creditsUrl, proxyOptions),
|
||||
fetchPlanLabel(accessToken, cfg.settingsUrl, proxyOptions),
|
||||
]);
|
||||
|
||||
if (billingResult.error === "auth") {
|
||||
return { message: "xAI OAuth token expired or unauthorized. Please re-authorize." };
|
||||
}
|
||||
|
||||
const quotas = {};
|
||||
let periodStart = null;
|
||||
let periodEnd = null;
|
||||
let onDemandCap = 0;
|
||||
|
||||
if (billingResult.data) {
|
||||
const config =
|
||||
billingResult.data.config && typeof billingResult.data.config === "object"
|
||||
? billingResult.data.config
|
||||
: billingResult.data;
|
||||
const monthlyLimit = moneyVal(config.monthlyLimit);
|
||||
const used = moneyVal(config.used);
|
||||
onDemandCap = moneyVal(config.onDemandCap) ?? 0;
|
||||
periodEnd = parseResetTime(config.billingPeriodEnd);
|
||||
periodStart = parseResetTime(config.billingPeriodStart);
|
||||
|
||||
// Absolute credit counts — do NOT put remaining credits on `remaining`
|
||||
// (QuotaTable treats remaining as a 0-100 percentage; same pitfall as Qoder).
|
||||
if (monthlyLimit != null && monthlyLimit > 0) {
|
||||
const usedSafe = Math.max(0, used ?? 0);
|
||||
quotas.api_usage = {
|
||||
used: usedSafe,
|
||||
total: monthlyLimit,
|
||||
remainingCredits: Math.max(0, monthlyLimit - usedSafe),
|
||||
unit: "credits",
|
||||
resetAt: periodEnd,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
if (onDemandCap > 0) {
|
||||
quotas.on_demand = {
|
||||
used: 0,
|
||||
total: onDemandCap,
|
||||
remainingCredits: onDemandCap,
|
||||
unit: "credits",
|
||||
resetAt: periodEnd,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Weekly SuperGrok window — percentage-based like Claude/Codex windows.
|
||||
if (weekly) {
|
||||
quotas.weekly = {
|
||||
used: weekly.usedPercent,
|
||||
total: 100,
|
||||
remaining: weekly.remainingPercent,
|
||||
remainingPercentage: weekly.remainingPercent,
|
||||
resetAt: weekly.resetAt || null,
|
||||
unlimited: false,
|
||||
};
|
||||
if (!periodStart && weekly.periodStart) periodStart = weekly.periodStart;
|
||||
}
|
||||
|
||||
if (Object.keys(quotas).length === 0) {
|
||||
const statusHint =
|
||||
billingResult.error === "http"
|
||||
? ` Billing API temporarily unavailable (${billingResult.status}).`
|
||||
: "";
|
||||
return {
|
||||
plan: planLabel || "xAI",
|
||||
message: `xAI connected. No quota allotment reported for this account.${statusHint}`,
|
||||
periodStart,
|
||||
periodEnd,
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
plan: planLabel || "xAI",
|
||||
periodStart,
|
||||
periodEnd,
|
||||
onDemandCap,
|
||||
quotas,
|
||||
};
|
||||
} catch (error) {
|
||||
return { message: `xAI connected. Unable to fetch billing: ${error.message}` };
|
||||
}
|
||||
}
|
||||
222
open-sse/services/usage/zed.js
Normal file
222
open-sse/services/usage/zed.js
Normal file
@@ -0,0 +1,222 @@
|
||||
/**
|
||||
* Zed usage — GET https://cloud.zed.dev/client/users/me
|
||||
* Auth: Authorization: {user_id} {access_token}
|
||||
*
|
||||
* Quota rows are derived from plan.usage (edit_predictions, optional model_requests)
|
||||
* and subscription_period.ended_at for billing-cycle reset.
|
||||
*/
|
||||
|
||||
import { fetchZedAuthenticatedUser } from "../../shared/zedAuth.js";
|
||||
import { parseResetTime, toFiniteNumber } from "./shared.js";
|
||||
|
||||
/** Map plan_v3 ids to dashboard labels (CodexBar-compatible). */
|
||||
export function formatZedPlanLabel(rawPlan) {
|
||||
const raw = String(rawPlan || "").trim();
|
||||
if (!raw) return "Zed";
|
||||
switch (raw.toLowerCase()) {
|
||||
case "zed_free":
|
||||
return "Zed Free";
|
||||
case "zed_pro":
|
||||
return "Zed Pro";
|
||||
case "zed_pro_trial":
|
||||
return "Zed Pro Trial";
|
||||
case "zed_student":
|
||||
return "Zed Student";
|
||||
case "zed_business":
|
||||
return "Zed Business";
|
||||
default:
|
||||
return raw
|
||||
.replace(/_/g, " ")
|
||||
.split(/\s+/)
|
||||
.map((word) => word.charAt(0).toUpperCase() + word.slice(1).toLowerCase())
|
||||
.join(" ");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse Zed UsageLimit JSON: "unlimited", a number, or { limited: N }.
|
||||
*/
|
||||
export function parseZedUsageLimit(limit) {
|
||||
if (limit == null) return { unlimited: false, total: 0 };
|
||||
|
||||
if (limit === "unlimited" || limit?.unlimited === true) {
|
||||
return { unlimited: true, total: 0 };
|
||||
}
|
||||
|
||||
if (typeof limit === "number" && Number.isFinite(limit)) {
|
||||
return { unlimited: false, total: Math.max(0, limit) };
|
||||
}
|
||||
|
||||
if (typeof limit === "string") {
|
||||
const trimmed = limit.trim();
|
||||
if (trimmed === "unlimited") return { unlimited: true, total: 0 };
|
||||
const parsed = Number(trimmed);
|
||||
if (Number.isFinite(parsed)) return { unlimited: false, total: Math.max(0, parsed) };
|
||||
}
|
||||
|
||||
const limited = limit.limited ?? limit.Limited;
|
||||
if (typeof limited === "number" && Number.isFinite(limited)) {
|
||||
return { unlimited: false, total: Math.max(0, limited) };
|
||||
}
|
||||
|
||||
return { unlimited: false, total: 0 };
|
||||
}
|
||||
|
||||
/** limit `{ limited: 0 }` on Pro/Student means token billing, not a 0-cap request quota. */
|
||||
export function isZedTokenBillingModelRequestsLimit(limitRaw) {
|
||||
const info = parseZedUsageLimit(limitRaw);
|
||||
return !info.unlimited && info.total === 0;
|
||||
}
|
||||
|
||||
function makeZedQuotaRow(name, usedRaw, limitRaw, resetAt = null) {
|
||||
const used = Math.max(0, toFiniteNumber(usedRaw, 0));
|
||||
const limitInfo = parseZedUsageLimit(limitRaw);
|
||||
|
||||
if (limitInfo.unlimited) {
|
||||
return {
|
||||
used,
|
||||
total: 0,
|
||||
remainingPercentage: 100,
|
||||
resetAt: resetAt || null,
|
||||
unlimited: true,
|
||||
};
|
||||
}
|
||||
|
||||
const total = limitInfo.total;
|
||||
if (total <= 0) {
|
||||
return {
|
||||
used,
|
||||
total: 0,
|
||||
remainingPercentage: 0,
|
||||
resetAt: resetAt || null,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
const clampedUsed = Math.min(used, total);
|
||||
const remaining = Math.max(0, total - clampedUsed);
|
||||
return {
|
||||
used: clampedUsed,
|
||||
total,
|
||||
remainingPercentage: (remaining / total) * 100,
|
||||
resetAt: resetAt || null,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
function usageBucketLimit(bucket) {
|
||||
if (!bucket || typeof bucket !== "object") return null;
|
||||
if (bucket.limit != null) return bucket.limit;
|
||||
return bucket;
|
||||
}
|
||||
|
||||
/**
|
||||
* Map /client/users/me JSON → { plan, quotas, message } for the dashboard.
|
||||
*/
|
||||
export function parseZedAuthenticatedUserUsage(userInfo) {
|
||||
const plan = userInfo?.plan || {};
|
||||
const planId =
|
||||
plan.plan_v3 || plan.plan_v2 || plan.plan || userInfo?.plan_v3 || null;
|
||||
const resetAt =
|
||||
parseResetTime(plan.subscription_period?.ended_at) ||
|
||||
parseResetTime(plan.subscriptionPeriod?.endedAt) ||
|
||||
null;
|
||||
|
||||
const quotas = {};
|
||||
const usage = plan.usage || {};
|
||||
|
||||
const editPredictions = usage.edit_predictions || usage.editPredictions;
|
||||
if (editPredictions) {
|
||||
quotas["Edit Predictions"] = makeZedQuotaRow(
|
||||
"Edit Predictions",
|
||||
editPredictions.used,
|
||||
editPredictions.limit,
|
||||
resetAt,
|
||||
);
|
||||
}
|
||||
|
||||
const modelRequests = usage.model_requests || usage.modelRequests;
|
||||
if (modelRequests) {
|
||||
const limitRaw =
|
||||
modelRequests.limit != null
|
||||
? modelRequests.limit
|
||||
: usageBucketLimit(modelRequests)?.limit;
|
||||
const limitInfo = parseZedUsageLimit(limitRaw);
|
||||
// Token-billed plans report model_requests.limit=0 — not a request quota.
|
||||
if (limitInfo.unlimited || limitInfo.total > 0) {
|
||||
quotas["Hosted Model Requests"] = makeZedQuotaRow(
|
||||
"Hosted Model Requests",
|
||||
modelRequests.used,
|
||||
limitRaw,
|
||||
resetAt,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const tokenBillingNote =
|
||||
modelRequests &&
|
||||
isZedTokenBillingModelRequestsLimit(
|
||||
modelRequests.limit ?? usageBucketLimit(modelRequests)?.limit,
|
||||
)
|
||||
? "Hosted AI models are billed per token (not request count). Edit Predictions are tracked below. Token spend is on dashboard.zed.dev."
|
||||
: null;
|
||||
|
||||
let planLabel = formatZedPlanLabel(planId);
|
||||
if (plan.trial_started_at || plan.trialStartedAt) {
|
||||
if (!/trial/i.test(planLabel)) planLabel = `${planLabel} (Trial active)`;
|
||||
}
|
||||
|
||||
let message = tokenBillingNote;
|
||||
if (plan.has_overdue_invoices || plan.hasOverdueInvoices) {
|
||||
message = "This Zed account has overdue invoices. Usage may be blocked until billing is resolved.";
|
||||
}
|
||||
|
||||
return {
|
||||
plan: planLabel,
|
||||
quotas,
|
||||
message,
|
||||
hasOverdueInvoices: !!(plan.has_overdue_invoices || plan.hasOverdueInvoices),
|
||||
trialStarted: !!(plan.trial_started_at || plan.trialStartedAt),
|
||||
planId: planId || null,
|
||||
resetAt,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string|null|undefined} accessToken
|
||||
* @param {object|null|undefined} providerSpecificData
|
||||
* @param {object|null|undefined} proxyOptions
|
||||
*/
|
||||
export async function getZedUsage(
|
||||
accessToken = null,
|
||||
providerSpecificData = {},
|
||||
proxyOptions = null,
|
||||
) {
|
||||
const psd = providerSpecificData || {};
|
||||
const userId = psd.userId;
|
||||
|
||||
if (!accessToken || typeof accessToken !== "string" || !accessToken.trim()) {
|
||||
return { message: "Zed access token not available. Re-connect Zed to view quota." };
|
||||
}
|
||||
if (!userId) {
|
||||
return { message: "Zed credential is missing user id. Re-connect Zed to view quota." };
|
||||
}
|
||||
|
||||
const credentials = {
|
||||
accessToken: accessToken.trim(),
|
||||
providerSpecificData: psd,
|
||||
};
|
||||
|
||||
try {
|
||||
const userInfo = await fetchZedAuthenticatedUser(credentials, { proxyOptions });
|
||||
return parseZedAuthenticatedUserUsage(userInfo);
|
||||
} catch (error) {
|
||||
const status = error?.status;
|
||||
if (status === 401 || status === 403) {
|
||||
return {
|
||||
message: "Zed authentication failed. Sign in again from the dashboard or Zed editor.",
|
||||
};
|
||||
}
|
||||
return { message: `Zed error: ${error.message || "Failed to fetch quota"}` };
|
||||
}
|
||||
}
|
||||
@@ -54,10 +54,13 @@ export const QODER_MODEL_MAP = {
|
||||
lite: "lite",
|
||||
// Frontier models
|
||||
qmodel: "qmodel",
|
||||
qfmodel: "qfmodel",
|
||||
qmodel_latest: "qmodel_latest",
|
||||
qmodel_38max: "qmodel_38max",
|
||||
dmodel: "dmodel",
|
||||
dfmodel: "dfmodel",
|
||||
gm51model: "gm51model",
|
||||
gmodel: "gmodel",
|
||||
gfmodel: "gfmodel",
|
||||
kmodel: "kmodel",
|
||||
mmodel: "mmodel",
|
||||
};
|
||||
|
||||
@@ -172,8 +172,8 @@ function getSystemId(credentials) {
|
||||
);
|
||||
}
|
||||
|
||||
async function fetchJson(url, options) {
|
||||
const res = await proxyAwareFetch(url, options);
|
||||
async function fetchJson(url, options, proxyOptions = null) {
|
||||
const res = await proxyAwareFetch(url, options, proxyOptions);
|
||||
const text = await res.text();
|
||||
let data = null;
|
||||
if (text) {
|
||||
@@ -203,11 +203,15 @@ export async function fetchZedAuthenticatedUser(credentials, options = {}) {
|
||||
const systemId = getSystemId(credentials);
|
||||
if (systemId) headers[ZED_HEADERS.systemId] = systemId;
|
||||
|
||||
return fetchJson(zedUrl(config, "cloudBaseUrl", "/client/users/me", ZED_CLOUD_BASE_URL), {
|
||||
method: "GET",
|
||||
headers,
|
||||
signal: options.signal ?? undefined,
|
||||
});
|
||||
return fetchJson(
|
||||
zedUrl(config, "cloudBaseUrl", "/client/users/me", ZED_CLOUD_BASE_URL),
|
||||
{
|
||||
method: "GET",
|
||||
headers,
|
||||
signal: options.signal ?? undefined,
|
||||
},
|
||||
options.proxyOptions ?? null,
|
||||
);
|
||||
}
|
||||
|
||||
function normalizeOrganizationId(value) {
|
||||
|
||||
@@ -58,6 +58,15 @@ export function extractThinking(body) {
|
||||
return { mode: "level", level: e };
|
||||
}
|
||||
|
||||
// OpenAI chat / Responses shape — check effort first (zai sends both thinking object and reasoning.effort)
|
||||
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
|
||||
if (typeof effort === "string" && effort) {
|
||||
const e = effort.toLowerCase();
|
||||
if (e === "none" || e === "off") return { mode: "none" };
|
||||
if (e === "auto") return { mode: "auto" };
|
||||
return { mode: "level", level: e };
|
||||
}
|
||||
|
||||
// Claude shape
|
||||
const t = body.thinking;
|
||||
if (t && typeof t === "object") {
|
||||
@@ -69,15 +78,6 @@ export function extractThinking(body) {
|
||||
}
|
||||
}
|
||||
|
||||
// OpenAI chat / Responses shape
|
||||
const effort = body.reasoning_effort ?? (typeof body.reasoning === "object" ? body.reasoning?.effort : null);
|
||||
if (typeof effort === "string" && effort) {
|
||||
const e = effort.toLowerCase();
|
||||
if (e === "none" || e === "off") return { mode: "none" };
|
||||
if (e === "auto") return { mode: "auto" };
|
||||
return { mode: "level", level: e };
|
||||
}
|
||||
|
||||
// Gemini shape (top-level, generationConfig, or request envelope)
|
||||
const tc = body.thinkingConfig || body.generationConfig?.thinkingConfig || body.request?.generationConfig?.thinkingConfig;
|
||||
if (tc && typeof tc === "object") {
|
||||
@@ -105,12 +105,16 @@ export function extractThinking(body) {
|
||||
// at the call-site where intent is snapshotted before format translation.
|
||||
export const captureThinking = extractThinking;
|
||||
|
||||
// Resolve thinking format: provider override > capability > derive(targetFormat).
|
||||
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
|
||||
|
||||
function resolveFormat(targetFormat, model, provider) {
|
||||
const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null;
|
||||
if (providerFmt) return providerFmt;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (caps.thinkingFormat) return caps.thinkingFormat;
|
||||
const isOpenAIWire = targetFormat === "openai" || targetFormat === "openai-responses";
|
||||
if (caps.thinkingFormat && !(isOpenAIWire && NATIVE_ONLY_FORMATS.has(caps.thinkingFormat))) {
|
||||
return caps.thinkingFormat;
|
||||
}
|
||||
return FORMAT_TO_NATIVE[targetFormat] || "openai";
|
||||
}
|
||||
|
||||
@@ -237,14 +241,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
}
|
||||
case "claude-adaptive": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
// output_config.effort alone does NOT turn thinking on: Anthropic requires
|
||||
// an explicit thinking:{type:"adaptive"} on Opus 4.6/4.7/4.8 and Sonnet 4.6
|
||||
// ("thinking is off unless you explicitly set it"), and Anthropic-compatible
|
||||
// shims (e.g. GitHub Copilot /v1/messages) default thinking off even for
|
||||
// Sonnet 5. Send both fields — the documented adaptive-thinking shape.
|
||||
body.thinking = { type: "adaptive" };
|
||||
// Models that can disable thinking need the explicit adaptive switch.
|
||||
// Permanently adaptive models such as Fable 5.1 accept effort directly.
|
||||
if (canDisable) body.thinking = { type: "adaptive" };
|
||||
else delete body.thinking;
|
||||
const level = toLevel(eff);
|
||||
body.output_config = { effort: level === "xhigh" ? "high" : level };
|
||||
body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level };
|
||||
break;
|
||||
}
|
||||
case "claude-budget": {
|
||||
@@ -270,6 +272,18 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
// Z.ai ignores thinking.disabled → must use enable_thinking:false to turn off.
|
||||
if (none && canDisable) { body.enable_thinking = false; delete body.thinking; break; }
|
||||
body.thinking = { type: "enabled" };
|
||||
// reasoning_effort is only read by z.ai from GLM-5.2 onward — older GLM ignores it
|
||||
// (see thinkingEffortSupported in capabilities.js). Skip on unsupported models so we
|
||||
// don't send a field the API doesn't recognize.
|
||||
if (caps.thinkingEffortSupported) {
|
||||
const zaiLvl = toLevel(eff);
|
||||
// GLM-5.3 only accepts exactly low|high|max (anything else errors); GLM-5.2 accepts
|
||||
// a wider set but z.ai maps low/medium->high and xhigh->max server-side anyway, so
|
||||
// this 3-value mapping matches both.
|
||||
body.reasoning_effort = (zaiLvl === "low" || zaiLvl === "minimal") ? "low"
|
||||
: (zaiLvl === "high" || zaiLvl === "medium") ? "high"
|
||||
: "max";
|
||||
}
|
||||
break;
|
||||
}
|
||||
case "qwen": {
|
||||
|
||||
@@ -151,3 +151,17 @@ export function fixMissingToolResponses(body) {
|
||||
return body;
|
||||
}
|
||||
|
||||
// Default `type: "custom"` on Claude-format tools that arrive without one.
|
||||
// Anthropic's Claude tool schema requires `type` to be explicitly set; strict gateways
|
||||
// (e.g., MiniMax Anthropic-compatible endpoint, error 2013) reject legacy payloads that
|
||||
// omit it with HTTP 400. Tools that already carry a truthy `type` (e.g., `computer_use`,
|
||||
// `bash`, `web_search_20250305`) are passed through untouched.
|
||||
//
|
||||
// Spread order matters: `{ ...tool, type: "custom" }` (spread first, override last)
|
||||
// ensures that falsy `type` values (null, undefined, "") in the original tool don't
|
||||
// overwrite the default. `{ type: "custom", ...tool }` would let `type: null` survive.
|
||||
export function defaultClaudeToolType(tools) {
|
||||
if (!Array.isArray(tools)) return tools;
|
||||
return tools.map(tool => tool?.type ? tool : { ...tool, type: "custom" });
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,18 @@ import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
|
||||
const CACHE_CONTROL_5M = { type: "ephemeral" };
|
||||
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
|
||||
|
||||
// Anthropic rejects a tool carrying BOTH defer_loading:true and cache_control
|
||||
// ("Tools defer_loading cannot use prompt caching", #3567). MCP clients put
|
||||
// deferred tools at the tail, which is exactly where the cache anchor lands.
|
||||
// Anchor on the last tool that CAN be cached instead of dropping caching.
|
||||
export function lastCacheableToolIndex(tools) {
|
||||
if (!Array.isArray(tools)) return -1;
|
||||
for (let i = tools.length - 1; i >= 0; i--) {
|
||||
if (tools[i]?.defer_loading !== true) return i;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Check if message has valid non-empty content
|
||||
export function hasValidContent(msg) {
|
||||
if (typeof msg.content === "string" && msg.content.trim()) return true;
|
||||
@@ -108,11 +120,24 @@ function buildThinkingPlaceholder(provider) {
|
||||
return block;
|
||||
}
|
||||
|
||||
// Anthropic validates server_tool_use ids against this pattern and rejects the
|
||||
// whole request with a 400 when one does not match. A combo that falls back to a
|
||||
// provider with its own built-in tools (z.ai/glm emits OpenAI-style `call_` ids for
|
||||
// its analyze_image tool) leaves such blocks in the history, so every later Claude
|
||||
// turn carries a poisoned id.
|
||||
const CLAUDE_SERVER_TOOL_USE_ID = /^srvtoolu_[a-zA-Z0-9_]+$/;
|
||||
|
||||
function hasForeignServerToolUseId(block) {
|
||||
return block?.type === CLAUDE_BLOCK.SERVER_TOOL_USE
|
||||
&& !CLAUDE_SERVER_TOOL_USE_ID.test(String(block.id ?? ""));
|
||||
}
|
||||
|
||||
// Normalize a native Claude passthrough body to match Anthropic Messages API spec.
|
||||
// Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject:
|
||||
// 1. thinking.type "adaptive" → unsupported on Haiku
|
||||
// 2. output_config.effort → unsupported on Haiku
|
||||
// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed
|
||||
// 4. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright
|
||||
export function normalizeClaudePassthrough(body, model = "") {
|
||||
if (!body || typeof body !== "object") return body;
|
||||
|
||||
@@ -164,6 +189,7 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
|
||||
// so foreign signatures leak into history and Anthropic rejects them).
|
||||
const thinkingEnabled = body.thinking?.type === "enabled";
|
||||
const droppedServerToolUseIds = new Set();
|
||||
if (Array.isArray(body.messages)) {
|
||||
for (const msg of body.messages) {
|
||||
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
|
||||
@@ -178,6 +204,10 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (hasForeignServerToolUseId(block)) {
|
||||
if (block.id != null) droppedServerToolUseIds.add(String(block.id));
|
||||
continue;
|
||||
}
|
||||
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
|
||||
kept.push(block);
|
||||
}
|
||||
@@ -188,6 +218,35 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
}
|
||||
}
|
||||
|
||||
// A dropped server_tool_use leaves its result behind; Anthropic rejects a
|
||||
// tool_result that references an id no block declares, so both halves must go.
|
||||
if (droppedServerToolUseIds.size > 0 && Array.isArray(body.messages)) {
|
||||
for (const msg of body.messages) {
|
||||
if (!Array.isArray(msg.content)) continue;
|
||||
const kept = msg.content.filter(block => !(
|
||||
(block?.type === CLAUDE_BLOCK.TOOL_RESULT || block?.type === CLAUDE_BLOCK.WEB_SEARCH_TOOL_RESULT)
|
||||
&& droppedServerToolUseIds.has(String(block.tool_use_id ?? ""))
|
||||
));
|
||||
if (kept.length !== msg.content.length) {
|
||||
msg.content = kept;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Drop empty text blocks and any message left with no content at all.
|
||||
// Anthropic rejects `messages.N.content` blocks with empty text (400
|
||||
// "text content blocks must be non-empty"); a message whose blocks were all
|
||||
// stripped above must be dropped, not padded with an empty placeholder.
|
||||
if (Array.isArray(body.messages)) {
|
||||
body.messages = body.messages.filter(msg => {
|
||||
if (typeof msg.content === "string") return msg.content.trim().length > 0;
|
||||
if (!Array.isArray(msg.content)) return true;
|
||||
msg.content = msg.content.filter(block =>
|
||||
!(block?.type === CLAUDE_BLOCK.TEXT && !String(block.text ?? "").trim()));
|
||||
return msg.content.length > 0;
|
||||
});
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -223,7 +282,7 @@ export function anchorClaudeCache(body) {
|
||||
}
|
||||
|
||||
if (Array.isArray(body.tools)) {
|
||||
const last = body.tools.length - 1;
|
||||
const last = lastCacheableToolIndex(body.tools);
|
||||
body.tools.forEach((tool, i) => {
|
||||
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
|
||||
else delete tool.cache_control;
|
||||
@@ -252,33 +311,6 @@ export function anchorClaudeCache(body) {
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
|
||||
// so foreign signatures leak into history and Anthropic rejects them).
|
||||
const thinkingEnabled = body.thinking?.type === "enabled";
|
||||
if (Array.isArray(body.messages)) {
|
||||
for (const msg of body.messages) {
|
||||
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
|
||||
let hasToolUse = false;
|
||||
let hasKeptThinking = false;
|
||||
const kept = [];
|
||||
for (const block of msg.content) {
|
||||
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
|
||||
if (isValidClaudeSignature(block.signature)) {
|
||||
hasKeptThinking = true;
|
||||
kept.push(block);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
|
||||
kept.push(block);
|
||||
}
|
||||
msg.content = kept;
|
||||
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
|
||||
msg.content.unshift(buildThinkingPlaceholder("claude"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -444,9 +476,10 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
|
||||
});
|
||||
}
|
||||
|
||||
const lastCacheable = lastCacheableToolIndex(body.tools);
|
||||
body.tools = body.tools.map((tool, i) => {
|
||||
const { cache_control, ...rest } = tool;
|
||||
if (i === body.tools.length - 1) {
|
||||
if (i === lastCacheable) {
|
||||
return { ...rest, cache_control: { type: "ephemeral", ttl: "1h" } };
|
||||
}
|
||||
return rest;
|
||||
|
||||
@@ -14,6 +14,8 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
|
||||
"uniqueItems", "contains",
|
||||
// 2020-12 keywords with no Gemini equivalent
|
||||
"unevaluatedProperties", "unevaluatedItems", "contentSchema",
|
||||
// Tuple-array keywords; converted to items first, leftovers stripped
|
||||
"prefixItems", "additionalItems",
|
||||
// Claude rejects these in VALIDATED mode
|
||||
"default", "examples",
|
||||
// JSON Schema meta keywords
|
||||
@@ -308,6 +310,37 @@ function ensureObjectType(obj) {
|
||||
for (const v of Object.values(obj)) if (v && typeof v === "object") ensureObjectType(v);
|
||||
}
|
||||
|
||||
// Convert prefixItems (tuple validation) to items — Gemini cannot express tuples,
|
||||
// and a type:"array" schema without items is rejected with "missing field"
|
||||
function convertPrefixItems(obj) {
|
||||
if (!obj || typeof obj !== "object") return;
|
||||
|
||||
if (Array.isArray(obj.prefixItems) && obj.prefixItems.length > 0) {
|
||||
const variants = obj.prefixItems.filter(s => s && s.type !== "null");
|
||||
if (!obj.items && variants.length === 1) {
|
||||
obj.items = variants[0];
|
||||
} else if (!obj.items && variants.length > 1) {
|
||||
obj.items = { anyOf: variants };
|
||||
}
|
||||
delete obj.prefixItems;
|
||||
}
|
||||
|
||||
for (const value of Object.values(obj)) {
|
||||
if (value && typeof value === "object") {
|
||||
convertPrefixItems(value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Gemini requires items on every type:"array" schema — fill a permissive placeholder
|
||||
function ensureArrayItems(obj) {
|
||||
if (!obj || typeof obj !== "object") return;
|
||||
if (obj.type === "array" && !obj.items) {
|
||||
obj.items = { type: "string" };
|
||||
}
|
||||
for (const v of Object.values(obj)) if (v && typeof v === "object") ensureArrayItems(v);
|
||||
}
|
||||
|
||||
// Clean JSON Schema for Antigravity API compatibility - removes unsupported keywords recursively
|
||||
export function cleanJSONSchemaForAntigravity(schema) {
|
||||
if (!schema || typeof schema !== "object") return schema;
|
||||
@@ -321,11 +354,13 @@ export function cleanJSONSchemaForAntigravity(schema) {
|
||||
|
||||
// Phase 2: Flatten complex structures
|
||||
mergeAllOf(cleaned);
|
||||
convertPrefixItems(cleaned);
|
||||
flattenAnyOfOneOf(cleaned);
|
||||
flattenTypeArrays(cleaned);
|
||||
|
||||
// Phase 2.5: Infer missing type=object when properties exist (Gemini requirement)
|
||||
ensureObjectType(cleaned);
|
||||
ensureArrayItems(cleaned);
|
||||
|
||||
// Phase 3: Remove all unsupported keywords at ALL levels (including inside arrays)
|
||||
removeUnsupportedKeywords(cleaned, UNSUPPORTED_SCHEMA_CONSTRAINTS);
|
||||
|
||||
@@ -23,6 +23,59 @@ export function normalizeResponsesInput(input) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Strict Responses upstreams reject overlong call_ids with InputValidationError (#393).
|
||||
export const MAX_RESPONSES_CALL_ID_LEN = 64;
|
||||
|
||||
// Fallback ids share one Date.now() when a batch of items is sanitized in a tight
|
||||
// loop — a per-process sequence keeps same-millisecond ids unique so
|
||||
// function_call ↔ function_call_output correlation never collides.
|
||||
let responsesCallIdSeq = 0;
|
||||
|
||||
export function clampResponsesCallId(id) {
|
||||
if (typeof id !== "string" || !id) return `call_${Date.now()}_${(responsesCallIdSeq += 1)}`;
|
||||
return id.length > MAX_RESPONSES_CALL_ID_LEN ? id.substring(0, MAX_RESPONSES_CALL_ID_LEN) : id;
|
||||
}
|
||||
|
||||
// Single-stringify: objects → JSON once; valid JSON strings pass through untouched;
|
||||
// anything else (partial fragments, empty) falls back to "{}" instead of
|
||||
// double-encoding and tripping upstream InputValidationError.
|
||||
export function coerceResponsesArguments(value) {
|
||||
if (value === undefined || value === null || value === "") return "{}";
|
||||
if (typeof value !== "string") {
|
||||
try {
|
||||
return JSON.stringify(value);
|
||||
} catch {
|
||||
return "{}";
|
||||
}
|
||||
}
|
||||
try {
|
||||
JSON.parse(value);
|
||||
return value;
|
||||
} catch {
|
||||
return "{}";
|
||||
}
|
||||
}
|
||||
|
||||
// function_call_output.output must be a string — never null/object.
|
||||
export function coerceResponsesOutput(value) {
|
||||
if (typeof value === "string") return value;
|
||||
if (value === undefined || value === null) return "";
|
||||
if (Array.isArray(value)) {
|
||||
return value.map((c) => {
|
||||
try {
|
||||
return c?.text ?? JSON.stringify(c);
|
||||
} catch {
|
||||
return String(c);
|
||||
}
|
||||
}).join("");
|
||||
}
|
||||
try {
|
||||
return JSON.stringify(value);
|
||||
} catch {
|
||||
return String(value);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert OpenAI Responses API format to standard chat completions format
|
||||
* Responses API uses: { input: [...], instructions: "..." }
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { FORMATS } from "./formats.js";
|
||||
import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js";
|
||||
import { prepareClaudeRequest } from "./formats/claude.js";
|
||||
import { cloakClaudeTools } from "../utils/claudeCloaking.js";
|
||||
import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js";
|
||||
import { filterToOpenAIFormat } from "./formats/openai.js";
|
||||
import { normalizeThinkingConfig } from "../services/provider.js";
|
||||
import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js";
|
||||
@@ -133,7 +133,7 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
|
||||
result = prepareClaudeRequest(result, provider, apiKey, connectionId, credentials?.rawHeaders, clientSessionId);
|
||||
}
|
||||
|
||||
// Claude cloaking: rename client tools with _cc suffix (anti-ban)
|
||||
// Claude cloaking: rename client tools with CLAUDE_TOOL_SUFFIX (anti-ban)
|
||||
// quirk: only providers flagged cloakToolsOnOAuth, and only with an OAuth token
|
||||
if (PROVIDERS[provider]?.quirks?.cloakToolsOnOAuth) {
|
||||
const apiKey = credentials?.accessToken || credentials?.apiKey || null;
|
||||
@@ -161,9 +161,12 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
|
||||
// Translate response chunk: target -> openai -> source
|
||||
export function translateResponse(targetFormat, sourceFormat, chunk, state) {
|
||||
ensureInitialized();
|
||||
// If same format, return as-is
|
||||
// If same format, return as-is — except the tool name may still be cloaked:
|
||||
// translateRequest() suffixes client tools for OAuth-cloaked Claude providers
|
||||
// even when no format conversion is needed, so streamed tool_use blocks must
|
||||
// be decloaked here or the client sees an unknown ("_ide"-suffixed) tool.
|
||||
if (sourceFormat === targetFormat) {
|
||||
return [chunk];
|
||||
return [decloakStreamChunk(chunk, state?.toolNameMap)];
|
||||
}
|
||||
|
||||
let results = [chunk];
|
||||
|
||||
@@ -327,7 +327,6 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
|
||||
};
|
||||
|
||||
if (profileArn) payload.profileArn = profileArn;
|
||||
if (systemPrompt) payload.systemPrompt = systemPrompt;
|
||||
if (additionalModelRequestFields) {
|
||||
payload.additionalModelRequestFields = additionalModelRequestFields;
|
||||
}
|
||||
|
||||
@@ -6,12 +6,15 @@
|
||||
*/
|
||||
import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { normalizeResponsesInput } from "../formats/responsesApi.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../formats/responsesApi.js";
|
||||
import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM } from "../schema/index.js";
|
||||
|
||||
// Responses API enforces max 64 chars on call_id (#393)
|
||||
const MAX_CALL_ID_LEN = 64;
|
||||
const clampCallId = (id) => (typeof id === "string" && id.length > MAX_CALL_ID_LEN ? id.substring(0, MAX_CALL_ID_LEN) : id);
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
|
||||
/**
|
||||
* Convert OpenAI Responses API request to OpenAI Chat Completions format
|
||||
@@ -249,6 +252,23 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract plain text from a system/developer message for Responses instructions.
|
||||
* Array content (text parts) is joined; anything else falls back to "" rather
|
||||
* than leaking "[object Object]" upstream.
|
||||
*/
|
||||
function extractInstructionsText(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (Array.isArray(content)) {
|
||||
return content.map((c) => {
|
||||
if (typeof c?.text === "string") return c.text;
|
||||
if (typeof c?.content === "string") return c.content;
|
||||
return "";
|
||||
}).filter(Boolean).join("\n");
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
/**
|
||||
* Ensure object schema always has properties field (required by Codex Responses API)
|
||||
*/
|
||||
@@ -300,7 +320,16 @@ function buildReasoningInputItem(msg) {
|
||||
*/
|
||||
export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) {
|
||||
// Body already in Responses API format (e.g. Cursor CLI calling /chat/completions with input[])
|
||||
if (body.input) return { ...body, model, stream: true };
|
||||
if (body.input) {
|
||||
const out = { ...body, model, stream: true };
|
||||
if (out.max_output_tokens === undefined) {
|
||||
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
|
||||
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
|
||||
}
|
||||
delete out.max_tokens;
|
||||
delete out.max_completion_tokens;
|
||||
return out;
|
||||
}
|
||||
|
||||
const result = {
|
||||
model,
|
||||
@@ -318,7 +347,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
|
||||
// Use the first instruction-bearing message as instructions.
|
||||
// OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt.
|
||||
if (!hasSystemMessage) {
|
||||
result.instructions = typeof msg.content === "string" ? msg.content : "";
|
||||
result.instructions = extractInstructionsText(msg.content);
|
||||
hasSystemMessage = true;
|
||||
}
|
||||
continue; // Skip instruction messages in input
|
||||
@@ -369,26 +398,24 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
|
||||
// Convert tool calls
|
||||
if (msg.role === ROLE.ASSISTANT && msg.tool_calls) {
|
||||
for (const tc of msg.tool_calls) {
|
||||
// Skip nameless calls — strict Responses upstreams reject them (#444)
|
||||
const name = typeof tc.function?.name === "string" ? tc.function.name.trim() : "";
|
||||
if (!name) continue;
|
||||
result.input.push({
|
||||
type: RESPONSES_ITEM.FUNCTION_CALL,
|
||||
call_id: clampCallId(tc.id),
|
||||
name: tc.function?.name || "_unknown",
|
||||
arguments: tc.function?.arguments || "{}"
|
||||
call_id: clampResponsesCallId(tc.id),
|
||||
name: name.slice(0, MAX_TOOL_NAME_LEN),
|
||||
arguments: coerceResponsesArguments(tc.function?.arguments)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Convert tool results - output must be a string for Responses API
|
||||
if (msg.role === ROLE.TOOL) {
|
||||
const output = typeof msg.content === "string"
|
||||
? msg.content
|
||||
: Array.isArray(msg.content)
|
||||
? msg.content.map(c => c.text || JSON.stringify(c)).join("")
|
||||
: JSON.stringify(msg.content);
|
||||
result.input.push({
|
||||
type: RESPONSES_ITEM.FUNCTION_CALL_OUTPUT,
|
||||
call_id: clampCallId(msg.tool_call_id),
|
||||
output
|
||||
call_id: clampResponsesCallId(msg.tool_call_id),
|
||||
output: coerceResponsesOutput(msg.content)
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -402,21 +429,30 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
|
||||
if (body.tools && Array.isArray(body.tools)) {
|
||||
result.tools = body.tools.map(tool => {
|
||||
if (tool.type === OPENAI_BLOCK.FUNCTION) {
|
||||
// Strict upstreams reject nameless/overlong tool declarations
|
||||
const name = typeof tool.function?.name === "string" ? tool.function.name.trim() : "";
|
||||
if (!name) return null;
|
||||
return {
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
name: tool.function.name,
|
||||
name: name.slice(0, MAX_TOOL_NAME_LEN),
|
||||
description: String(tool.function.description || ""),
|
||||
parameters: normalizeToolParameters(tool.function.parameters),
|
||||
strict: tool.function.strict
|
||||
};
|
||||
}
|
||||
return tool;
|
||||
});
|
||||
}).filter(Boolean);
|
||||
}
|
||||
|
||||
// Pass through other relevant fields
|
||||
if (body.temperature !== undefined) result.temperature = body.temperature;
|
||||
if (body.max_tokens !== undefined) result.max_tokens = body.max_tokens;
|
||||
if (body.max_output_tokens !== undefined) {
|
||||
result.max_output_tokens = body.max_output_tokens;
|
||||
} else if (body.max_completion_tokens !== undefined) {
|
||||
result.max_output_tokens = body.max_completion_tokens;
|
||||
} else if (body.max_tokens !== undefined) {
|
||||
result.max_output_tokens = body.max_tokens;
|
||||
}
|
||||
if (body.top_p !== undefined) result.top_p = body.top_p;
|
||||
if (body.reasoning !== undefined) result.reasoning = body.reasoning;
|
||||
if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" };
|
||||
|
||||
@@ -13,160 +13,204 @@ import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { randomUUID } from "crypto";
|
||||
import { ROLE, OPENAI_BLOCK } from "../schema/index.js";
|
||||
import { DEFAULT_IMAGE_MIME } from "../schema/index.js";
|
||||
import { parseDataUri } from "../concerns/image.js";
|
||||
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
|
||||
|
||||
function flattenText(content) {
|
||||
if (content == null) return "";
|
||||
if (typeof content === "string") return content;
|
||||
if (Array.isArray(content)) {
|
||||
const parts = [];
|
||||
for (const p of content) {
|
||||
if (typeof p === "string") parts.push(p);
|
||||
else if (p && typeof p === "object" && typeof p.text === "string") parts.push(p.text);
|
||||
}
|
||||
return parts.join("\n");
|
||||
}
|
||||
return String(content);
|
||||
if (content == null) return "";
|
||||
if (typeof content === "string") return content;
|
||||
if (Array.isArray(content)) {
|
||||
const parts = [];
|
||||
for (const p of content) {
|
||||
if (typeof p === "string") parts.push(p);
|
||||
else if (p && typeof p === "object" && typeof p.text === "string")
|
||||
parts.push(p.text);
|
||||
}
|
||||
return parts.join("\n");
|
||||
}
|
||||
return String(content);
|
||||
}
|
||||
|
||||
function toContentBlocks(content) {
|
||||
if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }];
|
||||
if (typeof content === "string") return [{ type: OPENAI_BLOCK.TEXT, text: content }];
|
||||
if (Array.isArray(content)) {
|
||||
const blocks = [];
|
||||
for (const part of content) {
|
||||
if (typeof part === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part });
|
||||
} else if (part && typeof part === "object") {
|
||||
if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
} else if (part.type === OPENAI_BLOCK.IMAGE_URL || part.type === OPENAI_BLOCK.IMAGE) {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: "[image omitted]" });
|
||||
} else if (typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
}
|
||||
}
|
||||
}
|
||||
return blocks.length ? blocks : [{ type: OPENAI_BLOCK.TEXT, text: "" }];
|
||||
}
|
||||
return [{ type: OPENAI_BLOCK.TEXT, text: String(content) }];
|
||||
if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }];
|
||||
if (typeof content === "string")
|
||||
return [{ type: OPENAI_BLOCK.TEXT, text: content }];
|
||||
if (Array.isArray(content)) {
|
||||
const blocks = [];
|
||||
for (const part of content) {
|
||||
if (typeof part === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part });
|
||||
} else if (part && typeof part === "object") {
|
||||
if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
} else if (
|
||||
part.type === OPENAI_BLOCK.IMAGE_URL ||
|
||||
part.type === OPENAI_BLOCK.IMAGE
|
||||
) {
|
||||
// CommandCode `/alpha/generate` accepts {type:"image", image:"<data URI | url>"} —
|
||||
// same shape the official command-code CLI sends (verified from CLI source).
|
||||
const src = part.source;
|
||||
let raw = part.image_url?.url || src?.data || src?.url || "";
|
||||
let parsed = parseDataUri(raw);
|
||||
if (!parsed && src?.type === "base64" && src?.data) {
|
||||
// Claude-style base64 source without a data-URI prefix → wrap it.
|
||||
raw = `data:${src.media_type || DEFAULT_IMAGE_MIME};base64,${src.data}`;
|
||||
parsed = parseDataUri(raw);
|
||||
}
|
||||
if (parsed) {
|
||||
blocks.push({
|
||||
type: "image",
|
||||
image: `data:${parsed.mimeType};base64,${parsed.base64}`,
|
||||
});
|
||||
} else if (raw) {
|
||||
blocks.push({ type: "image", image: raw });
|
||||
}
|
||||
} else if (typeof part.text === "string") {
|
||||
blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text });
|
||||
}
|
||||
}
|
||||
}
|
||||
return blocks.length ? blocks : [{ type: OPENAI_BLOCK.TEXT, text: "" }];
|
||||
}
|
||||
return [{ type: OPENAI_BLOCK.TEXT, text: String(content) }];
|
||||
}
|
||||
|
||||
function safeParseJson(s) {
|
||||
if (s == null) return {};
|
||||
if (typeof s !== "string") return s;
|
||||
try { return JSON.parse(s); } catch { return {}; }
|
||||
if (s == null) return {};
|
||||
if (typeof s !== "string") return s;
|
||||
try {
|
||||
return JSON.parse(s);
|
||||
} catch {
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
function convertMessages(messages = []) {
|
||||
const out = [];
|
||||
const systemTexts = [];
|
||||
const out = [];
|
||||
const systemTexts = [];
|
||||
|
||||
for (const m of messages) {
|
||||
if (!m) continue;
|
||||
const role = m.role;
|
||||
for (const m of messages) {
|
||||
if (!m) continue;
|
||||
const role = m.role;
|
||||
|
||||
if (role === ROLE.SYSTEM) {
|
||||
const t = flattenText(m.content);
|
||||
if (t) systemTexts.push(t);
|
||||
continue;
|
||||
}
|
||||
if (role === ROLE.SYSTEM) {
|
||||
const t = flattenText(m.content);
|
||||
if (t) systemTexts.push(t);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (role === ROLE.TOOL) {
|
||||
const value = typeof m.content === "string" ? m.content : flattenText(m.content);
|
||||
out.push({
|
||||
role: ROLE.TOOL,
|
||||
content: [{
|
||||
type: "tool-result",
|
||||
toolCallId: m.tool_call_id || "",
|
||||
toolName: m.name || "",
|
||||
output: { type: "text", value },
|
||||
}],
|
||||
});
|
||||
continue;
|
||||
}
|
||||
if (role === ROLE.TOOL) {
|
||||
const value =
|
||||
typeof m.content === "string" ? m.content : flattenText(m.content);
|
||||
out.push({
|
||||
role: ROLE.TOOL,
|
||||
content: [
|
||||
{
|
||||
type: "tool-result",
|
||||
toolCallId: m.tool_call_id || "",
|
||||
toolName: m.name || "",
|
||||
output: { type: "text", value },
|
||||
},
|
||||
],
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
if (role === ROLE.ASSISTANT) {
|
||||
const blocks = [];
|
||||
const text = flattenText(m.content);
|
||||
if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
|
||||
if (Array.isArray(m.tool_calls)) {
|
||||
for (const tc of m.tool_calls) {
|
||||
const fn = tc.function || {};
|
||||
blocks.push({
|
||||
type: "tool-call",
|
||||
toolCallId: tc.id || "",
|
||||
toolName: fn.name || "",
|
||||
input: safeParseJson(fn.arguments),
|
||||
});
|
||||
}
|
||||
}
|
||||
out.push({ role: ROLE.ASSISTANT, content: blocks.length ? blocks : [{ type: OPENAI_BLOCK.TEXT, text: "" }] });
|
||||
continue;
|
||||
}
|
||||
if (role === ROLE.ASSISTANT) {
|
||||
const blocks = [];
|
||||
const text = flattenText(m.content);
|
||||
if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
|
||||
if (Array.isArray(m.tool_calls)) {
|
||||
for (const tc of m.tool_calls) {
|
||||
const fn = tc.function || {};
|
||||
blocks.push({
|
||||
type: "tool-call",
|
||||
toolCallId: tc.id || "",
|
||||
toolName: fn.name || "",
|
||||
input: safeParseJson(fn.arguments),
|
||||
});
|
||||
}
|
||||
}
|
||||
out.push({
|
||||
role: ROLE.ASSISTANT,
|
||||
content: blocks.length
|
||||
? blocks
|
||||
: [{ type: OPENAI_BLOCK.TEXT, text: "" }],
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
out.push({ role: ROLE.USER, content: toContentBlocks(m.content) });
|
||||
}
|
||||
out.push({ role: ROLE.USER, content: toContentBlocks(m.content) });
|
||||
}
|
||||
|
||||
return { messages: out, system: systemTexts.join("\n\n") };
|
||||
return { messages: out, system: systemTexts.join("\n\n") };
|
||||
}
|
||||
|
||||
function convertTools(tools) {
|
||||
if (!Array.isArray(tools) || tools.length === 0) return undefined;
|
||||
const result = [];
|
||||
for (const t of tools) {
|
||||
if (!t) continue;
|
||||
if (t.type === OPENAI_BLOCK.FUNCTION && t.function) {
|
||||
result.push({
|
||||
name: t.function.name,
|
||||
description: t.function.description,
|
||||
input_schema: t.function.parameters || { type: "object" },
|
||||
});
|
||||
} else if (t.name && (t.input_schema || t.parameters)) {
|
||||
result.push({
|
||||
name: t.name,
|
||||
description: t.description,
|
||||
input_schema: t.input_schema || t.parameters,
|
||||
});
|
||||
}
|
||||
}
|
||||
return result.length ? result : undefined;
|
||||
if (!Array.isArray(tools) || tools.length === 0) return undefined;
|
||||
const result = [];
|
||||
for (const t of tools) {
|
||||
if (!t) continue;
|
||||
if (t.type === OPENAI_BLOCK.FUNCTION && t.function) {
|
||||
result.push({
|
||||
name: t.function.name,
|
||||
description: t.function.description,
|
||||
input_schema: t.function.parameters || { type: "object" },
|
||||
});
|
||||
} else if (t.name && (t.input_schema || t.parameters)) {
|
||||
result.push({
|
||||
name: t.name,
|
||||
description: t.description,
|
||||
input_schema: t.input_schema || t.parameters,
|
||||
});
|
||||
}
|
||||
}
|
||||
return result.length ? result : undefined;
|
||||
}
|
||||
|
||||
export function openaiToCommandCodeRequest(model, body, stream /* , credentials */) {
|
||||
const { messages, system } = convertMessages(body.messages);
|
||||
const params = {
|
||||
model,
|
||||
messages,
|
||||
stream: stream !== false,
|
||||
max_tokens: body.max_tokens ?? body.max_output_tokens ?? DEFAULT_MAX_TOKENS,
|
||||
temperature: body.temperature ?? 0.3,
|
||||
};
|
||||
export function openaiToCommandCodeRequest(
|
||||
model,
|
||||
body,
|
||||
stream /* , credentials */,
|
||||
) {
|
||||
const { messages, system } = convertMessages(body.messages);
|
||||
const params = {
|
||||
model,
|
||||
messages,
|
||||
stream: stream !== false,
|
||||
max_tokens: body.max_tokens ?? body.max_output_tokens ?? DEFAULT_MAX_TOKENS,
|
||||
temperature: body.temperature ?? 0.3,
|
||||
};
|
||||
|
||||
if (system) params.system = system;
|
||||
if (system) params.system = system;
|
||||
|
||||
const tools = convertTools(body.tools);
|
||||
if (tools) params.tools = tools;
|
||||
if (body.top_p != null) params.top_p = body.top_p;
|
||||
const tools = convertTools(body.tools);
|
||||
if (tools) params.tools = tools;
|
||||
if (body.top_p != null) params.top_p = body.top_p;
|
||||
|
||||
const today = new Date().toISOString().slice(0, 10);
|
||||
const today = new Date().toISOString().slice(0, 10);
|
||||
|
||||
return {
|
||||
threadId: randomUUID(),
|
||||
memory: "",
|
||||
config: {
|
||||
workingDir: process.cwd(),
|
||||
date: today,
|
||||
environment: process.platform,
|
||||
structure: [],
|
||||
isGitRepo: false,
|
||||
currentBranch: "",
|
||||
mainBranch: "",
|
||||
gitStatus: "",
|
||||
recentCommits: [],
|
||||
},
|
||||
params,
|
||||
};
|
||||
// environment format mirrors the official command-code CLI (getEnvironmentInfo):
|
||||
// `${platform}-${arch}, Node.js ${version}` (e.g. "darwin-arm64, Node.js v24.16.0").
|
||||
const environment = `${process.platform}-${process.arch}, Node.js ${process.version}`;
|
||||
|
||||
return {
|
||||
threadId: randomUUID(),
|
||||
memory: "",
|
||||
config: {
|
||||
workingDir: process.cwd(),
|
||||
date: today,
|
||||
environment,
|
||||
structure: [],
|
||||
isGitRepo: false,
|
||||
currentBranch: "",
|
||||
mainBranch: "",
|
||||
gitStatus: "",
|
||||
recentCommits: [],
|
||||
},
|
||||
params,
|
||||
};
|
||||
}
|
||||
|
||||
register(FORMATS.OPENAI, FORMATS.COMMANDCODE, openaiToCommandCodeRequest, null);
|
||||
|
||||
@@ -2,6 +2,7 @@ import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js";
|
||||
import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js";
|
||||
import { getGeminiThoughtSignatureSync } from "../../services/thoughtSignatureStore.js";
|
||||
function generateUUID() {
|
||||
return crypto.randomUUID();
|
||||
}
|
||||
@@ -46,7 +47,7 @@ function normalizeGeminiContents(contents) {
|
||||
}
|
||||
|
||||
// Core: Convert OpenAI request to Gemini format (base for all variants)
|
||||
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) {
|
||||
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) {
|
||||
const result = {
|
||||
model: model,
|
||||
contents: [],
|
||||
@@ -133,18 +134,27 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
|
||||
|
||||
if (msg.tool_calls && Array.isArray(msg.tool_calls)) {
|
||||
const toolCallIds = [];
|
||||
let firstFunctionCallSeen = false;
|
||||
for (const tc of msg.tool_calls) {
|
||||
if (tc.type !== OPENAI_BLOCK.FUNCTION) continue;
|
||||
|
||||
const args = tryParseJSON(tc.function?.arguments || "{}");
|
||||
parts.push({
|
||||
thoughtSignature: signature,
|
||||
const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId) : null;
|
||||
// First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig
|
||||
const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined);
|
||||
firstFunctionCallSeen = true;
|
||||
|
||||
const part = {
|
||||
functionCall: {
|
||||
id: tc.id,
|
||||
name: sanitizeGeminiFunctionName(tc.function.name),
|
||||
args: args
|
||||
}
|
||||
});
|
||||
};
|
||||
if (callSig) {
|
||||
part.thoughtSignature = callSig;
|
||||
}
|
||||
parts.push(part);
|
||||
toolCallIds.push(tc.id);
|
||||
}
|
||||
|
||||
@@ -232,13 +242,13 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
|
||||
}
|
||||
|
||||
// OpenAI -> Gemini (standard API)
|
||||
export function openaiToGeminiRequest(model, body, stream) {
|
||||
return openaiToGeminiBase(model, body, stream);
|
||||
export function openaiToGeminiRequest(model, body, stream, credentials = null) {
|
||||
return openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_AG_SIGNATURE, credentials?._clientSessionId);
|
||||
}
|
||||
|
||||
// OpenAI -> Gemini CLI (Cloud Code Assist)
|
||||
export function openaiToGeminiCLIRequest(model, body, stream) {
|
||||
const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE);
|
||||
export function openaiToGeminiCLIRequest(model, body, stream, credentials = null) {
|
||||
const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE, credentials?._clientSessionId);
|
||||
// Thinking is normalized centrally by applyThinking (thinkingUnified.js) after translation.
|
||||
|
||||
// Clean schema for tools
|
||||
@@ -335,18 +345,26 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
|
||||
const parts = [];
|
||||
|
||||
if (Array.isArray(msg.content)) {
|
||||
let firstToolUseSeen = false;
|
||||
for (const block of msg.content) {
|
||||
if (block.type === CLAUDE_BLOCK.TEXT) {
|
||||
parts.push({ text: block.text });
|
||||
} else if (block.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||
parts.push({
|
||||
thoughtSignature: signature,
|
||||
const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId) : null;
|
||||
const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined);
|
||||
firstToolUseSeen = true;
|
||||
|
||||
const part = {
|
||||
functionCall: {
|
||||
id: block.id,
|
||||
name: sanitizeGeminiFunctionName(block.name),
|
||||
args: block.input || {}
|
||||
}
|
||||
});
|
||||
};
|
||||
if (callSig) {
|
||||
part.thoughtSignature = callSig;
|
||||
}
|
||||
parts.push(part);
|
||||
} else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) {
|
||||
let content = block.content;
|
||||
if (Array.isArray(content)) {
|
||||
|
||||
@@ -420,7 +420,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
|
||||
if (profileArn) {
|
||||
payload.profileArn = profileArn;
|
||||
}
|
||||
if (systemPrompt) payload.systemPrompt = systemPrompt;
|
||||
if (additionalModelRequestFields) {
|
||||
payload.additionalModelRequestFields = additionalModelRequestFields;
|
||||
}
|
||||
|
||||
@@ -25,160 +25,198 @@ import { fallbackToolCallId } from "../concerns/toolCall.js";
|
||||
import { toOpenAIFinish } from "../concerns/finishReason.js";
|
||||
|
||||
function ensureState(state, model) {
|
||||
if (!state.responseId) {
|
||||
state.responseId = `chatcmpl-${Date.now()}`;
|
||||
state.created = Math.floor(Date.now() / 1000);
|
||||
state.model = state.model || model || "commandcode";
|
||||
state.chunkIndex = 0;
|
||||
state.toolIndex = 0;
|
||||
state.toolIndexById = new Map();
|
||||
state.openTools = new Set();
|
||||
state.openText = false;
|
||||
state.finishReason = null;
|
||||
state.usage = null;
|
||||
}
|
||||
if (!state.responseId) {
|
||||
state.responseId = `chatcmpl-${Date.now()}`;
|
||||
state.created = Math.floor(Date.now() / 1000);
|
||||
state.model = state.model || model || "commandcode";
|
||||
state.chunkIndex = 0;
|
||||
state.toolIndex = 0;
|
||||
state.toolIndexById = new Map();
|
||||
state.openTools = new Set();
|
||||
state.openText = false;
|
||||
state.finishReason = null;
|
||||
state.usage = null;
|
||||
}
|
||||
}
|
||||
|
||||
function makeChunk(state, delta, finishReason = null) {
|
||||
return buildChunk(
|
||||
{ id: state.responseId, created: state.created, model: state.model },
|
||||
delta,
|
||||
finishReason
|
||||
);
|
||||
return buildChunk(
|
||||
{ id: state.responseId, created: state.created, model: state.model },
|
||||
delta,
|
||||
finishReason,
|
||||
);
|
||||
}
|
||||
|
||||
const mapFinishReason = (reason) => toOpenAIFinish(reason, "commandcode");
|
||||
|
||||
export function commandCodeToOpenAIResponse(chunk, state) {
|
||||
if (!chunk) return null;
|
||||
if (!chunk) return null;
|
||||
|
||||
// Already-OpenAI chunk: pass through
|
||||
if (chunk && typeof chunk === "object" && chunk.object === "chat.completion.chunk") {
|
||||
return chunk;
|
||||
}
|
||||
// Already-OpenAI chunk: pass through
|
||||
if (
|
||||
chunk &&
|
||||
typeof chunk === "object" &&
|
||||
chunk.object === "chat.completion.chunk"
|
||||
) {
|
||||
return chunk;
|
||||
}
|
||||
|
||||
// Parse string lines coming out of upstream
|
||||
let event = chunk;
|
||||
if (typeof chunk === "string") {
|
||||
const line = chunk.trim();
|
||||
if (!line) return null;
|
||||
// Tolerate raw "data: {...}" framing if the upstream wrapper inserts it
|
||||
const json = line.startsWith("data:") ? line.slice(5).trim() : line;
|
||||
if (!json || json === "[DONE]") return null;
|
||||
try {
|
||||
event = JSON.parse(json);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
// Parse string lines coming out of upstream
|
||||
let event = chunk;
|
||||
if (typeof chunk === "string") {
|
||||
const line = chunk.trim();
|
||||
if (!line) return null;
|
||||
// Tolerate raw "data: {...}" framing if the upstream wrapper inserts it
|
||||
const json = line.startsWith("data:") ? line.slice(5).trim() : line;
|
||||
if (!json || json === "[DONE]") return null;
|
||||
try {
|
||||
event = JSON.parse(json);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
if (!event || typeof event !== "object" || !event.type) return null;
|
||||
if (!event || typeof event !== "object" || !event.type) return null;
|
||||
|
||||
ensureState(state, event.model);
|
||||
const out = [];
|
||||
ensureState(state, event.model);
|
||||
const out = [];
|
||||
|
||||
switch (event.type) {
|
||||
case "text-delta": {
|
||||
const text = event.text || event.delta || "";
|
||||
if (!text) break;
|
||||
const delta = state.chunkIndex === 0 ? { role: ROLE.ASSISTANT, content: text } : { content: text };
|
||||
state.chunkIndex++;
|
||||
state.openText = true;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "reasoning-delta": {
|
||||
const text = event.text || "";
|
||||
if (!text) break;
|
||||
// Map reasoning to OpenAI "reasoning_content" field (used by deepseek-reasoner-style clients).
|
||||
const delta = reasoningDelta(text, state.chunkIndex === 0);
|
||||
state.chunkIndex++;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "tool-input-start": {
|
||||
const id = event.id || event.toolCallId || fallbackToolCallId(state.toolIndex);
|
||||
let idx = state.toolIndexById.get(id);
|
||||
if (idx == null) {
|
||||
idx = state.toolIndex++;
|
||||
state.toolIndexById.set(id, idx);
|
||||
}
|
||||
state.openTools.add(id);
|
||||
const delta = {
|
||||
...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}),
|
||||
tool_calls: [{
|
||||
index: idx,
|
||||
id,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: event.toolName || "", arguments: "" },
|
||||
}],
|
||||
};
|
||||
state.chunkIndex++;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "tool-input-delta": {
|
||||
const id = event.id || event.toolCallId;
|
||||
const idx = state.toolIndexById.get(id);
|
||||
if (idx == null) break;
|
||||
const delta = {
|
||||
tool_calls: [{
|
||||
index: idx,
|
||||
function: { arguments: event.delta || event.inputTextDelta || "" },
|
||||
}],
|
||||
};
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "tool-call": {
|
||||
// Final consolidated tool call — only emit if we never saw tool-input-* deltas.
|
||||
const id = event.toolCallId;
|
||||
if (state.toolIndexById.has(id)) break;
|
||||
const idx = state.toolIndex++;
|
||||
state.toolIndexById.set(id, idx);
|
||||
const argsStr = typeof event.input === "string" ? event.input : JSON.stringify(event.input ?? {});
|
||||
const delta = {
|
||||
...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}),
|
||||
tool_calls: [{
|
||||
index: idx,
|
||||
id,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: event.toolName || "", arguments: argsStr },
|
||||
}],
|
||||
};
|
||||
state.chunkIndex++;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "finish-step": {
|
||||
state.finishReason = mapFinishReason(event.finishReason);
|
||||
if (event.usage) state.usage = event.usage;
|
||||
break;
|
||||
}
|
||||
case "finish": {
|
||||
const finishReason = state.finishReason || mapFinishReason(event.finishReason || "stop");
|
||||
const finalChunk = makeChunk(state, {}, finishReason);
|
||||
const totalUsage = event.totalUsage || state.usage;
|
||||
const usage = toOpenAIUsage(totalUsage, "commandcode");
|
||||
if (usage) finalChunk.usage = usage;
|
||||
out.push(finalChunk);
|
||||
break;
|
||||
}
|
||||
case "error": {
|
||||
state.finishReason = OPENAI_FINISH.STOP;
|
||||
const errVal = event.error ?? event.message ?? "unknown";
|
||||
const errStr = typeof errVal === "string" ? errVal : JSON.stringify(errVal);
|
||||
out.push(makeChunk(state, { content: `\n\n[CommandCode error: ${errStr}]` }));
|
||||
out.push(makeChunk(state, {}, OPENAI_FINISH.STOP));
|
||||
break;
|
||||
}
|
||||
// Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end,
|
||||
// provider-metadata, message-metadata, etc. They carry no client-visible content.
|
||||
default:
|
||||
break;
|
||||
}
|
||||
switch (event.type) {
|
||||
case "text-delta": {
|
||||
const text = event.text || event.delta || "";
|
||||
if (!text) break;
|
||||
const delta =
|
||||
state.chunkIndex === 0
|
||||
? { role: ROLE.ASSISTANT, content: text }
|
||||
: { content: text };
|
||||
state.chunkIndex++;
|
||||
state.openText = true;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "reasoning-delta": {
|
||||
const text = event.text || "";
|
||||
if (!text) break;
|
||||
// Map reasoning to OpenAI "reasoning_content" field (used by deepseek-reasoner-style clients).
|
||||
const delta = reasoningDelta(text, state.chunkIndex === 0);
|
||||
state.chunkIndex++;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "tool-input-start": {
|
||||
const id =
|
||||
event.id || event.toolCallId || fallbackToolCallId(state.toolIndex);
|
||||
let idx = state.toolIndexById.get(id);
|
||||
if (idx == null) {
|
||||
idx = state.toolIndex++;
|
||||
state.toolIndexById.set(id, idx);
|
||||
}
|
||||
state.openTools.add(id);
|
||||
const delta = {
|
||||
...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}),
|
||||
tool_calls: [
|
||||
{
|
||||
index: idx,
|
||||
id,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: event.toolName || "", arguments: "" },
|
||||
},
|
||||
],
|
||||
};
|
||||
state.chunkIndex++;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "tool-input-delta": {
|
||||
const id = event.id || event.toolCallId;
|
||||
const idx = state.toolIndexById.get(id);
|
||||
if (idx == null) break;
|
||||
const delta = {
|
||||
tool_calls: [
|
||||
{
|
||||
index: idx,
|
||||
function: { arguments: event.delta || event.inputTextDelta || "" },
|
||||
},
|
||||
],
|
||||
};
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "tool-call": {
|
||||
// Final consolidated tool call — only emit if we never saw tool-input-* deltas.
|
||||
const id = event.toolCallId;
|
||||
if (state.toolIndexById.has(id)) break;
|
||||
const idx = state.toolIndex++;
|
||||
state.toolIndexById.set(id, idx);
|
||||
const argsStr =
|
||||
typeof event.input === "string"
|
||||
? event.input
|
||||
: JSON.stringify(event.input ?? {});
|
||||
const delta = {
|
||||
...(state.chunkIndex === 0 ? { role: ROLE.ASSISTANT } : {}),
|
||||
tool_calls: [
|
||||
{
|
||||
index: idx,
|
||||
id,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: event.toolName || "", arguments: argsStr },
|
||||
},
|
||||
],
|
||||
};
|
||||
state.chunkIndex++;
|
||||
out.push(makeChunk(state, delta));
|
||||
break;
|
||||
}
|
||||
case "finish-step": {
|
||||
state.finishReason = mapFinishReason(event.finishReason);
|
||||
if (event.usage) state.usage = event.usage;
|
||||
break;
|
||||
}
|
||||
case "finish": {
|
||||
const finishReason =
|
||||
state.finishReason || mapFinishReason(event.finishReason || "stop");
|
||||
const finalChunk = makeChunk(state, {}, finishReason);
|
||||
const totalUsage = event.totalUsage || state.usage;
|
||||
const usage = toOpenAIUsage(totalUsage, "commandcode");
|
||||
if (usage) finalChunk.usage = usage;
|
||||
out.push(finalChunk);
|
||||
break;
|
||||
}
|
||||
case "error": {
|
||||
// Terminal upstream failure (AI SDK v5 error event) — NOT content. Emit an
|
||||
// OpenAI-shaped error chunk (chunk.error) so downstream — parseSSEToOpenAIResponse
|
||||
// for non-streaming, OpenAI SDK clients for streaming — treats the request as
|
||||
// failed instead of surfacing fake success content like "[CommandCode error: ...]".
|
||||
state.finishReason = OPENAI_FINISH.STOP;
|
||||
const errVal = event.error ?? event.message ?? "unknown";
|
||||
const errStr =
|
||||
typeof errVal === "string"
|
||||
? errVal
|
||||
: typeof errVal?.message === "string"
|
||||
? errVal.message
|
||||
: JSON.stringify(errVal);
|
||||
const errType =
|
||||
typeof errVal === "string"
|
||||
? "upstream_error"
|
||||
: errVal?.type || "upstream_error";
|
||||
const errChunk = makeChunk(state, {});
|
||||
errChunk.error = { message: errStr, type: errType };
|
||||
out.push(errChunk);
|
||||
out.push(makeChunk(state, {}, OPENAI_FINISH.STOP));
|
||||
break;
|
||||
}
|
||||
// Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end,
|
||||
// provider-metadata, message-metadata, etc. They carry no client-visible content.
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return out.length ? out : null;
|
||||
return out.length ? out : null;
|
||||
}
|
||||
|
||||
register(FORMATS.COMMANDCODE, FORMATS.OPENAI, null, commandCodeToOpenAIResponse);
|
||||
register(
|
||||
FORMATS.COMMANDCODE,
|
||||
FORMATS.OPENAI,
|
||||
null,
|
||||
commandCodeToOpenAIResponse,
|
||||
);
|
||||
|
||||
@@ -6,6 +6,7 @@ import { toOpenAIUsage } from "../concerns/usage.js";
|
||||
import { reasoningDelta } from "../concerns/reasoning.js";
|
||||
import { encodeDataUri } from "../concerns/image.js";
|
||||
import { toOpenAIFinish } from "../concerns/finishReason.js";
|
||||
import { storeGeminiThoughtSignature } from "../../services/thoughtSignatureStore.js";
|
||||
|
||||
// Build chunk meta for current gemini state
|
||||
function chunkMeta(state) {
|
||||
@@ -13,14 +14,18 @@ function chunkMeta(state) {
|
||||
}
|
||||
|
||||
// Build a tool_call chunk from a gemini functionCall part (shared by sig/non-sig branches)
|
||||
function emitFunctionCall(functionCall, state) {
|
||||
function emitFunctionCall(functionCall, state, signature = null) {
|
||||
const rawName = functionCall.name;
|
||||
// Restore original tool name from mapping (AG cloaking)
|
||||
const fcName = state.toolNameMap?.get(rawName) || rawName;
|
||||
const fcArgs = functionCall.args || {};
|
||||
const toolCallIndex = state.functionIndex++;
|
||||
const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`;
|
||||
if (signature) {
|
||||
storeGeminiThoughtSignature(callId, signature, state.sessionId);
|
||||
}
|
||||
const toolCall = {
|
||||
id: `${fcName}-${Date.now()}-${toolCallIndex}`,
|
||||
id: callId,
|
||||
index: toolCallIndex,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: fcName, arguments: JSON.stringify(fcArgs) },
|
||||
@@ -57,13 +62,21 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
if (content?.parts) {
|
||||
for (const part of content.parts) {
|
||||
const hasThoughtSig = part.thoughtSignature || part.thought_signature;
|
||||
if (hasThoughtSig && typeof hasThoughtSig === "string") {
|
||||
state.pendingThoughtSignature = hasThoughtSig;
|
||||
}
|
||||
const isThought = part.thought === true;
|
||||
|
||||
|
||||
// Handle thought signature (thinking mode)
|
||||
if (hasThoughtSig) {
|
||||
const hasTextContent = part.text !== undefined && part.text !== "";
|
||||
const hasFunctionCall = !!part.functionCall;
|
||||
|
||||
|
||||
// Standalone thoughtSignature part (no text, no functionCall): keep pending for next functionCall
|
||||
if (!hasTextContent && !hasFunctionCall) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (hasTextContent) {
|
||||
results.push(buildChunk(
|
||||
chunkMeta(state),
|
||||
@@ -71,9 +84,10 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
null
|
||||
));
|
||||
}
|
||||
|
||||
|
||||
if (hasFunctionCall) {
|
||||
results.push(emitFunctionCall(part.functionCall, state));
|
||||
results.push(emitFunctionCall(part.functionCall, state, hasThoughtSig));
|
||||
state.pendingThoughtSignature = null;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
@@ -92,7 +106,9 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
|
||||
// Function call
|
||||
if (part.functionCall) {
|
||||
results.push(emitFunctionCall(part.functionCall, state));
|
||||
const sig = state.pendingThoughtSignature || null;
|
||||
results.push(emitFunctionCall(part.functionCall, state, sig));
|
||||
state.pendingThoughtSignature = null;
|
||||
}
|
||||
|
||||
// Inline data (images)
|
||||
|
||||
@@ -446,6 +446,13 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
|
||||
state.created = Math.floor(Date.now() / 1000);
|
||||
state.toolCallIndex = 0;
|
||||
state.currentToolCallId = null;
|
||||
// item_id → chat tool_calls index. Deltas carry item_id; keying on it (not
|
||||
// stream position) keeps parallel calls separate when upstream emits all
|
||||
// output_item.added events before any done/delta. Lazily created so callers
|
||||
// that build their own state object (stream.js) need no changes.
|
||||
state.respToolChatIndex ??= new Map();
|
||||
// Indices that already received argument deltas (guards done-with-args).
|
||||
state.respToolArgsEmitted ??= new Set();
|
||||
}
|
||||
|
||||
// Text content delta
|
||||
@@ -464,16 +471,29 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Function call started (standard function_call or custom_tool_call)
|
||||
// Function call started (standard function_call or custom_tool_call).
|
||||
// Index is assigned here (not on done): attributing deltas by stream position
|
||||
// merges parallel calls into index 0 whenever upstream emits all addeds
|
||||
// before dones — the client then concatenates N JSON payloads into one
|
||||
// tool input and fails validation. The server item id is the correlator.
|
||||
if (eventType === "response.output_item.added" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) {
|
||||
const item = data.item;
|
||||
state.currentToolCallId = item.call_id || fallbackToolCallId();
|
||||
state.respToolChatIndex ??= new Map();
|
||||
const key = item.id || data.item_id || state.currentToolCallId;
|
||||
let idx;
|
||||
if (key && state.respToolChatIndex.has(key)) {
|
||||
idx = state.respToolChatIndex.get(key); // duplicate added (retry) — reuse
|
||||
} else {
|
||||
idx = state.toolCallIndex++;
|
||||
if (key) state.respToolChatIndex.set(key, idx);
|
||||
}
|
||||
|
||||
return buildChunk(
|
||||
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
|
||||
{
|
||||
tool_calls: [{
|
||||
index: state.toolCallIndex,
|
||||
index: idx,
|
||||
id: state.currentToolCallId,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: { name: item.name || "", arguments: "" }
|
||||
@@ -482,20 +502,39 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
|
||||
);
|
||||
}
|
||||
|
||||
// Function call arguments delta (standard or custom_tool_call variant)
|
||||
// Function call arguments delta (standard or custom_tool_call variant).
|
||||
// Routed by item_id so interleaved parallel fragments stay on their own call.
|
||||
if (eventType === "response.function_call_arguments.delta" || eventType === "response.custom_tool_call_input.delta") {
|
||||
const argsDelta = data.delta || "";
|
||||
if (!argsDelta) return null;
|
||||
|
||||
const known = data.item_id ? state.respToolChatIndex?.get(data.item_id) : undefined;
|
||||
const idx = known ?? Math.max(0, (state.toolCallIndex || 1) - 1);
|
||||
state.respToolArgsEmitted ??= new Set();
|
||||
state.respToolArgsEmitted.add(idx);
|
||||
return buildChunk(
|
||||
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
|
||||
{ tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsDelta } }] }
|
||||
{ tool_calls: [{ index: idx, function: { arguments: argsDelta } }] }
|
||||
);
|
||||
}
|
||||
|
||||
// Function call done (standard or custom_tool_call variant)
|
||||
// Function call done (standard or custom_tool_call variant).
|
||||
// Index was assigned at added-time; nothing to advance. Some upstreams send
|
||||
// complete arguments only here (no deltas) — emit them once in that case.
|
||||
if (eventType === "response.output_item.done" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) {
|
||||
state.toolCallIndex++;
|
||||
const key = data.item?.id || data.item_id;
|
||||
const idx = (key && state.respToolChatIndex?.get(key)) ?? Math.max(0, (state.toolCallIndex || 1) - 1);
|
||||
const fullArgs = data.item?.arguments;
|
||||
if (typeof fullArgs === "string" && fullArgs) {
|
||||
state.respToolArgsEmitted ??= new Set();
|
||||
if (!state.respToolArgsEmitted.has(idx)) {
|
||||
state.respToolArgsEmitted.add(idx);
|
||||
return buildChunk(
|
||||
{ id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK },
|
||||
{ tool_calls: [{ index: idx, function: { arguments: fullArgs } }] }
|
||||
);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
@@ -20,6 +20,8 @@ export const CLAUDE_BLOCK = {
|
||||
TOOL_RESULT: "tool_result",
|
||||
THINKING: "thinking",
|
||||
REDACTED_THINKING: "redacted_thinking",
|
||||
SERVER_TOOL_USE: "server_tool_use",
|
||||
WEB_SEARCH_TOOL_RESULT: "web_search_tool_result",
|
||||
};
|
||||
|
||||
// OpenAI Responses API item types.
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user