Compare commits
337 Commits
2a37a4085e
...
gitea/new_
| Author | SHA1 | Date | |
|---|---|---|---|
| 39101b3416 | |||
|
|
f01fb909e3 | ||
|
|
c4690307ce | ||
|
|
b54a3f9bb5 | ||
|
|
b65d2d0a6a | ||
|
|
6aea3875ef | ||
|
|
0249464d74 | ||
|
|
239bcfc568 | ||
|
|
199173fe58 | ||
|
|
8f20daac6b | ||
|
|
fdcba3e1b2 | ||
|
|
dd293d3c62 | ||
|
|
7a436d209c | ||
|
|
737b1f4d0a | ||
|
|
37a6b7e0f2 | ||
|
|
06112c136c | ||
|
|
90b0693423 | ||
|
|
fe347e4ea5 | ||
|
|
273f0c32cd | ||
|
|
c2148179c0 | ||
|
|
5d2cfbf3c5 | ||
|
|
3e4323e2fd | ||
|
|
975f28a57c | ||
|
|
ddfdb0df04 | ||
|
|
30464bc227 | ||
|
|
249f6c2fb8 | ||
|
|
f462837536 | ||
|
|
dc198dff1f | ||
| 29ccb84faf | |||
| 3481058e09 | |||
| 272dbcb9cc | |||
|
|
4a57df8bf9 | ||
|
|
a406381fad | ||
|
|
e571a8b6da | ||
|
|
8403576095 | ||
|
|
e7c269b8c9 | ||
|
|
95600db17b | ||
|
|
69cc1aa987 | ||
|
|
001b292876 | ||
|
|
a61fc6a01e | ||
|
|
1b72f02e3b | ||
|
|
2aa99d9f39 | ||
|
|
39e36d3d0c | ||
|
|
910db749aa | ||
|
|
6af26a9ee8 | ||
|
|
cbffeb9770 | ||
|
|
21583c03e5 | ||
|
|
0f488c7027 | ||
|
|
1a02713150 | ||
|
|
b53260ca54 | ||
|
|
d1de324586 | ||
|
|
ce9ac43da5 | ||
|
|
5798b30841 | ||
|
|
782c137b1f | ||
|
|
0d50fe3610 | ||
|
|
3279d31024 | ||
|
|
aa2bc53f3a | ||
|
|
c73bb2cd66 | ||
|
|
6c9fe6f78a | ||
|
|
44fd69d229 | ||
|
|
f84c667d42 | ||
|
|
20014b3104 | ||
|
|
f28e918e24 | ||
|
|
6431e35303 | ||
| 11089ab137 | |||
|
|
41a1b8003d | ||
|
|
b7446f8dd1 | ||
|
|
6886915f62 | ||
|
|
da0046550a | ||
|
|
402745dc1f | ||
|
|
f67d5a0c93 | ||
|
|
c7df895bbb | ||
|
|
be3bc764b1 | ||
|
|
2daf25ffbe | ||
|
|
7c2b1fe3e1 | ||
|
|
253199f16f | ||
|
|
c933eefc27 | ||
|
|
5c217d34f3 | ||
|
|
477b2aed0b | ||
|
|
9f42e7ac1c | ||
|
|
cf663f5300 | ||
|
|
822aa958d1 | ||
|
|
49185137b8 | ||
|
|
73e021b8a0 | ||
|
|
a8c9d3802c | ||
|
|
23ae82d8e3 | ||
|
|
8e15f0bdd8 | ||
|
|
058ceace48 | ||
|
|
d99bc8201d | ||
|
|
bc3be0cb28 | ||
|
|
092c84eac9 | ||
|
|
b3d6e089c6 | ||
|
|
efc80ba2e3 | ||
|
|
4641c2b76a | ||
|
|
93837af09f | ||
|
|
52917a6d49 | ||
|
|
f64229530e | ||
|
|
3ac100d524 | ||
|
|
ef18175226 | ||
|
|
725e2c1187 | ||
|
|
367fc546d8 | ||
|
|
20a43f5a2c | ||
|
|
aa14ef72e2 | ||
|
|
eafac37dcb | ||
|
|
c49efdf528 | ||
|
|
82b1bca42a | ||
|
|
f4f06f290c | ||
|
|
0c6ab4f99b | ||
|
|
6091ff597e | ||
|
|
2b65c49ff5 | ||
|
|
702b57c30d | ||
|
|
912ed295db | ||
| 28e26f4295 | |||
| 81a47f36c6 | |||
|
|
13b468b889 | ||
|
|
9300121366 | ||
|
|
5c399b6406 | ||
|
|
17c4cc7687 | ||
|
|
73cb89143c | ||
|
|
83af3f1853 | ||
|
|
accf2c5296 | ||
|
|
c712641123 | ||
|
|
da6aa90128 | ||
|
|
248d7da01c | ||
|
|
998bb3d975 | ||
|
|
8a81085a72 | ||
|
|
122f23eebc | ||
|
|
f6e7cabe60 | ||
|
|
45ec1d30bb | ||
|
|
781c18d837 | ||
|
|
1892ed77c8 | ||
|
|
1f10f9e5c4 | ||
|
|
832a34659e | ||
|
|
3288bbc47e | ||
|
|
807553e246 | ||
|
|
4a390685b3 | ||
|
|
537b3befd2 | ||
|
|
a7047a07d4 | ||
|
|
eee3515e54 | ||
|
|
40dffbce53 | ||
| a84ba559f3 | |||
|
|
35b950be81 | ||
|
|
7fee56bacd | ||
|
|
4ad1e7a4ba | ||
|
|
e3bf94ee25 | ||
|
|
628ff1eab5 | ||
|
|
e7b5f09d50 | ||
| 302795a613 | |||
| 02f097880d | |||
| 38ff11ee16 | |||
| e3ba5d2207 | |||
| ddfa789a31 | |||
| 6f52d7020c | |||
| 59f17b3725 | |||
| a835771c97 | |||
|
|
eb712ca821 | ||
|
|
11222eff0f | ||
|
|
e214fb1c30 | ||
|
|
f615a83cb2 | ||
|
|
e74db4d0a6 | ||
|
|
77e6a227fe | ||
|
|
1442cc73ce | ||
|
|
fb9fab0206 | ||
|
|
28cfd9facf | ||
|
|
1a3d446831 | ||
|
|
97f3ab97b1 | ||
|
|
0da803eef4 | ||
|
|
81f4f93082 | ||
|
|
cec672d9d9 | ||
|
|
ed963931b4 | ||
|
|
f388b5e56b | ||
|
|
b84681d5a4 | ||
|
|
2ab6a4c949 | ||
|
|
c08efdbe2b | ||
|
|
4eda76e2ab | ||
|
|
e0ffc7e2a1 | ||
|
|
6ab9ca9eb1 | ||
|
|
6efb97904b | ||
|
|
831001c322 | ||
|
|
e7dd72a8d7 | ||
|
|
5caa72f5fb | ||
|
|
e014cb537f | ||
|
|
d1d4e0f02b | ||
|
|
ac98dd9d32 | ||
|
|
a58902e4a7 | ||
|
|
98579f98c1 | ||
|
|
15687d1913 | ||
|
|
acb5c34cdc | ||
|
|
b870b5d41b | ||
|
|
1f190bd00b | ||
|
|
70f15aa50b | ||
|
|
1fe996db6a | ||
|
|
c24a854278 | ||
|
|
1fc2a81d65 | ||
|
|
f6c59d30b0 | ||
|
|
ac9120fde3 | ||
|
|
ee7a961633 | ||
|
|
f68d2f5ee5 | ||
|
|
ed1bd0c528 | ||
|
|
925cb4aade | ||
|
|
b9c92cb83c | ||
|
|
44e4b80bbe | ||
|
|
9d3f7646d1 | ||
|
|
009cac6326 | ||
|
|
38f031f4c9 | ||
|
|
90b52e06ff | ||
|
|
2203cd8f2b | ||
|
|
2fd99eae5d | ||
|
|
df85e16d7a | ||
|
|
dff648496c | ||
|
|
88676b3037 | ||
|
|
bb3cb43e09 | ||
|
|
4a371d1d9f | ||
|
|
d91e8b85e0 | ||
|
|
993c6eb469 | ||
|
|
28d005772a | ||
|
|
2a9213c5bd | ||
|
|
ec6692808b | ||
|
|
e5a13c3ab7 | ||
|
|
67d9182e1a | ||
|
|
9dbdca0e5e | ||
|
|
5a86f6a8d2 | ||
|
|
eb312bd470 | ||
|
|
fcfcced4ab | ||
|
|
56a40765e9 | ||
|
|
cadef6c4ff | ||
|
|
ab044e6d6d | ||
|
|
14401c433c | ||
|
|
e08ac6dada | ||
|
|
f9d82c6575 | ||
|
|
2f17352cc2 | ||
|
|
90a0005845 | ||
|
|
e79ae6e7c5 | ||
|
|
a68ada1c83 | ||
|
|
c4af43faa3 | ||
|
|
d7f7d70dd5 | ||
|
|
9c45b27cd7 | ||
|
|
e6f5724b4b | ||
|
|
d01724556a | ||
|
|
40eed18688 | ||
|
|
0532f00d84 | ||
|
|
9c650e1d54 | ||
|
|
1a3db1efae | ||
|
|
f0a6d35818 | ||
|
|
548e32aacf | ||
|
|
abb20d9f39 | ||
| 86112cee6d | |||
| 5c6048759c | |||
| d0f202a75d | |||
| f0adfb205a | |||
| 1d56e2dbc5 | |||
| 55f10c11e5 | |||
| eedad6c5ea | |||
| 99752a397c | |||
| bb8d67ba9c | |||
| 144dda2ac2 | |||
| bc9719fac7 | |||
| 6770f6ba0b | |||
| c61dc6de46 | |||
| b5c0f10610 | |||
| 1256f29d92 | |||
| de9e00c66d | |||
|
|
699edac327 | ||
|
|
540ebbe682 | ||
|
|
e1115e2839 | ||
|
|
27f3710c8b | ||
|
|
59d858b639 | ||
|
|
92259214db | ||
|
|
b04c03c6b5 | ||
|
|
8b2b2fefb5 | ||
|
|
86694ed8d0 | ||
|
|
8af5e752da | ||
|
|
8ed9da7165 | ||
|
|
7e5f5a8813 | ||
|
|
345cdcf6a5 | ||
|
|
65197ad11c | ||
|
|
e02bde4a70 | ||
|
|
30fec4318e | ||
|
|
67271d859e | ||
|
|
b566b20ade | ||
|
|
6d30ce6de5 | ||
|
|
5b417f9bf2 | ||
|
|
b57c041345 | ||
|
|
8a527fec91 | ||
|
|
70ba0024b0 | ||
|
|
80afb59907 | ||
|
|
10a923da11 | ||
|
|
01858feca0 | ||
|
|
e2a4fe048f | ||
|
|
b44bb09f72 | ||
|
|
456f2a2635 | ||
|
|
cd4003bc8b | ||
|
|
71dcdc1053 | ||
| a3182a7265 | |||
| 386b25ff7f | |||
|
|
15223724c3 | ||
|
|
35f86e5828 | ||
|
|
41588bea01 | ||
|
|
03f8487cc7 | ||
|
|
99639c0540 | ||
|
|
e41d85037d | ||
|
|
02c66fe2bd | ||
|
|
dcdd4628b3 | ||
|
|
6498b3122f | ||
|
|
8e59093db7 | ||
|
|
cd13d904d7 | ||
|
|
fe547f4dc0 | ||
|
|
b480892952 | ||
|
|
3fab15ae3e | ||
|
|
d06e0d26c6 | ||
|
|
b11be8be0a | ||
|
|
42c691b3ea | ||
|
|
a7941ddab4 | ||
|
|
646b3b9ba3 | ||
|
|
baebc9a06e | ||
|
|
25e4bf1c6c | ||
|
|
c570fe33ae | ||
|
|
d0751bcff7 | ||
|
|
948dd8f89b | ||
|
|
86131b9ca4 | ||
|
|
651df2f0e2 | ||
|
|
da8691f866 | ||
|
|
13ed14568d | ||
|
|
1eb37db32d | ||
|
|
d433c0b295 | ||
|
|
3292dfc102 | ||
|
|
9138c99391 | ||
|
|
41606a37a3 | ||
|
|
2abe8b855c | ||
|
|
c06cc08453 | ||
|
|
d6df6576c5 | ||
|
|
786b3013ba | ||
|
|
0648e9e420 | ||
|
|
ae4f76c433 | ||
|
|
918b3c87a1 | ||
|
|
0e5da70cb1 | ||
|
|
f260a1817b |
@@ -14,6 +14,10 @@ NODE_ENV=production
|
||||
API_KEY_SECRET=endpoint-proxy-api-key-secret
|
||||
MACHINE_ID_SALT=endpoint-proxy-salt
|
||||
ENABLE_REQUEST_LOGS=false
|
||||
# Console verbosity: DEBUG | INFO | WARN | ERROR. Default INFO. In production set
|
||||
# ERROR to only print important errors (hides ▶ POST / 📊 DONE / [COMBO] / [CHAT]).
|
||||
# Can also be changed at runtime from dashboard Settings → Logging.
|
||||
# LOG_LEVEL=ERROR
|
||||
OBSERVABILITY_ENABLED=true
|
||||
AUTH_COOKIE_SECURE=false
|
||||
REQUIRE_API_KEY=false
|
||||
|
||||
BIN
.github/issue-assets/combo-defaults/cursor-model-list.png
vendored
Normal file
BIN
.github/issue-assets/combo-defaults/cursor-model-list.png
vendored
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 38 KiB |
BIN
.github/issue-assets/combo-defaults/override-openai-base-url.png
vendored
Normal file
BIN
.github/issue-assets/combo-defaults/override-openai-base-url.png
vendored
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 48 KiB |
429
.github/workflows/docker-publish.yml
vendored
429
.github/workflows/docker-publish.yml
vendored
@@ -5,22 +5,164 @@ on:
|
||||
tags:
|
||||
- "v*"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
release_tag:
|
||||
description: "Existing vX.Y.Z tag to publish"
|
||||
required: true
|
||||
type: string
|
||||
promote_latest:
|
||||
description: "Promote this republish to latest"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
# Keep every release in one FIFO queue. A per-tag group would still allow an
|
||||
# older release to finish after a newer release and move latest backwards.
|
||||
concurrency:
|
||||
group: docker-publish-${{ github.repository }}
|
||||
cancel-in-progress: false
|
||||
queue: max
|
||||
|
||||
env:
|
||||
GHCR_IMAGE: ghcr.io/${{ github.repository }}
|
||||
DOCKERHUB_IMAGE: decolua/9router
|
||||
|
||||
jobs:
|
||||
build-and-push:
|
||||
prepare:
|
||||
name: Validate release
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
outputs:
|
||||
tag: ${{ steps.release.outputs.tag }}
|
||||
version: ${{ steps.release.outputs.version }}
|
||||
commit: ${{ steps.release.outputs.commit }}
|
||||
publish_dockerhub: ${{ steps.release.outputs.publish_dockerhub }}
|
||||
promote_latest: ${{ steps.release.outputs.promote_latest }}
|
||||
ghcr_image: ${{ steps.release.outputs.ghcr_image }}
|
||||
|
||||
steps:
|
||||
- name: Check out release tag
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ inputs.release_tag || github.ref_name }}
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Validate tag and package versions
|
||||
id: release
|
||||
env:
|
||||
RELEASE_TAG: ${{ inputs.release_tag || github.ref_name }}
|
||||
REPOSITORY: ${{ github.repository }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
PROMOTE_LATEST_INPUT: ${{ inputs.promote_latest && 'true' || 'false' }}
|
||||
run: |
|
||||
node <<'NODE'
|
||||
const fs = require("fs");
|
||||
const { execFileSync } = require("child_process");
|
||||
|
||||
const tag = process.env.RELEASE_TAG || "";
|
||||
const match = /^v((?:0|[1-9]\d*)\.(?:0|[1-9]\d*)\.(?:0|[1-9]\d*)(?:-[0-9A-Za-z-]+(?:\.[0-9A-Za-z-]+)*)?)$/.exec(tag);
|
||||
|
||||
if (tag.includes("+")) {
|
||||
console.error(`Build metadata is not supported in Docker release tags: ${tag}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (!match) {
|
||||
console.error(`Expected a Docker-safe semver tag like v0.5.81 or v0.5.81-rc.1, received: ${tag || "<empty>"}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const version = match[1];
|
||||
if (version.length > 128 || !/^[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}$/.test(version)) {
|
||||
console.error(`Version is not a valid Docker tag: ${version}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const prerelease = version.includes("-")
|
||||
? version.slice(version.indexOf("-") + 1).split(".")
|
||||
: [];
|
||||
for (const identifier of prerelease) {
|
||||
if (/^\d+$/.test(identifier) && identifier.length > 1 && identifier.startsWith("0")) {
|
||||
console.error(`Numeric prerelease identifiers cannot contain leading zeroes: ${identifier}`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
const rootVersion = require("./package.json").version;
|
||||
const cliVersion = require("./cli/package.json").version;
|
||||
|
||||
if (rootVersion !== version) {
|
||||
console.error(`package.json version ${rootVersion} does not match tag ${tag}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (cliVersion !== version) {
|
||||
console.error(`cli/package.json version ${cliVersion} does not match tag ${tag}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const commit = execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim();
|
||||
const publishDockerHub = process.env.REPOSITORY === "decolua/9router";
|
||||
const ghcrImage = `ghcr.io/${process.env.REPOSITORY.toLowerCase()}`;
|
||||
const isPrerelease = version.includes("-");
|
||||
const promoteLatest = (process.env.EVENT_NAME === "push" && !isPrerelease)
|
||||
|| process.env.PROMOTE_LATEST_INPUT === "true";
|
||||
const output = process.env.GITHUB_OUTPUT;
|
||||
|
||||
fs.appendFileSync(output, `tag=${tag}\n`);
|
||||
fs.appendFileSync(output, `version=${version}\n`);
|
||||
fs.appendFileSync(output, `commit=${commit}\n`);
|
||||
fs.appendFileSync(output, `publish_dockerhub=${publishDockerHub}\n`);
|
||||
fs.appendFileSync(output, `promote_latest=${promoteLatest}\n`);
|
||||
fs.appendFileSync(output, `ghcr_image=${ghcrImage}\n`);
|
||||
|
||||
console.log(`Validated ${tag} at ${commit}`);
|
||||
console.log(`latest promotion: ${promoteLatest ? "enabled" : "disabled"}`);
|
||||
NODE
|
||||
|
||||
build:
|
||||
name: Build ${{ matrix.platform }}
|
||||
needs: prepare
|
||||
runs-on: ${{ matrix.runner }}
|
||||
timeout-minutes: 60
|
||||
env:
|
||||
GHCR_IMAGE: ${{ needs.prepare.outputs.ghcr_image }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- platform: linux/amd64
|
||||
suffix: amd64
|
||||
runner: ubuntu-24.04
|
||||
- platform: linux/arm64
|
||||
suffix: arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Check out release source at validated commit
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ needs.prepare.outputs.commit }}
|
||||
path: source
|
||||
fetch-depth: 1
|
||||
|
||||
- uses: docker/setup-buildx-action@v3
|
||||
- name: Check out publishing Dockerfile
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ github.workflow_sha }}
|
||||
path: workflow
|
||||
sparse-checkout: |
|
||||
Dockerfile
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@v3
|
||||
@@ -29,32 +171,267 @@ jobs:
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Build and push platform image by digest
|
||||
id: build
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: source
|
||||
file: workflow/Dockerfile
|
||||
platforms: ${{ matrix.platform }}
|
||||
outputs: type=image,name=${{ env.GHCR_IMAGE }},push-by-digest=true,name-canonical=true,push=true
|
||||
build-args: |
|
||||
APP_VERSION=${{ needs.prepare.outputs.version }}
|
||||
ALPINE_MIRROR=${{ vars.ALPINE_MIRROR || 'dl-cdn.alpinelinux.org' }}
|
||||
NPM_REGISTRY=${{ vars.NPM_REGISTRY || 'https://registry.npmjs.org/' }}
|
||||
labels: |
|
||||
org.opencontainers.image.source=https://github.com/${{ github.repository }}
|
||||
org.opencontainers.image.revision=${{ needs.prepare.outputs.commit }}
|
||||
org.opencontainers.image.version=${{ needs.prepare.outputs.version }}
|
||||
cache-from: type=gha,scope=9router-${{ matrix.suffix }}
|
||||
cache-to: type=gha,mode=max,scope=9router-${{ matrix.suffix }}
|
||||
provenance: false
|
||||
sbom: false
|
||||
|
||||
- name: Smoke-test platform image before publishing digest artifact
|
||||
env:
|
||||
GHCR_IMAGE: ${{ env.GHCR_IMAGE }}
|
||||
IMAGE_DIGEST: ${{ steps.build.outputs.digest }}
|
||||
PLATFORM: ${{ matrix.platform }}
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
[[ "$IMAGE_DIGEST" =~ ^sha256:[0-9a-f]{64}$ ]]
|
||||
|
||||
container="9router-platform-smoke-${GITHUB_RUN_ID}-${{ matrix.suffix }}"
|
||||
trap 'docker rm -f "$container" >/dev/null 2>&1 || true' EXIT
|
||||
|
||||
docker run --detach \
|
||||
--name "$container" \
|
||||
--platform "$PLATFORM" \
|
||||
--publish 20128:20128 \
|
||||
"${GHCR_IMAGE}@${IMAGE_DIGEST}"
|
||||
|
||||
for attempt in {1..45}; do
|
||||
if curl --fail --silent --show-error http://127.0.0.1:20128/api/health; then
|
||||
echo "${PLATFORM} health check passed"
|
||||
exit 0
|
||||
fi
|
||||
if (( attempt % 5 == 0 )); then
|
||||
echo "Waiting for ${PLATFORM} health check (${attempt}/45)" >&2
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
echo "${PLATFORM} health check failed; container logs follow:" >&2
|
||||
docker logs "$container" || true
|
||||
exit 1
|
||||
|
||||
- name: Save image digest
|
||||
env:
|
||||
IMAGE_DIGEST: ${{ steps.build.outputs.digest }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
test -n "$IMAGE_DIGEST"
|
||||
mkdir -p "$RUNNER_TEMP/digests"
|
||||
printf '%s\n' "$IMAGE_DIGEST" > "$RUNNER_TEMP/digests/${{ matrix.suffix }}.txt"
|
||||
|
||||
- name: Upload image digest
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: digests-${{ matrix.suffix }}
|
||||
path: ${{ runner.temp }}/digests/${{ matrix.suffix }}.txt
|
||||
if-no-files-found: error
|
||||
|
||||
publish:
|
||||
name: Publish and verify manifest
|
||||
needs:
|
||||
- prepare
|
||||
- build
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
env:
|
||||
GHCR_IMAGE: ${{ needs.prepare.outputs.ghcr_image }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
steps:
|
||||
- name: Download platform digests
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
pattern: digests-*
|
||||
path: ${{ runner.temp }}/digests
|
||||
merge-multiple: true
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Create and verify version manifest
|
||||
env:
|
||||
GHCR_IMAGE: ${{ env.GHCR_IMAGE }}
|
||||
VERSION: ${{ needs.prepare.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
shopt -s nullglob
|
||||
digest_files=("$RUNNER_TEMP"/digests/*.txt)
|
||||
|
||||
if [[ "${#digest_files[@]}" -ne 2 ]]; then
|
||||
echo "Expected two platform digests, found ${#digest_files[@]}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
sources=()
|
||||
for digest_file in "${digest_files[@]}"; do
|
||||
digest="$(tr -d '\n' < "$digest_file")"
|
||||
if [[ ! "$digest" =~ ^sha256:[0-9a-f]{64}$ ]]; then
|
||||
echo "Invalid image digest in $digest_file: $digest" >&2
|
||||
exit 1
|
||||
fi
|
||||
sources+=("${GHCR_IMAGE}@${digest}")
|
||||
done
|
||||
|
||||
docker buildx imagetools create \
|
||||
--tag "${GHCR_IMAGE}:${VERSION}" \
|
||||
"${sources[@]}"
|
||||
|
||||
docker buildx imagetools inspect "${GHCR_IMAGE}:${VERSION}" | tee "$RUNNER_TEMP/version-manifest.txt"
|
||||
docker buildx imagetools inspect --raw "${GHCR_IMAGE}:${VERSION}" > "$RUNNER_TEMP/version-manifest.json"
|
||||
|
||||
expected=$'linux/amd64\nlinux/arm64'
|
||||
actual="$(jq -r '[.manifests[] | select(.platform != null and .platform.os != null and .platform.architecture != null) | "\(.platform.os)/\(.platform.architecture)"] | sort | .[]' "$RUNNER_TEMP/version-manifest.json")"
|
||||
if [[ "$actual" != "$expected" ]]; then
|
||||
echo "Version manifest platforms do not match exactly:" >&2
|
||||
printf '%s\n' "$actual" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Smoke-test resolved version manifest
|
||||
env:
|
||||
GHCR_IMAGE: ${{ env.GHCR_IMAGE }}
|
||||
VERSION: ${{ needs.prepare.outputs.version }}
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
container="9router-manifest-smoke-${GITHUB_RUN_ID}"
|
||||
trap 'docker rm -f "$container" >/dev/null 2>&1 || true' EXIT
|
||||
|
||||
docker run --detach \
|
||||
--name "$container" \
|
||||
--platform linux/amd64 \
|
||||
--publish 20128:20128 \
|
||||
"${GHCR_IMAGE}:${VERSION}"
|
||||
|
||||
for attempt in {1..30}; do
|
||||
if curl --fail --silent --show-error http://127.0.0.1:20128/api/health; then
|
||||
echo "Resolved version manifest health check passed"
|
||||
exit 0
|
||||
fi
|
||||
if (( attempt % 5 == 0 )); then
|
||||
echo "Waiting for resolved manifest health check (${attempt}/30)" >&2
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
echo "Resolved version manifest health check failed; container logs follow:" >&2
|
||||
docker logs "$container" || true
|
||||
exit 1
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
if: needs.prepare.outputs.publish_dockerhub == 'true'
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Extract metadata
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: |
|
||||
${{ env.GHCR_IMAGE }}
|
||||
${{ env.DOCKERHUB_IMAGE }}
|
||||
tags: |
|
||||
type=semver,pattern={{version}}
|
||||
type=raw,value=latest,enable={{is_default_branch}}
|
||||
- name: Publish version image to Docker Hub
|
||||
if: needs.prepare.outputs.publish_dockerhub == 'true'
|
||||
env:
|
||||
DOCKERHUB_IMAGE: ${{ env.DOCKERHUB_IMAGE }}
|
||||
GHCR_IMAGE: ${{ env.GHCR_IMAGE }}
|
||||
VERSION: ${{ needs.prepare.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker buildx imagetools create \
|
||||
--tag "${DOCKERHUB_IMAGE}:${VERSION}" \
|
||||
"${GHCR_IMAGE}:${VERSION}"
|
||||
|
||||
- name: Build and push
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
cache-from: type=registry,ref=${{ env.GHCR_IMAGE }}:buildcache
|
||||
cache-to: type=registry,ref=${{ env.GHCR_IMAGE }}:buildcache,mode=max
|
||||
platforms: linux/amd64,linux/arm64
|
||||
provenance: false
|
||||
sbom: false
|
||||
docker buildx imagetools inspect "${DOCKERHUB_IMAGE}:${VERSION}" | tee "$RUNNER_TEMP/dockerhub-version-manifest.txt"
|
||||
docker buildx imagetools inspect --raw "${DOCKERHUB_IMAGE}:${VERSION}" > "$RUNNER_TEMP/dockerhub-version-manifest.json"
|
||||
|
||||
expected=$'linux/amd64\nlinux/arm64'
|
||||
actual="$(jq -r '[.manifests[] | select(.platform != null and .platform.os != null and .platform.architecture != null) | "\(.platform.os)/\(.platform.architecture)"] | sort | .[]' "$RUNNER_TEMP/dockerhub-version-manifest.json")"
|
||||
if [[ "$actual" != "$expected" ]]; then
|
||||
echo "Docker Hub version manifest platforms do not match exactly:" >&2
|
||||
printf '%s\n' "$actual" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Record latest promotion policy
|
||||
env:
|
||||
PROMOTE_LATEST: ${{ needs.prepare.outputs.promote_latest }}
|
||||
VERSION: ${{ needs.prepare.outputs.version }}
|
||||
run: |
|
||||
if [[ "$PROMOTE_LATEST" == "true" ]]; then
|
||||
echo "### Latest promotion" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- Policy: promote \`latest\` after the verified ${VERSION} manifest." >> "$GITHUB_STEP_SUMMARY"
|
||||
else
|
||||
echo "### Latest promotion" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- Policy: leave \`latest\` unchanged; this is a numbered-tag-only manual republish." >> "$GITHUB_STEP_SUMMARY"
|
||||
fi
|
||||
|
||||
- name: Promote verified version to latest
|
||||
if: needs.prepare.outputs.promote_latest == 'true'
|
||||
env:
|
||||
DOCKERHUB_IMAGE: ${{ env.DOCKERHUB_IMAGE }}
|
||||
GHCR_IMAGE: ${{ env.GHCR_IMAGE }}
|
||||
PUBLISH_DOCKERHUB: ${{ needs.prepare.outputs.publish_dockerhub }}
|
||||
VERSION: ${{ needs.prepare.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
docker buildx imagetools create \
|
||||
--tag "${GHCR_IMAGE}:latest" \
|
||||
"${GHCR_IMAGE}:${VERSION}"
|
||||
|
||||
if [[ "$PUBLISH_DOCKERHUB" == "true" ]]; then
|
||||
docker buildx imagetools create \
|
||||
--tag "${DOCKERHUB_IMAGE}:latest" \
|
||||
"${GHCR_IMAGE}:${VERSION}"
|
||||
fi
|
||||
|
||||
docker buildx imagetools inspect "${GHCR_IMAGE}:latest" | tee "$RUNNER_TEMP/ghcr-latest-manifest.txt"
|
||||
docker buildx imagetools inspect --raw "${GHCR_IMAGE}:latest" > "$RUNNER_TEMP/ghcr-latest-manifest.json"
|
||||
|
||||
expected=$'linux/amd64\nlinux/arm64'
|
||||
actual="$(jq -r '[.manifests[] | select(.platform != null and .platform.os != null and .platform.architecture != null) | "\(.platform.os)/\(.platform.architecture)"] | sort | .[]' "$RUNNER_TEMP/ghcr-latest-manifest.json")"
|
||||
if [[ "$actual" != "$expected" ]]; then
|
||||
echo "GHCR latest manifest platforms do not match exactly:" >&2
|
||||
printf '%s\n' "$actual" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ "$PUBLISH_DOCKERHUB" == "true" ]]; then
|
||||
docker buildx imagetools inspect "${DOCKERHUB_IMAGE}:latest" | tee "$RUNNER_TEMP/dockerhub-latest-manifest.txt"
|
||||
docker buildx imagetools inspect --raw "${DOCKERHUB_IMAGE}:latest" > "$RUNNER_TEMP/dockerhub-latest-manifest.json"
|
||||
actual="$(jq -r '[.manifests[] | select(.platform != null and .platform.os != null and .platform.architecture != null) | "\(.platform.os)/\(.platform.architecture)"] | sort | .[]' "$RUNNER_TEMP/dockerhub-latest-manifest.json")"
|
||||
if [[ "$actual" != "$expected" ]]; then
|
||||
echo "Docker Hub latest manifest platforms do not match exactly:" >&2
|
||||
printf '%s\n' "$actual" >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
{
|
||||
echo "### Published Docker images"
|
||||
echo "- GHCR: \`${GHCR_IMAGE}:${VERSION}\`"
|
||||
echo "- GHCR latest: \`${GHCR_IMAGE}:latest\`"
|
||||
if [[ "$PUBLISH_DOCKERHUB" == "true" ]]; then
|
||||
echo "- Docker Hub: \`${DOCKERHUB_IMAGE}:${VERSION}\`"
|
||||
echo "- Docker Hub latest: \`${DOCKERHUB_IMAGE}:latest\`"
|
||||
fi
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
176
.github/workflows/tray-binaries.yml
vendored
Normal file
176
.github/workflows/tray-binaries.yml
vendored
Normal file
@@ -0,0 +1,176 @@
|
||||
name: Build macOS tray binary (arm64)
|
||||
|
||||
# systray2 ships only an x86_64 tray_darwin_release, so Apple Silicon users need
|
||||
# Rosetta 2 for the menubar icon. This builds the native arm64 overlay that
|
||||
# cli/hooks/trayRuntime.js downloads from the `tray-binaries` release.
|
||||
#
|
||||
# Manual-only: the artifact's sha256 is pinned in cli/hooks/trayRuntime.js and
|
||||
# verified on every download, so a new build is only publishable together with a
|
||||
# matching pin. Running this with publish=true against a mismatched pin fails
|
||||
# rather than silently bricking every Apple Silicon client.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
publish:
|
||||
description: "Upload to the tray-binaries release (requires sha to match ARM64_TRAY_SHA256)"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
concurrency:
|
||||
group: tray-binaries-${{ github.repository }}
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# Pinned because -trimpath only makes the build reproducible for a given Go
|
||||
# version and macOS SDK. Bumping this changes the sha256.
|
||||
GO_VERSION: "1.27.1"
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: Build darwin/arm64
|
||||
runs-on: macos-15
|
||||
timeout-minutes: 20
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: 22
|
||||
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version: ${{ env.GO_VERSION }}
|
||||
# The Go module lives in a temp clone of the upstream repo, so there is
|
||||
# no go.sum at the workspace root for setup-go's cache to key on.
|
||||
cache: false
|
||||
|
||||
- name: Record SDK provenance
|
||||
run: |
|
||||
{
|
||||
echo "runner macOS: $(sw_vers -productVersion)"
|
||||
echo "Xcode: $(xcodebuild -version | head -1)"
|
||||
echo "clang: $(clang --version | head -1)"
|
||||
echo "Go: $(go version)"
|
||||
} | tee sdk-provenance.txt
|
||||
|
||||
- name: Build
|
||||
run: node cli/scripts/buildTrayArm64.js
|
||||
|
||||
- name: Compare against pinned checksum
|
||||
id: sha
|
||||
run: |
|
||||
BUILT=$(shasum -a 256 cli/.tray-build/tray_darwin_arm64 | cut -d' ' -f1)
|
||||
# Whitespace-tolerant, and a missing constant must fail loudly: a null
|
||||
# match would otherwise surface as an opaque TypeError from [1].
|
||||
PINNED=$(node -e '
|
||||
const m = require("fs").readFileSync("cli/hooks/trayRuntime.js", "utf8")
|
||||
.match(/ARM64_TRAY_SHA256\s*=\s*"([0-9a-f]{64})"/);
|
||||
if (!m) { console.error("::error::ARM64_TRAY_SHA256 not found in cli/hooks/trayRuntime.js"); process.exit(1); }
|
||||
process.stdout.write(m[1]);
|
||||
')
|
||||
{
|
||||
echo "built=$BUILT"
|
||||
echo "pinned=$PINNED"
|
||||
if [ "$BUILT" = "$PINNED" ]; then echo "match=true"; else echo "match=false"; fi
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Write job summary
|
||||
run: |
|
||||
{
|
||||
echo "### tray_darwin_arm64"
|
||||
echo ""
|
||||
echo "| | |"
|
||||
echo "|---|---|"
|
||||
echo "| built sha256 | \`${{ steps.sha.outputs.built }}\` |"
|
||||
echo "| pinned sha256 | \`${{ steps.sha.outputs.pinned }}\` |"
|
||||
echo "| match | ${{ steps.sha.outputs.match }} |"
|
||||
echo ""
|
||||
echo '```'
|
||||
cat sdk-provenance.txt
|
||||
echo '```'
|
||||
echo ""
|
||||
if [ "${{ steps.sha.outputs.match }}" = "true" ]; then
|
||||
echo "Pin already matches — safe to re-run with \`publish=true\`."
|
||||
else
|
||||
echo "⚠️ Pin does **not** match. To publish this build, set \`ARM64_TRAY_SHA256\`"
|
||||
echo "in \`cli/hooks/trayRuntime.js\` to the built sha256 above and land that"
|
||||
echo "change first. Publishing without it makes every Apple Silicon client fail"
|
||||
echo "checksum verification and fall back to the Rosetta binary."
|
||||
fi
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
# Uploaded before the mismatch gate below, so a publish run that fails on a
|
||||
# checksum mismatch still leaves the bytes downloadable — that is exactly
|
||||
# the run where a maintainer needs them to verify the new sha256.
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: tray_darwin_arm64
|
||||
path: |
|
||||
cli/.tray-build/tray_darwin_arm64
|
||||
sdk-provenance.txt
|
||||
|
||||
- name: Refuse to publish on checksum mismatch
|
||||
if: ${{ inputs.publish && steps.sha.outputs.match != 'true' }}
|
||||
run: |
|
||||
echo "::error::publish requested but built sha256 != ARM64_TRAY_SHA256"
|
||||
echo " built: ${{ steps.sha.outputs.built }}"
|
||||
echo " pinned: ${{ steps.sha.outputs.pinned }}"
|
||||
echo "Update cli/hooks/trayRuntime.js and land it before publishing."
|
||||
exit 1
|
||||
|
||||
- name: Publish to tray-binaries release
|
||||
if: ${{ inputs.publish && steps.sha.outputs.match == 'true' }}
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
# Clients fetch from the hardcoded ARM64_TRAY_URL, so publishing from a
|
||||
# different repo would gate an asset stream nobody downloads and silently
|
||||
# decouple the sha pin from the bytes Apple Silicon users execute.
|
||||
URL_REPO=$(node -e '
|
||||
const m = require("fs").readFileSync("cli/hooks/trayRuntime.js", "utf8")
|
||||
.match(/"https:\/\/github\.com\/([^\/"]+\/[^\/"]+)\/releases\/download\/tray-binaries\/tray_darwin_arm64"/);
|
||||
if (!m) { console.error("::error::ARM64_TRAY_URL not found in cli/hooks/trayRuntime.js"); process.exit(1); }
|
||||
process.stdout.write(m[1]);
|
||||
')
|
||||
if [ "${{ github.repository }}" != "$URL_REPO" ]; then
|
||||
echo "::error::publishing to ${{ github.repository }}, but ARM64_TRAY_URL points clients at $URL_REPO"
|
||||
echo "Repoint ARM64_TRAY_URL in cli/hooks/trayRuntime.js at this repo, or run the publish from $URL_REPO."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# gh release upload does not create the release, so bootstrap it on the
|
||||
# first publish run rather than failing with "release not found".
|
||||
if ! gh release view tray-binaries >/dev/null 2>&1; then
|
||||
echo "Release 'tray-binaries' does not exist yet — creating it"
|
||||
gh release create tray-binaries --latest=false \
|
||||
--title "Native macOS tray binaries" \
|
||||
--notes "Built by .github/workflows/tray-binaries.yml. Provenance and the pinned sha256 live in that workflow and in cli/hooks/trayRuntime.js (ARM64_TRAY_SHA256)."
|
||||
fi
|
||||
|
||||
gh release upload tray-binaries cli/.tray-build/tray_darwin_arm64 --clobber
|
||||
|
||||
echo "Uploaded. Verifying public download URL..."
|
||||
URL="https://github.com/${{ github.repository }}/releases/download/tray-binaries/tray_darwin_arm64"
|
||||
GOT=""
|
||||
for attempt in 1 2 3; do
|
||||
if curl -fsSL --max-time 60 -o /tmp/verify "$URL"; then
|
||||
GOT=$(shasum -a 256 /tmp/verify | cut -d' ' -f1)
|
||||
if [ "$GOT" = "${{ steps.sha.outputs.built }}" ]; then break; fi
|
||||
fi
|
||||
# A just-uploaded asset can 404 or serve stale bytes until the CDN catches up.
|
||||
echo "attempt $attempt: got '${GOT:-<download failed>}' — retrying in 15s"
|
||||
sleep 15
|
||||
done
|
||||
if [ "$GOT" != "${{ steps.sha.outputs.built }}" ]; then
|
||||
echo "::error::downloaded asset sha256 '${GOT:-<none>}' != built ${{ steps.sha.outputs.built }} after 3 attempts"
|
||||
exit 1
|
||||
fi
|
||||
echo "✅ $URL serves the expected bytes"
|
||||
|
||||
20
.gitignore
vendored
20
.gitignore
vendored
@@ -87,5 +87,25 @@ graphify-out/*
|
||||
.PR/
|
||||
.next-analyze/*
|
||||
|
||||
# Kiro local workspace state
|
||||
.kiro/
|
||||
9router-*
|
||||
|
||||
# Local sensitive / temp files
|
||||
.engine-token.txt
|
||||
.tmp-prov.json
|
||||
.tmp-providers.json
|
||||
.oauth-session.json
|
||||
.dev-server.log
|
||||
.dev-server-err.log
|
||||
.start-dev.ps1
|
||||
start-dev-silent.cjs
|
||||
debug.log
|
||||
|
||||
# CommandCode CLI local state (auth/taste/projects)
|
||||
.commandcode/
|
||||
|
||||
# Pi subagent run artifacts
|
||||
.pi-subagents/
|
||||
# Vitest local artifacts
|
||||
.vitest/
|
||||
|
||||
517
CHANGELOG.md
517
CHANGELOG.md
@@ -1,17 +1,467 @@
|
||||
# Unreleased
|
||||
# v0.5.91 (2026-09-26)
|
||||
|
||||
## Features
|
||||
|
||||
- **Stream error patterns**: per-provider `streamErrorPatterns` setting (UI: provider page → Stream Error Patterns) — HTTP-200 streams whose first bytes match configured patterns (plain text or `/regex/`) are treated as failed requests: fallback works for non-streaming and early stream errors, and late streaming errors are logged as FAILED. Zero overhead when unconfigured.
|
||||
- **Providers**: add Token Harbor provider and four OpenAI-compatible aggregator providers (dahl, atria, agnes, bai)
|
||||
- **Claude**: forward `x-claude-code-session-id` on OAuth requests; merge client `anthropic-beta` flags and forward rate-limit headers; return thinking text to OpenAI-format clients
|
||||
- **Codex**: add GPT-6 Sol and Luna support
|
||||
- **CLI Tools**: support multiple model profiles for Codex CLI
|
||||
- **Hermes**: multi-role model config (delegation + auxiliary slots)
|
||||
- **OpenCode Go**: complete the Go catalog (40 models) with auto-fetch + family endpoint regex
|
||||
- **Usage**: show and redeem free limit resets for cc accounts
|
||||
- **Cline**: expose the `cline-free/*` tier and price it at zero
|
||||
- **Combos**: display vision adapter models in an ordered table view
|
||||
|
||||
## Fixes
|
||||
- **Claude**: decloak tool names when `toolNameMap` misses (#4342); update spoofed cli version to 2.1.280 to support Opus 5.5
|
||||
- **Providers API**: make POST `/api/providers` O(1) and refuse silent key overwrite (#4350)
|
||||
- **Capabilities**: stop caching the catalog source per module copy (#4351)
|
||||
- **OAuth**: stop Zed paste-token crash and add IDE auto-import (#4359)
|
||||
- **Dashboard**: resolve combo limits with the server's capabilities (#4360); lazy-load charts and `marked`, preload in background on idle
|
||||
- **Responses**: carry the streamed output items in `response.completed` (#4307)
|
||||
- **STT**: dispatch live-API-only Gemini models over the Live WebSocket transport (#4006)
|
||||
- **Gemini**: guard terminal model turns and unresponded functionCalls in `normalizeGeminiContents`
|
||||
- **Command Code**: replay raw byte chunks to preserve all NDJSON lines
|
||||
- **Translator**: stop emitting empty `<think>` markers into OpenAI content
|
||||
- **CLI Tools**: refresh Codex settings after apply (#4347); keep existing `ANTHROPIC_AUTH_TOKEN` when applying Claude settings
|
||||
- **Tray**: native arm64 macOS menubar binary, no Rosetta required
|
||||
- **CLI**: filter model selector by active connections and noAuth providers
|
||||
- **Usage**: key live byApiKey stats by full api key to prevent team-key collision and preserve API key usage attribution
|
||||
- **Tailscale**: cap enable-flow health wait at 20s
|
||||
|
||||
- **CommandCode**: in-stream `{"type":"error"}` events now emit OpenAI error chunks + an executor early-peek → 502 fallback instead of fake success content (`[CommandCode error: ...]`).
|
||||
# v0.5.86 (2026-09-23)
|
||||
|
||||
## Features
|
||||
- **Xiaomi MiMo**: server-assisted desktop login for headless/Docker deployments, five account clusters (cn/sgp/ams/ru/in), and v2.6 pro/flash/pro-ultraspeed models with dual-route (account service vs. cloud API)
|
||||
- **Claude**: add Claude Opus 5.5 support
|
||||
- **i18n**: translate React text rewrites via characterData mutation observer
|
||||
|
||||
## Fixes
|
||||
- **Proxy Pools**: keep request headers intact through Vercel/Cloudflare/Deno relays (spreading a `Headers` instance yielded `{}`, dropping auth and content-type)
|
||||
- **Xiaomi MiMo login**: keep the session in the httpOnly cookie only, require dashboard auth on the proxy branch, and stop forwarding authorization headers upstream
|
||||
|
||||
# v0.5.85 (2026-09-22)
|
||||
|
||||
## Features
|
||||
- **System One**: add `/v1/systemone` decision endpoint for Jev models (OpenCode Zen and OpenRouter lanes), wire into sidebar and Media Providers page with interactive probe testing
|
||||
- **CLI Tools**: add dynamic configuration, settings APIs, and official logos for Pi, OMP, Crush, ForgeCode, Smelt, and CodeWhale
|
||||
- **Analytics & Usage**: add Requests mode, provider/model breakdown charts, All Time period filter, and refined overview cards
|
||||
- **Combos**: add Cursor/Claude Default presets; support bulk select/delete and bulk strategy changes (Fallback / Round Robin / Fusion)
|
||||
- **Model Capabilities**: expose model capability metadata on `/v1/models` and aggregate capabilities across combo targets
|
||||
- **OpenCode Zen & MiMo**: add OpenCode Zen (`opencode-zen`) provider with free-tier fingerprint; switch default vision fallback to MiMo V2.6 Flash Free
|
||||
- **Qoder CN**: add `qoder-cn` provider for qoder.com.cn with OAuth flow, COSY protocol, and CN gateway routing
|
||||
|
||||
## Fixes
|
||||
- **Translator**: map Claude `refusal` stop_reason to `content_filter` and surface explanation; strip replayed reasoning fields for Groq, Mistral, and Cerebras (#4220)
|
||||
- **Antigravity**: drop requestType `agent` to avoid false 429 `RESOURCE_EXHAUSTED`; separate weekly and short-window (5-hour) quotas and deduplicate dashboard rows
|
||||
- **Responses API**: report usage on `response.completed` so clients can auto-compact (#3432)
|
||||
- **Hugging Face**: migrate to Inference Providers router (`router.huggingface.co`), expand image models catalog, and add STT route
|
||||
- **Qoder**: prevent signed request replay (`403/103 Duplicate request`), handle code 110 billing blocks, and preserve upstream SSE error status
|
||||
- **Performance**: bound usage `lastUsed` scan to a 2-day window; map large budget tokens to `max` reasoning tier
|
||||
- **Docker**: publish verified multi-platform images (linux/amd64 and linux/arm64) with configurable apk build mirrors
|
||||
|
||||
# v0.5.81 (2026-09-18)
|
||||
|
||||
## Features
|
||||
- **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow
|
||||
- **Claude Code**: add 1M-context toggle (`[1m]` marker) and drive `CLAUDE_CODE_AUTO_COMPACT_WINDOW` directly from the dashboard
|
||||
- **Models**: add DeepSeek-V4.1-Flash to DeepSeek provider, CodeBuddy-Intl, and Ollama (`deepseek-v4.1-flash:cloud`); enable `low`..`max` reasoning effort levels and vision capability for DeepSeek-V4.*
|
||||
- **i18n**: integrate Persian (fa) translation
|
||||
|
||||
## Fixes
|
||||
- **Cursor**: stop AgentService empty turns (`OUT 0`) and silent hangs — fold system prompts instead of `custom_system_prompt`, send `ModelDetails`, read Composer/Grok `thinking_delta`, ack request-context without echoing MCP tools, and reject IDE execs so the model can continue
|
||||
- **RTK**: for Cursor, compress source-format `tool_result` / `role:tool` **before** translation — its translator rewrites those shapes, so post-translate compression missed them. Other providers keep the post-translate pass unchanged
|
||||
- **OpenCode / OpenCode Go**: resolve 403 `FreeTierError` and 429 rate limits with canonical session format, valid User-Agent, and stable upstream session reuse; force stream and declare `forceStream` for free-tier SSE aggregation; cloak decoy tools, normalize Muse Free tool choice, and strip prior reasoning items on Responses models; route Union Alpha via Messages API
|
||||
- **Kiro**: preserve underscores in tool names (`mcp__server__tool`) and restore client tool names in responses; use neutral placeholder for tool-result-only turns; forward tool-result images
|
||||
- **Stream**: report aborts after HTTP 200 in-band (per-format error frames) instead of closing silently
|
||||
- **Command Code**: preserve images and `reasoning_effort` on `/alpha/generate`; retry transient stream errors and avoid fake stop chunks; add Quota Tracker support
|
||||
- **Zed**: harden OAuth lifecycle (preserve `systemId`, renew proxy timeout), support live model resolution, and lower display priority in OAuth list
|
||||
- **Antigravity**: scope cached thought signatures to model family; strip Claude Code billing headers from system prompts; sanitize Hermes system identity
|
||||
- **Codex**: route bare `codex-auto-review` requests to the Codex provider (#4135)
|
||||
- **Auth**: do not cool down an account for request-scoped 4xx errors
|
||||
- **Usage**: improve DeepSeek credit balance display as currency credit instead of 0/total quota bar
|
||||
- **Model Catalog**: scope synced catalog to gateways and declare vision capabilities for DeepSeek V4.1-Flash IDs
|
||||
|
||||
# v0.5.75 (2026-09-10)
|
||||
|
||||
## Features
|
||||
- **Video**: add OpenRouter and Vertex AI (Veo) video generation on `/v1/videos/*` via a provider adapter layer; poll requests resolve their provider from `x-connection-id` or `?provider=`
|
||||
- **Antigravity**: add weekly quota tracking (Gemini weekly / Claude & GPT weekly) and free-tier handling from `retrieveUserQuotaSummary` (#3892)
|
||||
- **Codex**: add GPT Image 2.5, Flare and Sunburst image models with multi-image support; add the same ids to the OpenAI catalog
|
||||
- **Qoder**: surface usage to all clients and stop inlining large attachments — images upload through `/api/v2/image/upload` like qodercli, oversized file blocks become stubs, context tier auto-escalates
|
||||
- **OpenCode Go**: add newly published models (glm-5.3, kimi-k3, deepseek-flash, longcat-2.0, hy4-preview, hy3 on chat/completions; qwen3.8-max, qwen3.8-flash on `/messages`; grok-4.6, gpt-5.6-luna on Responses) and list `deepseek-v4.1-flash` first in the catalog
|
||||
- **CLI tools**: group the model selector by provider with full-text search and manual custom model ID entry
|
||||
- **CodeBuddy-CN**: replace `deepseek-v4-flash` with `deepseek-v4.1-flash`
|
||||
|
||||
## Fixes
|
||||
- **Tools**: scope Claude tool type defaulting to gateways declaring `requireClaudeToolType` — the global default broke Anthropic-compatible endpoints that only accept the legacy typeless tool shape (#3905)
|
||||
- **Claude**: cap re-anchored `cache_control` at the 4-marker budget so a spent budget no longer 400s and triggers a full combo failover; wrap bare single-object content turns before the mid-conversation-system fold
|
||||
- **Cline / Airforce**: unwrap the `{"success":true,"data":…}` envelope on non-stream chat completions (#3644); add the live Cline/ClinePass model catalog and refresh Airforce free models
|
||||
- **Cline**: stop `workos:`-prefixing ClinePass API keys (401 on every request, #2333) and add clinepass token refresh
|
||||
- **Kiro**: never send a top-level `systemPrompt` (`400 REQUEST_BODY_INVALID`); route requests through current runtime surfaces (#3776)
|
||||
- **Codex**: strip Unicode-property tool schema patterns the validator rejects (#3922); restore the `Version` header and single-source the CLI version
|
||||
- **DeepSeek**: keep Anthropic-only tool types when forwarding to `/anthropic/v1/messages`
|
||||
- **Qoder**: drop the Responses usage plumbing from shared translator/handler code, which changed token accounting for every provider, not just Qoder
|
||||
- **Antigravity**: normalize contents and handle intermediate tool responses; protect the OAuth token-refresh path from Google anti-abuse rate limits (#3813)
|
||||
- **Providers**: clear stale connection health state (`modelLock_*`, `backoffLevel`, `rateLimitedUntil`, `errorCode`) when a connection is re-validated (#3810, #3830); remove the duplicate `qwen` provider that shadowed `alims-intl`
|
||||
- **Video / Vertex**: reject job ids and model ids that would escape the request URL path (SSRF)
|
||||
- **Usage**: parse the Fable weekly limit from `limits[]` instead of fabricating a row (#3847)
|
||||
- **Auth**: set a 24h `maxAge` on the dashboard session cookie
|
||||
|
||||
# v0.5.70 (2026-09-08)
|
||||
|
||||
## Features
|
||||
- **Providers**: creating a compatible / custom-embedding node now registers the endpoint only — the API key is added afterwards from the node's page, like the built-in providers. The create dialogs drop the API Key / Model ID / Check fields, and `POST /api/provider-nodes` no longer accepts credentials at all, so a node can never be half-created
|
||||
- **Providers**: compatible nodes now use the same model rows as built-in providers — capability badges, copy, per-model test, alias handling and the Add/Edit Model modal with vision + reasoning toggles, replacing the weaker read-only list
|
||||
- **Providers**: compatible nodes get the built-in bulk toolbars: Test All / Disable / Active / Select All over connections, and Test All Models / Disable All / Active All over models, with per-row disable and a restore strip for disabled models
|
||||
- **Providers**: wire the dead "Fetch Models" button on compatible nodes to the live upstream catalog, de-duplicating against already-added models
|
||||
|
||||
## Fixes
|
||||
- **Usage**: nested combos (comboA lists comboB, comboC, …) stay one slot each — the inner combo always runs as fallback to produce a single answer, failed hops are not written to Details/usage, and streaming no longer inserts a 0-token placeholder. One user message against a nested fallback combo is one request row with real tokens; fusion of nested combos is N panel slots + judge, not every nested leaf
|
||||
- **Models**: persist per-model capability assertions for custom and compatible providers and honor them everywhere — unsupported media is stripped on the chat path, `/v1/models` and `/api/models` report what the user asserted, and thinking translation follows it (asserting `reasoning:false` now actually strips thinking fields, `reasoning:true` emits them)
|
||||
- **Models**: partial capability edits merge instead of overwriting, so toggling vision off no longer erases a stored reasoning assertion
|
||||
- **Capabilities**: keep server-injected readers (synced catalog, user-asserted capabilities) in process-wide state — Next.js compiles startup and each API route into separate bundles with their own module instances, so a boot-time install was invisible to every request handler and the models.dev catalog contributed nothing to upstream requests since 0532f00d
|
||||
- **Dashboard**: thinking-level picker and model-row suffix reflect user-asserted reasoning on compatible nodes
|
||||
- **Providers**: `Default Model` is optional when adding an API key to a compatible node — the node's own model list (and the picker in the test modals) already determine what gets probed, and the built-in fallback still covers connection checks
|
||||
- **Providers**: restore the `useCopyToClipboard` import dropped from the provider detail page, which crashed the route with `ReferenceError` for every provider
|
||||
- **DB**: restore `getModelAliases` / `setModelAlias` / `deleteModelAlias` re-exports dropped from the `localDb` shim by 86112cee, which broke `GET /api/models` and `GET /v1/models` at import time
|
||||
- **Providers**: remove dead `PassthroughModelsSection` (never passed props, superseded by the shared model rows)
|
||||
- **Media Providers**: creating a custom embedding node reports that a key still has to be added, instead of claiming a key was saved; the edit dialog keeps its API Key + Check affordance since a stored key already exists there
|
||||
- **Build**: self-host Inter instead of fetching it through `next/font/google` at build time — a Docker / mirrored builder with no route to `fonts.googleapis.com` failed the entire image build on `Failed to fetch 'Inter' from Google Fonts`. The seven `@font-face` rules and their `unicode-range`s copy what `next/font` emitted (a `latin`-only file would have dropped Vietnamese diacritics) and the latin subset is preloaded as before, so rendered metrics are unchanged
|
||||
|
||||
# v0.5.69 (2026-09-05)
|
||||
|
||||
## Features
|
||||
- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities
|
||||
- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`)
|
||||
- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys
|
||||
- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819)
|
||||
- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through
|
||||
- **CLI tools**: replace Copilot MITM with VS Code extension setup guide
|
||||
- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace
|
||||
|
||||
## Fixes
|
||||
- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792)
|
||||
- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813)
|
||||
- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797)
|
||||
- **Dashboard**: dynamic mode label for local/remote detection (#3801)
|
||||
- **Codex**: format reset credit API errors cleanly (#3778)
|
||||
- **Security**: guard cowork MCP tools probe against SSRF (#3783)
|
||||
- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800)
|
||||
- **Logger**: suppress noisy background token refresh logs
|
||||
- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory
|
||||
|
||||
# v0.5.65 (2026-09-03)
|
||||
|
||||
## Features
|
||||
- **Fetch**: add Ollama Cloud web fetch provider
|
||||
- **Gemini / Antigravity**: add Gemini 3.8 Flash support and bump IDE fingerprint to 2.11.0
|
||||
- **Claude**: add Claude Fable 5.1 support (adaptive thinking with `output_config.effort`), bump Claude Code fingerprint to 2.1.258 for new-model access
|
||||
- **Providers**: add client-side status filter (All / Active / Inactive / No connection) on the Providers dashboard; add max height and scroll for connection list
|
||||
- **Providers & Models**: streamline tokenrouter model catalog down to 22 flagship/newest models and add missing provider icons; refresh Codebuddy-CN catalog (add hy4-preview/hy3/glm-5.3/kimi-k3-1, drop EOL glm-5.0/glm-4.7)
|
||||
- **Models**: capability toggles (vision, reasoning) when adding custom models with upsert and live caps refresh
|
||||
- **CLI tools**: support saving and managing custom API key presets
|
||||
- **Quota**: add usage and rate-limit tracking for Groq via `x-ratelimit-*` headers
|
||||
- **i18n**: complete Indonesian translation (1391 keys)
|
||||
|
||||
## Fixes
|
||||
- **Security**: close SSRF guard bypasses in `ssrfGuard.js` (alternate IPv6 encodings, hostname trailing dots, wildcard DNS resolution check, safe redirect handling) (#3714)
|
||||
- **Model markers**: strip the `[1m]` context marker Claude Code appends to model names (`claude-opus-5[1m]`) preventing model resolution failures (#3690)
|
||||
- **Claude**: drop `server_tool_use` blocks carrying foreign IDs to avoid Anthropic 400 rejections; never anchor cache breakpoints on `defer_loading` tools (#3567)
|
||||
- **Antigravity**: strike-break optimistic quota readings that keep 429ing by blocking the connection+model pair for 15m after 3 strikes (#3681); preserve client identity on model catalog requests (#3414)
|
||||
- **Auth**: protect root `/responses` rewrite requiring API key validation in dashboardGuard
|
||||
- **Chat & Docker**: return 503 Service Unavailable when all credentials are rate-limited; explicitly bundle `node-machine-id` into standalone Docker runtime image
|
||||
- **OpenCode**: route Muse Spark models to `/zen/v1/responses` and declare vision support; filter inactive free model
|
||||
- **Kiro**: preserve inline images as OpenAI-compatible `image_url` parts in OpenAI MITM; remove redundant top-level `systemPrompt` from payload
|
||||
- **Usage**: read Responses-shape `cached_tokens` in `extractUsageFromResponse` for non-streaming traffic
|
||||
- **Models**: support single model lookup with provider-prefixed IDs (e.g. `cc/claude-sonnet-5`)
|
||||
- **Translator**: route Gemini thinking through `reasoning_effort` on OpenAI-compatible wire; convert `prefixItems` and ensure array items in Gemini schema sanitizer
|
||||
- **UI**: apply persisted theme before first paint to prevent flash on reload; translate combo vision adapter label
|
||||
|
||||
# v0.5.59 (2026-08-29)
|
||||
|
||||
## Features
|
||||
- **Search**: new web search providers — Antigravity (Google Search grounding
|
||||
on the existing OAuth account pool, citations keyed and merged by URL) and
|
||||
Xquik (X search with `x-api-key` auth, cursor pagination, credit-based
|
||||
usage), both on `POST /v1/search`. Based on #3437 by @Nautilaceae
|
||||
- **Search**: ollama-search and zai-search borrow a chat provider's API key
|
||||
instead of requiring their own connection, driven by a new
|
||||
`credentialFallback` registry field. zai-search later folded into the `glm`
|
||||
provider itself so the web search page shows the shared connection
|
||||
- **Models**: daily background sync of model capabilities from models.dev —
|
||||
modalities keyed by model id (majority of sources must declare one),
|
||||
context/output limits keyed by provider + model, strictly additive and
|
||||
sitting below the hand-written tables. ETag + mtime cache, 60s startup
|
||||
delay, `MODEL_CATALOG_SYNC=off` to disable
|
||||
- **Models**: add GLM-5.3-Flash (1M context, natively multimodal), DeepSeek
|
||||
V4 Vision, Grok 4.5/4.6 (500k context); correct glm-4.6v/4.5v video input
|
||||
and output limits, backfill glm-4.6v on glm-cn
|
||||
- **Usage**: show the Zed plan quota on the dashboard — plan, edit
|
||||
predictions, hosted model requests and billing-cycle reset; unlimited rows
|
||||
render as "N used · Unlimited"
|
||||
- **Usage**: track GPT-5.3-Codex-Spark quota windows (spark_session /
|
||||
spark_weekly) from the Codex usage response (#3431)
|
||||
- **Antigravity**: quota-aware routing — on 409/429 fetch live quota for the
|
||||
exact per-model resetAt and skip only the exhausted account/model pair;
|
||||
report the earliest reset when every account is blocked (#3561)
|
||||
- **Antigravity**: map image `size` to the aspect-ratio model suffix (-WxH);
|
||||
add the Gemini 3.7 Flash tiers to MITM defaultModels so they show up in
|
||||
the dashboard model-mapping table
|
||||
- **Dashboard**: bulk import Grok CLI accounts from JSON — paste an array or
|
||||
drag-drop multiple .json files, all OAuth connections created in a single
|
||||
call, mirroring the codex flow
|
||||
- **CLI tools**: endpoint presets shared across every tool card through one
|
||||
live-resyncing store, instead of per-card localStorage copies that never
|
||||
saw each other's saved endpoints
|
||||
- **Token Saver**: configurable compression timeout (`headroomTimeoutMs`) —
|
||||
the fixed 3000 ms made busy machines time out and send inconsistently
|
||||
compressed bodies, hurting prompt caching
|
||||
- **i18n**: pt-BR expanded to 1132 terms
|
||||
|
||||
## Fixes
|
||||
- **Claude Code**: add Claude Fable 5.1 and advertise Claude Code 2.1.258 in
|
||||
both the request header and billing identity; use its permanent adaptive-thinking
|
||||
mode with `output_config.effort`
|
||||
- **Stream**: record usage when a client closes on the terminal event — the
|
||||
Responses API has no [DONE] sentinel, so codex closed the socket on
|
||||
`response.completed` and cancelled the reader before flush() ran its usage
|
||||
side effects; the tail now lives in a once-guarded finalizeStream(). Also
|
||||
stop logging a disconnect for every completed Responses call
|
||||
- **Stream**: parse the trailing NDJSON line an Ollama stream leaves behind
|
||||
without a closing newline — the final chunk carrying `done_reason` and the
|
||||
token counts was dropped
|
||||
- **Session**: read the Claude Code session id from the
|
||||
`x-claude-code-session-id` header — `metadata.user_id` is dropped by
|
||||
Responses translation, splitting one conversation across several
|
||||
`prompt_cache_key` values and missing the upstream prefix cache
|
||||
- **Usage**: preserve nested `cached_tokens` — the top-level-only read
|
||||
persisted `cached_tokens: 0` for every Responses-format provider (codex,
|
||||
grok-cli, …), billing cache hits at the full input rate
|
||||
- **Usage**: GLM quotas accept CREDIT_LIMIT plans and multi-interval windows
|
||||
(5h session / 7d weekly) instead of overwriting a single "session" key
|
||||
- **Models**: the catalog sync no longer erases its own output — deltas were
|
||||
measured against the previous run's writes (the second run cut `providers`
|
||||
from 20 entries to 5); one vote per provider in the modality tally, ETag
|
||||
restored from file on startup, and the worker thread dropped after the
|
||||
bundler rewrote its path into a module-not-found error
|
||||
- **Executor**: CommandCode returns errors as a `type:"error"` event inside
|
||||
an HTTP 200 NDJSON stream — peek the first events before committing, abort
|
||||
and return a real 4xx/5xx so combo/account fallback triggers instead of
|
||||
streaming the error text as content
|
||||
- **Search**: scope failure locks on the credential-fallback path — a failing
|
||||
search locked `modelLock___all` and took the shared glm key offline for
|
||||
chat as well; locks are now attributed to the connection's owner and
|
||||
scoped to `websearch:<provider>`
|
||||
- **Providers**: connection tests get a 15s AbortSignal timeout instead of
|
||||
hanging and exhausting the browser socket pool; guard undefined provider
|
||||
names on the providers page
|
||||
- **Antigravity**: sanitize competing-client branding via a config-driven
|
||||
rule table (Zed's Claude-agent prompt, opencode → antigravity) — upstream
|
||||
answers 429 Quota Exhausted. Applied in the executor so the shared
|
||||
openai-to-gemini translator leaves gemini/vertex/zed untouched
|
||||
- **MiniMax**: preserve images on the sourceFormat-matched OpenAI transport
|
||||
— MiniMax-M3 resolved a Claude-shaped body posted to the OpenAI endpoint,
|
||||
silently dropping `image_url` blocks (#3418)
|
||||
- **Claude**: decloak tool names in same-format streaming passthrough —
|
||||
OAuth-cloaked names (CLAUDE_TOOL_SUFFIX) leaked to the client and every
|
||||
tool call was rejected as unknown
|
||||
- **Tools**: default a missing `tools[].type` to "custom" on Claude-format
|
||||
requests — strict Anthropic-compatible gateways (MiniMax) reject the
|
||||
request with 400 otherwise
|
||||
- **Translator**: zai thinkingFormat sends the top-level `reasoning_effort`
|
||||
object GLM-5.2+ requires — every GLM-5.x request ran at the model default
|
||||
(max); gated on GLM-5.2+ since older GLM does not read it (#2721)
|
||||
- **RTK**: system prompt injection matches each target wire format
|
||||
(Chat/Responses/Claude/Gemini/Kiro) and is exact-idempotent across retries,
|
||||
so distinct prompts sharing a long prefix are no longer collapsed (#3202).
|
||||
Also set the diagnostic before the silent null return on Responses
|
||||
translation failure so the panel is no longer blank
|
||||
- **OpenCode**: route muse-spark through /zen/v1/responses (it 500s on
|
||||
chat/completions), normalizing the Chat fields the Responses API rejects
|
||||
and clamping max/ultra effort to xhigh
|
||||
- **CLI**: install better-sqlite3 without build tools on Node 22+ (N-API
|
||||
13.0.3 ships per-platform prebuilds, `--ignore-scripts` skips the implicit
|
||||
node-gyp build); Node < 22 stays on 12.6.2, working installs untouched
|
||||
- **CLI tools**: send the API key Codex actually reads —
|
||||
`[model_providers.9router.http_headers]` instead of auth.json (which left
|
||||
every request 401 and clobbered an existing ChatGPT login); subagent model
|
||||
moved to `agents.default_subagent_model`
|
||||
- **OAuth**: refresh Cline tokens with the extension JSON contract
|
||||
- **Dashboard**: clamp the API key mask length — keys shorter than 8 chars
|
||||
threw RangeError and crashed the media-provider detail page
|
||||
- **UI**: wait for the Material Symbols font itself before revealing icons —
|
||||
`document.fonts.ready` resolved before the 4MB woff2 even started loading,
|
||||
leaving icons blank until a second load
|
||||
|
||||
# v0.5.55 (2026-08-14)
|
||||
|
||||
## Features
|
||||
- **Auth**: native SAML 2.0 SSO alongside OIDC — AuthnRequest generation, ACS
|
||||
assertion handling, SP metadata export, admin config test, replay-protected
|
||||
via a `saml_state` cookie matched against `InResponseTo`
|
||||
- **Providers**: add Alibaba Token Plan (`token-plan.ap-southeast-1`) — the
|
||||
fourth Alibaba key type, Singapore-only and OpenAI-compatible transport only
|
||||
- **Providers**: add `glm-5.3` to GLM Coding and GLM (China)
|
||||
- **Providers**: Kimchi accepts API keys as well as OAuth (dual auth), with a
|
||||
working Test Connection for both modes
|
||||
- **Antigravity**: add Gemini 3.7 Flash and its tiered high/medium/low variants
|
||||
(also in the Gemini registry) with pricing and quota tracking
|
||||
- **TTS**: add Fish Audio — model id travels in an HTTP `model` header, voice
|
||||
is a `reference_id` (preset or cloned voice model)
|
||||
- **OpenCode-Go**: route by request format via declared transports instead of
|
||||
forcing every client into `/messages` — Codex/OpenAI clients no longer pay a
|
||||
lossy Responses→OpenAI→Claude double translation. Per-model `supportedFormats`
|
||||
guard; the bespoke executor is gone (its shared `_lastModel` cache could cross
|
||||
auth headers between concurrent requests)
|
||||
- **Usage**: dedup + cache Claude quota calls (120s TTL keyed by access token,
|
||||
in-flight promise dedup, last-good read on soft failure) to stop multiple
|
||||
tabs tripping 429; manual refresh (↻) sends `force=1` to bypass the cache
|
||||
|
||||
## Fixes
|
||||
- **Docker**: ship `sql.js` in the image so the pure-JS DB fallback can start —
|
||||
file tracing carried the package's JS without `dist/sql-wasm.wasm`, so a
|
||||
container with no native driver aborted with ENOENT and never got a database
|
||||
(#3248)
|
||||
- **Usage**: read Gemini `usageMetadata` out of the antigravity `{ response }`
|
||||
envelope — every non-streaming antigravity request logged `IN 0 | OUT 0`
|
||||
(#3260)
|
||||
- **Claude**: re-anchor passthrough cache breakpoints — the client's own
|
||||
`cache_control` markers point at pre-normalization offsets, so the tail was
|
||||
re-cached every request. Last system block and last tool pinned at 1h TTL,
|
||||
last assistant turn at 5m, mid-conversation system messages folded into the
|
||||
neighbouring user turn instead of hoisted into `body.system`
|
||||
- **Combos**: detect images from Hermes and attachment payloads (`images[]`,
|
||||
`experimental_attachments`, message-level `image_url`/`audio_url`, inline
|
||||
`data:` URIs) so the Vision Adapter auto-switch fires for Hermes/Ollama/
|
||||
Vercel AI SDK shapes
|
||||
- **Kiro**: intercept chat via `x-amz-target` — Kiro IDE 1.0.228+ moved
|
||||
`GenerateAssistantResponse` to `POST /` + header, bypassing MITM. Also emit
|
||||
the now-mandatory initial-response frame and map the `auto` model slot
|
||||
- **Kiro**: report real output tokens and stop discarding usable turns
|
||||
- **Qoder**: detect billing blocks at stream start and return a synthetic 403
|
||||
so combo/account fallback triggers instead of leaking the error into chat
|
||||
- **Antigravity**: strip competitive system prompts (Zed IDE's Claude-agent
|
||||
prompt) that Antigravity flags with a 429 Quota Exhausted
|
||||
- **OpenCode**: send the official client fingerprint on free-tier requests so
|
||||
the Console stops classifying traffic as unidentified and rate-limiting it;
|
||||
session id resolves conversation-stable to preserve prompt caching
|
||||
- **Responses**: don't close the message on an empty `tool_calls` array — some
|
||||
providers attach one to every chunk, and the truthy check ended the message
|
||||
on the first content token (#3234)
|
||||
- **Translator**: preserve `prompt_cache_key` when converting chat to responses
|
||||
- **Models**: expose snake_case token limits on `/v1/models`
|
||||
- **Combos**: strip `stream_options` from the Fusion panel fan-out to avoid a
|
||||
DeepSeek 400 (#3024); raise the dashboard model-test probe budget to 1024 and
|
||||
soft-pass reasoning-only responses (#3010)
|
||||
- **Headroom**: the toggle reflects the `headroomEnabled` setting even when the
|
||||
proxy is down — it previously showed OFF while the engine kept calling
|
||||
`/v1/compress`; proxy status stays visible via the status chip
|
||||
- **Hermes**: add the `api_key` parameter to the model block in YAML config
|
||||
- **Providers**: add llm7 to provider test support
|
||||
|
||||
## Docs
|
||||
- **i18n**: add Spanish, French, and Brazilian Portuguese README translations
|
||||
|
||||
## Security
|
||||
- **Real IP**: `x-9r-real-ip` and the Host fallback were trusted from
|
||||
client-controlled headers whenever `custom-server.js` was not in the request
|
||||
path (`npm run start`, `start:bun`), letting a remote caller pose as local to
|
||||
skip API key auth and reach `LOCAL_ONLY_PATHS` (`/api/mcp/*`,
|
||||
`/api/tunnel/enable`, `/api/auth/reset-password`). The server now stamps a
|
||||
per-process `x-9r-peer-token` on every request it sanitizes and only trusts
|
||||
`x-9r-real-ip` behind it — falling back to Host in development and failing
|
||||
closed in production (GHSA-pjm4-8fpg-f9p6). Also fixes IPv6 loopback
|
||||
detection (`::1`, `::ffff:127.0.0.1`) and routes `npm run start` /
|
||||
`start:bun` through `custom-server.js`
|
||||
- **Search**: `resolveBaseUrl()` rejects client-supplied non-public baseUrls
|
||||
(SSRF guard on `/v1/search`)
|
||||
- **Login**: fresh-install remote login with the default password returns 403
|
||||
without issuing a JWT
|
||||
- **Usage**: `/api/usage/request-details` redacts request/response payloads
|
||||
|
||||
# v0.5.50 (2026-08-05)
|
||||
|
||||
## Features
|
||||
- **Providers**: add TokenRouter (300+ models via OpenAI-compatible gateway) with
|
||||
exact per-model pricing for 110 models and `reasoning_effort` thinking config
|
||||
- **Providers**: add Self-hosted STT / TTS / Embedding — point 9Router at your own
|
||||
OpenAI-compatible speech and embedding servers (whisper.cpp, faster-whisper,
|
||||
Kokoro-FastAPI, llama-server, vLLM, Infinity). Unlike the named cloud providers
|
||||
these read `baseUrl` per connection, so one provider can front several machines
|
||||
- **Combos**: default-enable vision/audio capacity adapter (auto-routes to a
|
||||
vision/audio-capable model when the target lacks that capability, falling back
|
||||
to `oc/mimo-v2.5-free`), wired into chat handler routing
|
||||
- **Endpoint**: auto-provision a "Default Key" for first-time users so `/v1`
|
||||
works without a manual dashboard step
|
||||
- **Codex**: support GPT-5.6 Max/Ultra reasoning-level overrides (cx/ routes only)
|
||||
- **Qoder**: support PAT (Personal Access Token) connections end-to-end, alongside
|
||||
OAuth device flow
|
||||
- **CLI tools**: add OpenDesign (manalkaff/opendesign) support
|
||||
- **Headroom**: report effective payload savings (tool schema/history bytes broken
|
||||
out, byte-savings % reflects actual outbound reduction)
|
||||
- **Ollama**: Cloud quota tracker (session + weekly) + proactive background OAuth
|
||||
token refresh scheduler for all providers
|
||||
|
||||
## Fixes
|
||||
- **Providers**: remove Qwen (OAuth flow stopped working reliably)
|
||||
- **Passthrough**: detect codex-tui/Codex Desktop as native Codex client — they
|
||||
were falling through to the translator and losing fields like `reasoning.summary`
|
||||
- **OAuth**: scope antigravity header fixes to loadCodeAssist/onboardUser only
|
||||
- **OAuth**: keep `open` external in the build so xAI/Grok token refresh works on
|
||||
Windows
|
||||
- **OAuth**: declare missing `searchParams` in register-session handler (was a
|
||||
500 instead of JSON on error)
|
||||
- **DB**: `ENABLE_REQUEST_LOGS` env var now overrides the UI setting correctly;
|
||||
observability defaults to off (opt-in)
|
||||
- **Translator**: preserve Codex Responses Lite tool use across chat-native
|
||||
OpenAI-compatible providers
|
||||
- **Translator**: don't drop image-only user messages in `prepareClaudeRequest`
|
||||
- **Translator**: drop JSON Schema keywords Gemini rejects (`uniqueItems`,
|
||||
`contains`, `multipleOf`, `unevaluatedProperties`, `unevaluatedItems`,
|
||||
`contentSchema`)
|
||||
- **Claude**: remove global header cache that leaked one client's identity
|
||||
headers onto another client/account sharing the server; gate `anthropic-beta`
|
||||
by model instead
|
||||
- **Antigravity**: drop retired Gemini 3.0 quota tiers, show Gemini 3.6 Flash
|
||||
usage bars
|
||||
- **Cloudflare AI**: declare API key authentication (dashboard showed "No
|
||||
connections" despite an active key)
|
||||
- **GitHub Copilot**: hold monthly-exhausted accounts until UTC month reset
|
||||
instead of only cooling down 120s
|
||||
- **CodeBuddy**: dodge Tencent CN content filter, add usage tracking, normalize
|
||||
codebuddy-intl messages
|
||||
- **Usage**: stop losing cached prompt tokens in the forced-SSE→JSON path
|
||||
- **Grok CLI**: display the public subscription tier from the OAuth token claim
|
||||
- **Providers**: count apikey connections for Ollama free-tier card; free-tier/
|
||||
apikey providers without `authModes` now default to apikey (were treated
|
||||
oauth-only)
|
||||
- **Build**: include static/public assets in standalone output (login page hung
|
||||
on 404s when run via PM2)
|
||||
- **Server**: support IntelliJ IDEA OpenAI-compatible clients over HTTP (h2c
|
||||
upgrade handling)
|
||||
- **Auth**: redirect already-logged-in sessions away from `/login`
|
||||
- **CLI tools**: enable Apply button for dynamic OpenAI/Anthropic-compatible
|
||||
provider connections
|
||||
- **CLI**: include complete API artifacts in the CLI package
|
||||
- **TTS**: a bare self-hosted model name is the MODEL, not the voice — `kokoro`
|
||||
was parsed as a voice against a default model, 404ing or synthesising with the
|
||||
wrong one
|
||||
- **Embeddings**: self-hosted embeddings no longer fall back to `api.openai.com`
|
||||
when a connection has no `baseUrl` — that silently sent the input text and API
|
||||
key to OpenAI under a provider named "Self-hosted"
|
||||
- **Embeddings**: an adapter that rejects a misconfigured connection now returns
|
||||
400 with the reason instead of escaping the handler uncaught
|
||||
- **Embeddings**: bound the upstream fetch with `FETCH_CONNECT_TIMEOUT_MS` — an
|
||||
endpoint that drops packets never returns headers, so the request previously
|
||||
hung indefinitely
|
||||
|
||||
## Docs
|
||||
- **i18n**: fix port typo, add RTK Token Saver feature descriptions
|
||||
|
||||
# v0.5.45 (2026-07-30)
|
||||
|
||||
## Features
|
||||
|
||||
- **TTS**: add Xiaomi MiMo text-to-speech (preset voices 冰糖/茉莉/苏打/白桦/Mia/Chloe/Milo/Dean, style control, language hint dropdown with Auto-detect, i18n for Style label/placeholder)
|
||||
- **Providers**: add Poolside (OpenAI-compatible)
|
||||
- **Providers**: add api-airforce, baidu, bazaarlink, bluesminds, kilo-gateway, llm7, morph, sambanova, tencent
|
||||
- **OAuth**: zed / trae / windsurf providers + harden callback proxies
|
||||
@@ -24,7 +474,6 @@
|
||||
- **Usage**: SuperGrok weekly pool via gRPC-web
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Refresh**: rotate `refresh_token` between retry attempts
|
||||
- **Kiro**: canonicalize tool history and route API keys correctly
|
||||
- **Kiro**: normalize dashboard thinking intensity models
|
||||
@@ -39,21 +488,18 @@
|
||||
- **Dashboard**: flex quota rows, thin global scrollbars, no hidden-row overflow
|
||||
|
||||
## Docs
|
||||
|
||||
- **i18n**: expand pt-BR translation to 986 terms
|
||||
- README: Indonesian translation
|
||||
|
||||
# v0.5.40 (2026-07-20)
|
||||
|
||||
## Features
|
||||
|
||||
- **i18n**: add Khmer (km) translations
|
||||
- **CLI tools**: configure Grok Build subagent models
|
||||
- **Kimi**: merge OAuth into dual-auth provider, add K3 / K2.7 models
|
||||
- **Dashboard**: ProviderTopology flow animation
|
||||
|
||||
## Fixes
|
||||
|
||||
- **DB**: resolve better-sqlite3 parameter binding crash
|
||||
- **Translator**: pass `service_tier` through OpenAI → Responses conversion
|
||||
- **Kiro**: map GPT-5.6 reasoning effort fields
|
||||
@@ -64,10 +510,10 @@
|
||||
- **Cursor**: HTTP/2 AgentService support + version bump 3.12.17
|
||||
- **Dashboard**: cut duplicate API/icon spam, lazy-load provider assets
|
||||
|
||||
|
||||
# v0.5.35 (2026-07-16)
|
||||
|
||||
## Features
|
||||
|
||||
- **xAI**: Grok Imagine video generation (`/v1/videos`) + CLI
|
||||
- **CLI tools**: Grok Build setup — choose separate main/general-purpose/explore/plan models and preserve each model's context window
|
||||
- **GitHub Copilot**: route Claude models through Copilot's native `/v1/messages`
|
||||
@@ -78,7 +524,6 @@
|
||||
- **i18n**: Thai (th) + Persian (fa) translations / README
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Providers**: bulk-add API keys no longer overwrite existing keys (gap-fill `Key N`)
|
||||
- **Anthropic**: lowercase `anthropic-version` header to prevent duplication on `/v1/messages`
|
||||
- **Alicode-intl**: use DashScope compatible-mode endpoint so standard keys work
|
||||
@@ -91,17 +536,14 @@
|
||||
- **Translator**: strip `client_metadata` when converting openai-responses → openai
|
||||
|
||||
## Improvements
|
||||
|
||||
- **Perf**: skip inactive background services on startup
|
||||
|
||||
## Docs
|
||||
|
||||
- README: Persian YouTube tutorial
|
||||
|
||||
# v0.5.30 (2026-07-10)
|
||||
|
||||
## Features
|
||||
|
||||
- **Perplexity**: add Agent API provider (#2492)
|
||||
- **Grok CLI**: add Grok CLI / Grok Build provider with OAuth device-code flow (#2502)
|
||||
- **Featherless**: add OpenAI-compatible provider presets
|
||||
@@ -113,7 +555,6 @@
|
||||
- **Proxy-Pools**: auto-rotate strategy for no-auth providers (#2409)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Cloudflare-AI**: support accountId in bulk key import (#2449)
|
||||
- **DB**: backup on schema change, MCP child cleanup, codex models, usage providers OOM
|
||||
- **Codex**: avoid bare-email OAuth dedup (#2477)
|
||||
@@ -130,7 +571,6 @@
|
||||
- **Pricing**: update Claude/Codex model rates and add new models
|
||||
|
||||
## Improvements
|
||||
|
||||
- **i18n(zh-CN)**: complete Chinese translations for all UI strings (#2436)
|
||||
- **API**: caching for tunnel and version status endpoints
|
||||
- **Perf**: faster dev startup and lighter bundle
|
||||
@@ -138,14 +578,12 @@
|
||||
# v0.5.20 (2026-07-07)
|
||||
|
||||
## Features
|
||||
|
||||
- **Thinking**: per-model thinking level picker on provider page — appends `(level)` suffix to copied model names for forced reasoning effort across all formats (openai, claude, gemini, deepseek, kimi, qwen, zai, minimax, hunyuan, step)
|
||||
- **RTK**: add JS-native git-log filter (#2423)
|
||||
- **Caveman**: add targeted upstream-aligned style rules (#2424)
|
||||
- **i18n**: add Farsi (fa) language support (#2385)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Thinking**: strip `(level)` suffix from upstream `body.model` so providers no longer reject requests
|
||||
- **Translator**: preserve developer instructions in openai-responses conversion (#2434)
|
||||
- **count_tokens**: count structured Anthropic blocks (#2419)
|
||||
@@ -159,14 +597,12 @@
|
||||
# v0.5.18 (2026-07-03)
|
||||
|
||||
## Features
|
||||
|
||||
- **Usage**: track cached tokens + correct input/output/cache cost (#2209) — hodtien
|
||||
- **Codex**: show reset credit expiry details (#2290) — Rafli Ahmad Zulfikar
|
||||
- **NVIDIA**: add new models and capabilities — decolua
|
||||
- **ClinePass**: add provider support — sternelee
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Usage**: dedupe streaming request-details log entries — Qin Li
|
||||
- **Claude**: drop foreign thinking signatures in passthrough — decolua
|
||||
- Prevent non-SSE stream pipe crash and cross-IdP account overwrites (#2244) — KunN-21
|
||||
@@ -183,13 +619,11 @@
|
||||
# v0.5.15 (2026-06-29)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Kimchi OAuth provider — Nant361
|
||||
- Refine Qwen vision/video + thinking model patterns — decolua
|
||||
- Opt-in Codex auto-ping quota keep-alive — Emirhan
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Responses**: handle response.done terminal events (#2142) — rifuki
|
||||
- **Headroom**: skip unsafe responses tool history (#2132) — Sutarto Jordan Chrisfivo
|
||||
- **Translator**: map mid-conversation system message to user (claude→openai) — decolua
|
||||
@@ -206,7 +640,6 @@
|
||||
# v0.5.12 (2026-06-26)
|
||||
|
||||
## Features
|
||||
|
||||
- Add token-saver dashboard page — decolua
|
||||
- Add bulk delete for provider connections — teddytkz
|
||||
- Resolve GitHub Copilot model catalog from upstream — caiqinzhou
|
||||
@@ -215,7 +648,6 @@
|
||||
- Overhaul Blackbox provider catalog + WebUI test support — suryacagur
|
||||
|
||||
## Fixes
|
||||
|
||||
- Provider thinking compatibility (DeepSeek/Gemini) — Mink Nguyen
|
||||
- Stop double-counting streaming usage at source — decolua
|
||||
- Usage logging dedupe to reduce stats churn — Mink Nguyen
|
||||
@@ -244,13 +676,11 @@
|
||||
# v0.5.8 (2026-06-21)
|
||||
|
||||
## Features
|
||||
|
||||
- **Antigravity**: native image generation support (image models tagged kind:image, hiển thị trong media-providers UI)
|
||||
- **CodeBuddy CN**: API key auth + credit quota tracker
|
||||
- **CodeBuddy CN**: short model prefix alias "cbcn"
|
||||
|
||||
## Fixes
|
||||
|
||||
- **MiniMax-M3**: enable vision capability
|
||||
- **Headroom**: support Docker sidecar proxy
|
||||
- **Antigravity**: image executor fixes
|
||||
@@ -264,14 +694,12 @@
|
||||
# v0.5.6 (2026-06-20)
|
||||
|
||||
## Features
|
||||
|
||||
- **Ponytail**: minimalist code generation feature
|
||||
- **Headroom**: proxy lifecycle management + dashboard UI (one-click start/stop, install detection, status probing, token saver, claude↔openai shape conversion)
|
||||
- **CodeBuddy CN**: new OAuth provider (copilot.tencent.com) — 15-model catalog, /v2 inference, forced streaming, OpenAI-style reasoning
|
||||
- **OpenCode-Go**: align models with official endpoints; route Qwen 3.7 MiniMax via /v1/messages, GLM/Kimi/DeepSeek/MiMo via /chat/completions
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Anthropic-compatible validation**: use POST /v1/messages (GET /models not spec, false "invalid" for valid keys)
|
||||
- **CLI tools**: tolerate JSONC configs in all 8 settings routes (opencode, openclaw, kilo, droid, cowork, copilot, claude, cline)
|
||||
- **Gemini/Antigravity**: preserve 'pattern' in tool schema translation (glob/grep)
|
||||
@@ -282,7 +710,6 @@
|
||||
# v0.5.4 (2026-06-18)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Kiro**: honor thinking effort budgets
|
||||
- **AG/Kiro/Xiaomi**: provider fixes
|
||||
- **Combo/Fusion**: flatten tool history in panel calls to prevent 503
|
||||
@@ -292,7 +719,6 @@
|
||||
# v0.5.2 (2026-06-17)
|
||||
|
||||
## Features
|
||||
|
||||
- **Combo Fusion strategy** — fans the prompt out to all member models in parallel, then a configurable judge model synthesizes one final answer (quorum-grace, anonymized sources, graceful degradation)
|
||||
- **Per-combo strategy selector** — pick `fallback` / `round-robin` / `fusion` / `capacity` per combo (replaces the old round-robin toggle), with a judge picker for fusion
|
||||
- **Capacity auto-switch** — reorders models per request so images/PDFs route to capable models first
|
||||
@@ -300,7 +726,6 @@
|
||||
- **Claude auto-ping** — warms the 5h quota window right after reset so a fresh window starts immediately (per-connection toggle)
|
||||
|
||||
## Fixes
|
||||
|
||||
- **Claude 429**: stop hammering the OAuth usage endpoint — cache resetAt, throttle quota refresh to 3 min, cool down after a 429 (chat unaffected)
|
||||
- **Usage logs always empty**: missing `await` on `getAdapter()` in `getRecentLogs` made `/api/usage/logs` & `/api/usage/request-logs` return nothing
|
||||
- **Executors**: strip params unsupported by the provider/model (drops deprecated `temperature` for claude-opus-4 → Anthropic 400)
|
||||
@@ -312,13 +737,11 @@
|
||||
- **Security**: SSRF hardening on web fetch
|
||||
|
||||
## Internal
|
||||
|
||||
- Large **open-sse / translator refactor** (~40 commits): unified provider/model registry (LiteLLM-style `models[]` + `kind` field, 100 co-located registry files), single-sourced media/OAuth/refresh/token URLs, registry-based dispatch for usage & token-refresh, DRY translator concerns (buildUsage, encodeDataUri, finishReasonMap, chunkBuilder, reasoningDelta…), ESM-safe registry init, large-file splits, dead-code removal, and golden/no-regression test gates
|
||||
|
||||
# v0.4.80 (2026-06-13)
|
||||
|
||||
## Features
|
||||
|
||||
- Vercel AI Gateway: support embeddings, images and credit usage (#1183)
|
||||
- Add MiMo Free no-auth provider (#1789)
|
||||
- Vertex: support ADC `authorized_user` credential
|
||||
@@ -327,7 +750,6 @@
|
||||
- Kiro: enable multi-endpoint failover for GenerateAssistantResponse (#1722)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Security: re-auth on DB export/import + SSRF guard on web fetch
|
||||
- Auth: real client IP rate-limiting + remote default-password guard
|
||||
- Cerebras/Mistral: strip unsupported `client_metadata` from downstream requests (#1742)
|
||||
@@ -346,13 +768,11 @@
|
||||
- Dashboard: show provider node name instead of connection name in topology (#1770) + show explicit `kind="llm"` combos on combos page (#1684)
|
||||
|
||||
## Docs
|
||||
|
||||
- README: add Indonesian 9Router tutorial video (#1709)
|
||||
|
||||
# v0.4.71 (2026-06-06)
|
||||
|
||||
## Features
|
||||
|
||||
- Caveman: add wenyan classical Chinese levels and sync upstream prompts; locale-based visibility on endpoint page
|
||||
- i18n: endpoint exposure notice across multiple languages + Russian README
|
||||
- Antigravity: add gemini-3.5-flash-extra-low (Low) model
|
||||
@@ -361,7 +781,6 @@
|
||||
- MiniMax: add MiniMax-M3 + update Quota Tracker coding/CN (#1631)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Codex: harden streaming timeouts (stall/connect raised to 60s, configurable per-provider), accept `response.done` event, and always emit a terminal `response.failed` + `[DONE]` for Responses passthrough when a stream closes, stalls, or aborts before a terminal event — prevents codex clients from hanging (#1648, #1680, #1688, #1618)
|
||||
- Codex: durable OAuth refresh lifecycle (#1664)
|
||||
- Tunnel: skip virtual interfaces to prevent false netchange watchdog
|
||||
@@ -375,25 +794,21 @@
|
||||
- Model-test: route image/STT probes to their real endpoints, harden STT ping; add opencode-go + xiaomi-tokenplan to connection test (#1576, #1628)
|
||||
|
||||
## Improvements
|
||||
|
||||
- Dashboard: reorganize menu actions across sidebar/header/profile
|
||||
- Translator: add data-driven coverage, bug-exposing cases, and real provider smoke tests
|
||||
|
||||
# v0.4.66 (2026-05-29)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Qoder provider: device-flow OAuth, COSY signing, WAF-bypass body encoding, live model catalog, dashboard quota tracker, 11 models (#1372)
|
||||
- Add new models: Claude Opus 4.8 (Claude Code), GPT 5.4 Mini (Codex)
|
||||
|
||||
## Fixes
|
||||
|
||||
- DeepSeek thinking mode: echo `reasoning_content` back on follow-up/tool-call turns so OpenCode-free and custom providers no longer 400 with "reasoning_content must be passed back" (#1543)
|
||||
- Reasoning injector: match deepseek/kimi model ids case-insensitively (covers custom providers using capitalized model names)
|
||||
- OpenCode suggested-models: include free models without the `-free` suffix, e.g. `big-pickle` (#1535)
|
||||
|
||||
## Improvements
|
||||
|
||||
- Codex: trim sunset models, keep gpt-5.5 / gpt-5.4 / gpt-5.3-codex family, add gpt-5.4-mini
|
||||
- volcengine-ark: refresh model list (add DeepSeek-V4-Flash/Pro, drop EOL entries)
|
||||
- Lower stream stall timeout 35s → 30s for faster hang detection
|
||||
@@ -401,21 +816,18 @@
|
||||
# v0.4.63 (2026-05-26)
|
||||
|
||||
## Fixes
|
||||
|
||||
- GitHub Copilot: never route Gemini/Claude models to the `/responses` endpoint; prevents misleading "does not support Responses API" 400s (#1062)
|
||||
- proxyFetch: restore missing `Readable` import causing runtime `ReferenceError` in DNS-bypass fetch path
|
||||
|
||||
## Improvements
|
||||
|
||||
- Lower stream stall timeout from 60s → 35s for faster hang detection
|
||||
|
||||
# v0.4.62 (2026-05-26)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Codex: auto-retry when upstream drops mid-stream (no more hangs)
|
||||
- Codex: fix random 400/404 errors, tool-calling failures, and unstable prompt cache
|
||||
- MITM: support Antigravity 2.x
|
||||
- MITM: support Antigravity 2.x
|
||||
- Sanitize Read tool args to prevent retry loops from non-Anthropic models (#1144)
|
||||
- Implement json_schema fallback for OpenAI-compatible providers without native Structured Output (#1343)
|
||||
- Strip empty Read pages argument in OpenAI-to-Claude translator (#1354)
|
||||
@@ -424,30 +836,25 @@
|
||||
- Gemini CLI: reuse stored OAuth project IDs for quota checks and show clearer setup guidance when the project is missing (#1271, #1428)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Cloudflare Workers proxy deployer and pool integration (#1360)
|
||||
- Add Deno Deploy relays support and improved proxy pools dashboard layout (#1437)
|
||||
|
||||
## Improvements
|
||||
|
||||
- Refactor Tunnel into dedicated Cloudflare and Tailscale manager modules
|
||||
- Refactor tokenRefresh service with in-flight dedup to prevent refresh_token_reused errors
|
||||
|
||||
# v0.4.59 (2026-05-21)
|
||||
|
||||
## Fixes
|
||||
|
||||
- OAuth: fix login flow on Windows
|
||||
|
||||
# v0.4.58 (2026-05-21)
|
||||
|
||||
## Features
|
||||
|
||||
- xAI Grok provider (OAuth, API key, image)
|
||||
- Provider limits: paginated accounts with page size controls
|
||||
|
||||
## Fixes
|
||||
|
||||
- Tailscale: fix connection status on Windows (#1300)
|
||||
- Tunnel: fix false "checking" when tunnel URL is reachable
|
||||
- Stream: fix pipe errors on client disconnect/abort
|
||||
@@ -455,13 +862,11 @@
|
||||
# v0.4.55 (2026-05-18)
|
||||
|
||||
## Features
|
||||
|
||||
- Xiaomi MiMo Token Plan: region selector (Singapore / China / Europe) — keys are cluster-specific
|
||||
- Antigravity: risk confirmation dialog before first connection
|
||||
- Gemini CLI: surface upstream retry delay on 429 errors
|
||||
|
||||
## Fixes
|
||||
|
||||
- MITM: cannot kill process on macOS under sudo (lsof not found in PATH)
|
||||
- Stream: false-positive stall timeout on Claude reasoning / Kiro responses
|
||||
- Tunnel: cannot re-enable after disable (stuck state)
|
||||
@@ -470,19 +875,16 @@
|
||||
- Antigravity OAuth: metadata now matches the official client
|
||||
|
||||
## Improvements
|
||||
|
||||
- Gemini CLI: bump engine to 0.34.0
|
||||
- Re-hide `qwen` (OAuth EOL) and `iflow` (not ready) providers
|
||||
|
||||
# v0.4.52 (2026-05-17)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Vercel AI Gateway provider support (#1183)
|
||||
- rtk: Kiro format tool result compression — handle conversationState.history & currentMessage, preserve error results, ~13.6% savings (#1194)
|
||||
|
||||
## Fixes
|
||||
|
||||
- openclaw: normalize agent.model object form `{primary, fallbacks}` before .startsWith → fix TypeError & 'not configured' status (#1216)
|
||||
- Usage Details pagination: stay inside mobile viewport <640px (#1218)
|
||||
- Fix test model error
|
||||
@@ -492,7 +894,6 @@
|
||||
# v0.4.50 (2026-05-16)
|
||||
|
||||
## Fixes
|
||||
|
||||
- Fix duplicate tray icon on macOS when hiding to tray
|
||||
- Fix tray not showing in background mode on macOS
|
||||
- Fix hide to tray broken on Windows/Linux
|
||||
@@ -501,13 +902,11 @@
|
||||
# v0.4.49 (2026-05-16)
|
||||
|
||||
## Features
|
||||
|
||||
- Add Kiro provider support: full request/response translation, live model listing, reasoning content support
|
||||
- Add `buildOutput` RTK filter with autodetect for npm/yarn/cargo build logs
|
||||
- Add MITM warning notification in tray and dashboard
|
||||
|
||||
## Improvements
|
||||
|
||||
- Add modalities (input/output) to model configuration for OpenCode
|
||||
- Fix tray hide-to-tray: keep current process alive instead of spawning detached child (fixes macOS NSStatusItem ghost icon)
|
||||
- Fix tray kill: graceful shutdown with SIGTERM/SIGKILL escalation
|
||||
@@ -516,11 +915,9 @@
|
||||
- Update i18n across 32 languages
|
||||
|
||||
## Fixes
|
||||
|
||||
- Fix model check (test-models) blocked by dashboardGuard: pass machineId-based CLI token in internal self-calls
|
||||
|
||||
# v0.4.46 (2026-05-15)
|
||||
|
||||
## Breaking Changes
|
||||
|
||||
- Tunnel public URL changed — old tunnel links no longer work, please reconnect to get the new URL
|
||||
|
||||
11
CLAUDE.md
11
CLAUDE.md
@@ -88,4 +88,15 @@ Pre-translate hooks that compress `tool_result` content in-place to cut tokens.
|
||||
- `custom-server.js` wraps the Next standalone server to derive client IP from the TCP socket and strip attacker-controlled `X-Forwarded-For` — trusting forwarding headers only from a loopback reverse proxy. Preserve this when touching request/IP/rate-limit code.
|
||||
- Security-sensitive env: `JWT_SECRET` (session cookie), `INITIAL_PASSWORD` (default `123456` — must override), `API_KEY_SECRET`, `MACHINE_ID_SALT`. Full env contract in `.env.example` and ARCHITECTURE.md's env matrix.
|
||||
- Binary/protobuf upstreams (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — they're handled inside their own executor, not the translator.
|
||||
- **Security-first on PRs**: Security is the top priority when reviewing or creating PRs. Audit authentication, credential/token storage & leaks, header manipulation (`X-Forwarded-For`), and SSRF risks before functional logic. Always include explicit security warnings/notes when reporting PR reviews or changes to the user.
|
||||
- Versioning: root and `cli/` are versioned independently; changes are logged in `CHANGELOG.md`. Commit style is Conventional Commits (`fix(translator): …`, `feat(...)`).
|
||||
|
||||
<!-- BEGIN:nextjs-agent-rules -->
|
||||
|
||||
# This is NOT the Next.js you know
|
||||
|
||||
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` (resolved from this file's directory; in monorepos the `next` package may not be visible from the repo root) before writing any code. Heed deprecation notices.
|
||||
|
||||
This block is written and re-added by `next dev` — verify at `node_modules/next/dist/server/lib/generate-agent-files.js`. Removing it from a diff only re-creates the uncommitted change; committing it with your work keeps the tree clean.
|
||||
|
||||
<!-- END:nextjs-agent-rules -->
|
||||
|
||||
67
DOCKER.md
67
DOCKER.md
@@ -100,6 +100,12 @@ docker rm -f 9router
|
||||
# re-run the quick start command
|
||||
```
|
||||
|
||||
To pin a specific version instead of following `latest`, use a numbered image tag:
|
||||
|
||||
```bash
|
||||
docker pull decolua/9router:0.5.81
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
# 🛠 For Developers
|
||||
@@ -107,7 +113,7 @@ docker rm -f 9router
|
||||
## Build image locally (test)
|
||||
|
||||
```bash
|
||||
cd app && docker build -t 9router .
|
||||
docker build -t 9router .
|
||||
|
||||
docker run --rm -p 20128:20128 \
|
||||
-v "$HOME/.9router:/app/data" \
|
||||
@@ -115,18 +121,67 @@ docker run --rm -p 20128:20128 \
|
||||
9router
|
||||
```
|
||||
|
||||
The Dockerfile uses the official Alpine and npm registries by default. Regional mirrors can be supplied when needed:
|
||||
|
||||
```bash
|
||||
docker build \
|
||||
--build-arg ALPINE_MIRROR=mirrors.aliyun.com \
|
||||
--build-arg NPM_REGISTRY=https://registry.npmmirror.com/ \
|
||||
-t 9router .
|
||||
```
|
||||
|
||||
## Publish (automatic via CI)
|
||||
|
||||
Push a git tag `v*` → GitHub Actions builds multi-platform (amd64+arm64) and pushes to:
|
||||
- `ghcr.io/decolua/9router:v{version}` + `:latest`
|
||||
- `decolua/9router:v{version}` + `:latest`
|
||||
Push a Docker-safe semver git tag `vX.Y.Z` (or a prerelease such as `vX.Y.Z-rc.1`) → GitHub Actions builds `linux/amd64` and `linux/arm64` on native runners, health-checks each platform image, verifies the resulting manifest and `/api/health`, then publishes:
|
||||
|
||||
- `ghcr.io/decolua/9router:X.Y.Z` + `:latest`
|
||||
- `decolua/9router:X.Y.Z` + `:latest`
|
||||
|
||||
The `v` prefix is used only for the git tag; image tags omit it. A stable tag push promotes `latest`, but a prerelease tag such as `vX.Y.Z-rc.1` publishes only its numbered image by default. Prereleases require an explicit manual `promote_latest` opt-in. Promotion happens only after both native platform builds, both platform health checks, manifest inspection, and the resolved-manifest smoke test succeed. A failed or timed-out platform build therefore cannot move `latest`.
|
||||
|
||||
The workflow rejects SemVer build metadata such as `v1.2.3+build.7` because the `+` form is not a valid Docker image tag. The git tag and both `package.json` versions must match exactly.
|
||||
|
||||
```bash
|
||||
# Use scripts/release.js (recommended)
|
||||
node scripts/release.js "Release title" "Notes"
|
||||
|
||||
# Or manually
|
||||
git tag v0.4.x && git push origin v0.4.x
|
||||
git tag v0.5.81 && git push origin v0.5.81
|
||||
```
|
||||
|
||||
Workflow: `app/.github/workflows/docker-publish.yml`
|
||||
To republish an existing tag, run the `Build and Push Docker Image` workflow manually and provide the exact tag, for example `v0.5.81`, in the `release_tag` input. Manual runs publish the numbered tag but leave `latest` unchanged by default:
|
||||
|
||||
```text
|
||||
release_tag: v0.5.81
|
||||
promote_latest: false
|
||||
```
|
||||
|
||||
The `promote_latest` checkbox is an explicit opt-in for changing `latest`. Use it when a deliberate rollback or recovery should make that version the current default:
|
||||
|
||||
```text
|
||||
release_tag: v0.5.75
|
||||
promote_latest: true
|
||||
```
|
||||
|
||||
Numbered image tags are mutable because a republish can replace their manifest. For a deployment that must be immutable, pin the image digest instead:
|
||||
|
||||
```bash
|
||||
docker pull decolua/9router@sha256:<verified-digest>
|
||||
```
|
||||
|
||||
The release workflow runs `/api/health` on each native `amd64` and `arm64` platform image before it uploads the digest artifact or assembles the multi-platform manifest. It then runs a second health check against the resolved version manifest before any requested `latest` promotion.
|
||||
|
||||
During recovery, the selected tag remains the application source while the Dockerfile from the workflow revision is used, so an older tag can be rebuilt with the current publishing fixes.
|
||||
|
||||
The workflow is tag-driven. Creating a git tag does not automatically create a GitHub Release, so the Releases page and the published package/image tags can be at different versions unless a maintainer creates a release separately.
|
||||
|
||||
The upstream repository needs these repository secrets for Docker Hub publishing:
|
||||
|
||||
- `DOCKERHUB_USERNAME`
|
||||
- `DOCKERHUB_TOKEN`
|
||||
|
||||
GHCR publishing uses the workflow's `GITHUB_TOKEN` with package write permission. Forks can publish to their own GHCR namespace, but Docker Hub publication is restricted to the upstream `decolua/9router` repository.
|
||||
|
||||
The optional repository variables `ALPINE_MIRROR` and `NPM_REGISTRY` can override the default package mirrors used by the CI Docker build.
|
||||
|
||||
Workflow: `.github/workflows/docker-publish.yml`
|
||||
|
||||
45
Dockerfile
45
Dockerfile
@@ -1,24 +1,51 @@
|
||||
# syntax=docker/dockerfile:1.7
|
||||
ARG NODE_IMAGE=node:22-alpine
|
||||
ARG ALPINE_MIRROR=dl-cdn.alpinelinux.org
|
||||
ARG NPM_REGISTRY=https://registry.npmjs.org/
|
||||
ARG APP_VERSION=unknown
|
||||
|
||||
FROM ${NODE_IMAGE} AS base
|
||||
ARG ALPINE_MIRROR
|
||||
WORKDIR /app
|
||||
|
||||
FROM base AS builder
|
||||
# Use the official Alpine mirror by default. A repository variable/build arg can
|
||||
# override it for environments that require a regional mirror.
|
||||
RUN if [ "$ALPINE_MIRROR" != "dl-cdn.alpinelinux.org" ]; then \
|
||||
sed -i "s|dl-cdn.alpinelinux.org|${ALPINE_MIRROR}|g" /etc/apk/repositories; \
|
||||
fi
|
||||
|
||||
RUN apk --no-cache upgrade && apk --no-cache add python3 make g++ linux-headers
|
||||
FROM base AS builder
|
||||
ARG NPM_REGISTRY
|
||||
|
||||
RUN apk add --no-cache python3 make g++ linux-headers
|
||||
|
||||
COPY package.json ./
|
||||
RUN --mount=type=cache,target=/root/.npm \
|
||||
npm install
|
||||
npm install \
|
||||
--registry="${NPM_REGISTRY}" \
|
||||
--fetch-retries=5 \
|
||||
--fetch-retry-factor=2 \
|
||||
--fetch-retry-mintimeout=10000 \
|
||||
--fetch-retry-maxtimeout=120000 \
|
||||
--fetch-timeout=300000
|
||||
|
||||
COPY . ./
|
||||
ENV NEXT_TELEMETRY_DISABLED=1
|
||||
# Inter is self-hosted (public/fonts + src/app/fonts-inter.css), so this build needs no
|
||||
# route to fonts.googleapis.com / fonts.gstatic.com — only the npm mirror above is required.
|
||||
RUN npm run build
|
||||
|
||||
FROM ${NODE_IMAGE} AS runner
|
||||
ARG ALPINE_MIRROR
|
||||
ARG APP_VERSION
|
||||
WORKDIR /app
|
||||
|
||||
LABEL org.opencontainers.image.title="9router"
|
||||
RUN if [ "$ALPINE_MIRROR" != "dl-cdn.alpinelinux.org" ]; then \
|
||||
sed -i "s|dl-cdn.alpinelinux.org|${ALPINE_MIRROR}|g" /etc/apk/repositories; \
|
||||
fi
|
||||
|
||||
LABEL org.opencontainers.image.title="9router" \
|
||||
org.opencontainers.image.version="${APP_VERSION}"
|
||||
|
||||
ENV NODE_ENV=production
|
||||
ENV PORT=20128
|
||||
@@ -37,13 +64,19 @@ COPY --from=builder /app/src/mitm ./src/mitm
|
||||
COPY --from=builder /app/node_modules/node-forge ./node_modules/node-forge
|
||||
# Ensure `next` is available at runtime in case tracing did not include it.
|
||||
COPY --from=builder /app/node_modules/next ./node_modules/next
|
||||
# sql.js loads dist/sql-wasm.wasm by path at runtime; tracing only follows JS imports,
|
||||
# so the last-resort DB driver would abort with ENOENT on the missing binary.
|
||||
COPY --from=builder /app/node_modules/sql.js ./node_modules/sql.js
|
||||
# node-machine-id is createRequire-loaded at runtime; tracing omits it.
|
||||
COPY --from=builder /app/node_modules/node-machine-id ./node_modules/node-machine-id
|
||||
|
||||
RUN mkdir -p /app/data && chown -R node:node /app && \
|
||||
mkdir -p /app/data-home && chown node:node /app/data-home && \
|
||||
ln -sf /app/data-home /root/.9router 2>/dev/null || true
|
||||
|
||||
# Fix permissions at runtime (handles mounted volumes)
|
||||
RUN apk --no-cache upgrade && apk --no-cache add su-exec && \
|
||||
# Avoid a full distribution upgrade in the runtime image. It makes builds less
|
||||
# reproducible and is unrelated to installing the runtime entrypoint helper.
|
||||
RUN apk add --no-cache su-exec && \
|
||||
printf '#!/bin/sh\nchown -R node:node /app/data /app/data-home 2>/dev/null\nexec su-exec node "$@"\n' > /entrypoint.sh && \
|
||||
chmod +x /entrypoint.sh
|
||||
|
||||
|
||||
92
README.md
92
README.md
@@ -17,7 +17,7 @@
|
||||
|
||||
[🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Setup](#-setup-guide) • [🌐 Website](https://9router.com)
|
||||
|
||||
[🇻🇳 Tiếng Việt](./i18n/README.vi.md) • [🇨🇳 中文](./i18n/README.zh-CN.md) • [🇯🇵 日本語](./i18n/README.ja-JP.md) • [🇷🇺 Русский](./i18n/README.ru.md) • [🇹🇭 ไทย](./i18n/README.th.md) • [🇮🇷 فارسی](./i18n/README.fa_IR.md) • [🇮🇩 Indonesia](./i18n/README.id-ID.md)
|
||||
[🇧🇷 Português (Brasil)](./i18n/README.pt-BR.md) • [🇻🇳 Tiếng Việt](./i18n/README.vi.md) • [🇨🇳 中文](./i18n/README.zh-CN.md) • [🇯🇵 日本語](./i18n/README.ja-JP.md) • [🇷🇺 Русский](./i18n/README.ru.md) • [🇹🇭 ไทย](./i18n/README.th.md) • [🇮🇷 فارسی](./i18n/README.fa_IR.md) • [🇮🇩 Indonesia](./i18n/README.id-ID.md) • [🇪🇸 Español](./i18n/README.es.md) • [🇫🇷 Français](./i18n/README.fr.md)
|
||||
|
||||
</div>
|
||||
|
||||
@@ -110,7 +110,22 @@ PORT=20128 NEXT_PUBLIC_BASE_URL=http://localhost:20128 npm run dev
|
||||
Production mode:
|
||||
|
||||
```bash
|
||||
# Create Temporary Memory For Build
|
||||
sudo fallocate -l 2G /swapfile_temp
|
||||
sudo chmod 600 /swapfile_temp
|
||||
sudo mkswap /swapfile_temp
|
||||
sudo swapon /swapfile_temp
|
||||
|
||||
export MAKEFLAGS="-j1"
|
||||
export DLIB_NO_GUI_SUPPORT=1
|
||||
export CFLAGS="-mno-avx"
|
||||
|
||||
npm run build
|
||||
|
||||
# Clear temporary swap
|
||||
sudo swapoff /swapfile_temp
|
||||
sudo rm /swapfile_temp
|
||||
|
||||
PORT=20128 HOSTNAME=0.0.0.0 NEXT_PUBLIC_BASE_URL=http://localhost:20128 npm run start
|
||||
```
|
||||
|
||||
@@ -215,7 +230,14 @@ Default URLs:
|
||||
<b>🇻🇳 Tiếng Việt</b><br/>
|
||||
<sub>Hướng Dẫn Setup OpenClaw + 9Router: Tạo Bot Zalo AI Tự Động Từ A-Z<br/>by <a href="https://github.com/tuanminhhole">tuanminhhole</a></sub>
|
||||
</td>
|
||||
<td align="center" width="320"></td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.youtube.com/watch?v=hgnE7MKi3Y4">
|
||||
<img src="https://img.youtube.com/vi/hgnE7MKi3Y4/maxresdefault.jpg" alt="Bye Limit! Cara Bikin Sistem 'AI Unlimited' 100% Gratis Dengan 9Router!
|
||||
" width="300"/>
|
||||
</a><br/>
|
||||
<b>🇮🇩 Indonesia</b><br/>
|
||||
<sub>Bye Limit! Cara Bikin Sistem "AI Unlimited" 100% Gratis Dengan 9Router!<br/>by <a href="https://www.youtube.com/@neptiver">neptiver</a></sub>
|
||||
</td>
|
||||
<td align="center" width="320"></td>
|
||||
<td align="center" width="320"></td>
|
||||
</tr>
|
||||
@@ -285,6 +307,32 @@ Default URLs:
|
||||
<b>Kilo Code</b>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/opendesign.png" width="60" alt="OpenDesign"/><br/>
|
||||
<b>OpenDesign</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/jcode.png" width="60" alt="jcode"/><br/>
|
||||
<b>jcode</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/grok-cli.png" width="60" alt="Grok Build"/><br/>
|
||||
<b>Grok Build</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/devin-cli.png" width="60" alt="Devin CLI"/><br/>
|
||||
<b>Devin CLI</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/deepseek-tui.png" width="60" alt="DeepSeek TUI"/><br/>
|
||||
<b>DeepSeek TUI</b>
|
||||
</td>
|
||||
<td align="center" width="120">
|
||||
<img src="./public/providers/qwen.png" width="60" alt="Qwen Code"/><br/>
|
||||
<b>Qwen Code</b>
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
@@ -441,6 +489,46 @@ Default URLs:
|
||||
<p><i>...and 20+ more providers including Nebius, Chutes, Hyperbolic, and custom OpenAI/Anthropic compatible endpoints</i></p>
|
||||
</div>
|
||||
|
||||
### 🏠 Self-hosted Providers
|
||||
|
||||
For speech and embeddings served from **your own** machine — whisper.cpp,
|
||||
faster-whisper, Speaches, Kokoro-FastAPI, openedai-speech, llama.cpp/llama-server,
|
||||
vLLM, Infinity, text-embeddings-inference, or anything else that speaks the OpenAI
|
||||
shape.
|
||||
|
||||
| Provider | Endpoint used | Typical server |
|
||||
| --- | --- | --- |
|
||||
| **Self-hosted STT** | `/v1/audio/transcriptions` | whisper.cpp, faster-whisper |
|
||||
| **Self-hosted TTS** | `/v1/audio/speech` | Kokoro-FastAPI, openedai-speech |
|
||||
| **Self-hosted Embedding** | `/v1/embeddings` | llama-server, vLLM, Infinity |
|
||||
|
||||
Every other speech provider is a named cloud service with a fixed endpoint. These
|
||||
three read their address from **each connection**, so one provider can front
|
||||
several machines and load-balance across them like any other.
|
||||
|
||||
Set it on the connection as `providerSpecificData.baseUrl`:
|
||||
|
||||
| Provider | Give it | Result |
|
||||
| --- | --- | --- |
|
||||
| Self-hosted STT | the full URL — `http://host:8080/v1/audio/transcriptions` | used as-is |
|
||||
| Self-hosted TTS | the server root — `http://host:8880` | `+ /v1/audio/speech` |
|
||||
| Self-hosted Embedding | the **OpenAI base**, `/v1` included — `http://host:8080/v1` | `+ /embeddings` |
|
||||
|
||||
> **Mind the `/v1` on embeddings.** The adapter appends `/embeddings`, so
|
||||
> `http://host:8080` resolves to `http://host:8080/embeddings` and misses the
|
||||
> OpenAI route — llama-server answers **501**. Give it the same base URL an OpenAI
|
||||
> client would use. A full `.../v1/embeddings` is also accepted, so a value pasted
|
||||
> from a `curl` example works too.
|
||||
|
||||
The API key is not checked by most local servers, but the field must be non-empty:
|
||||
it is what gives the connection a credentials record, and `baseUrl` lives there.
|
||||
Any placeholder works.
|
||||
|
||||
Self-hosted Embedding has **no cloud fallback by design** — a connection saved
|
||||
without a `baseUrl` is reported as a configuration error rather than quietly
|
||||
falling back to `api.openai.com`, which would send your input text and API key to
|
||||
a third party under a provider named "Self-hosted".
|
||||
|
||||
---
|
||||
|
||||
## 💡 Key Features
|
||||
|
||||
1
cli/.gitignore
vendored
1
cli/.gitignore
vendored
@@ -1,2 +1,3 @@
|
||||
app/*
|
||||
node_modules/*
|
||||
.tray-build/
|
||||
|
||||
@@ -111,7 +111,7 @@ Any tool supporting OpenAI/Claude-compatible API works.
|
||||
Full docs, advanced setup, video tutorials & development guide:
|
||||
|
||||
- **GitHub**: https://github.com/decolua/9router
|
||||
- **Full README**: https://github.com/decolua/9router/blob/main/app/README.md
|
||||
- **Full README**: https://github.com/decolua/9router/blob/master/README.md
|
||||
- **Website**: https://9router.com
|
||||
|
||||
---
|
||||
|
||||
@@ -6,7 +6,13 @@ const fs = require("fs");
|
||||
const os = require("os");
|
||||
const path = require("path");
|
||||
|
||||
const BETTER_SQLITE3_VERSION = "12.6.2";
|
||||
// Gate the pinned version by Node major, mirroring src/lib/db/driver.js gating
|
||||
// style: 13.x is N-API and ships per-platform prebuilds inside the package, so
|
||||
// it needs no ABI-specific download. It requires Node >= 22; older runtimes stay
|
||||
// on 12.6.2, which fetches an ABI-specific binary via prebuild-install.
|
||||
const [NODE_MAJOR] = process.versions.node.split(".").map(Number);
|
||||
const USE_NAPI_BUILD = NODE_MAJOR >= 22;
|
||||
const BETTER_SQLITE3_VERSION = USE_NAPI_BUILD ? "13.0.3" : "12.6.2";
|
||||
const SQL_JS_VERSION = "1.14.1";
|
||||
|
||||
function getDataDir() {
|
||||
@@ -45,9 +51,23 @@ function hasModule(name) {
|
||||
return fs.existsSync(path.join(getRuntimeNodeModules(), name, "package.json"));
|
||||
}
|
||||
|
||||
function isGlibcRuntime() {
|
||||
try { return Boolean(process.report?.getReport()?.header?.glibcVersionRuntime); } catch { return true; }
|
||||
}
|
||||
|
||||
// 12.x compiles/downloads into build/Release; 13.x ships prebuilds/<platform>-<arch>.node.
|
||||
function getBetterSqliteBinary() {
|
||||
const root = path.join(getRuntimeNodeModules(), "better-sqlite3");
|
||||
const platform = process.platform === "linux" && !isGlibcRuntime() ? "linuxmusl" : process.platform;
|
||||
return [
|
||||
path.join(root, "build", "Release", "better_sqlite3.node"),
|
||||
path.join(root, "prebuilds", `${platform}-${process.arch}.node`),
|
||||
].find((file) => fs.existsSync(file));
|
||||
}
|
||||
|
||||
function isBetterSqliteBinaryValid() {
|
||||
const binary = path.join(getRuntimeNodeModules(), "better-sqlite3", "build", "Release", "better_sqlite3.node");
|
||||
if (!fs.existsSync(binary)) return false;
|
||||
const binary = getBetterSqliteBinary();
|
||||
if (!binary) return false;
|
||||
try {
|
||||
const fd = fs.openSync(binary, "r");
|
||||
const buf = Buffer.alloc(4);
|
||||
@@ -91,6 +111,7 @@ function runNpmInstall({ cwd, pkgs, extraArgs = [], timeout = 180000 }) {
|
||||
function npmInstall(pkgs, opts = {}) {
|
||||
const cwd = ensureRuntimeDir();
|
||||
const extra = opts.optional ? ["--no-save"] : [];
|
||||
if (opts.ignoreScripts) extra.push("--ignore-scripts");
|
||||
if (!opts.silent) console.log("⏳ Installing SQLite engine (first run)...");
|
||||
const res = runNpmInstall({ cwd, pkgs, extraArgs: extra, timeout: opts.timeout || 180000 });
|
||||
if (!res.ok && !opts.silent) {
|
||||
@@ -129,7 +150,10 @@ function ensureSqliteRuntime({ silent = false } = {}) {
|
||||
return { betterSqlite: true, sqlJs: sqlJsOk };
|
||||
}
|
||||
|
||||
const ok = npmInstall([`better-sqlite3@${BETTER_SQLITE3_VERSION}`], { optional: true, silent });
|
||||
// npm injects an implicit `node-gyp rebuild` for any package carrying a
|
||||
// binding.gyp, which would demand build tools even though 13.x already bundles
|
||||
// the binary — skip scripts so the bundled prebuild is used as-is.
|
||||
const ok = npmInstall([`better-sqlite3@${BETTER_SQLITE3_VERSION}`], { optional: true, silent, ignoreScripts: USE_NAPI_BUILD });
|
||||
return {
|
||||
betterSqlite: ok && hasModule("better-sqlite3") && isBetterSqliteBinaryValid(),
|
||||
sqlJs: sqlJsOk,
|
||||
|
||||
@@ -5,9 +5,17 @@
|
||||
//
|
||||
// We use the maintained `systray2` fork. The original `systray@1.0.5` package
|
||||
// bundles a 2017 x86_64 Go binary whose Mach-O headers are rejected by modern
|
||||
// dyld (macOS 14+), so the tray silently fails to register on Apple Silicon.
|
||||
// dyld (macOS 14+), so it fails to load at all.
|
||||
//
|
||||
// Note that systray2 is NOT an Apple Silicon fix: like its predecessor it ships
|
||||
// only an x86_64 `tray_darwin_release`, and picks it by process.platform with no
|
||||
// process.arch branch, so there is no native slice to select. On arm64 macOS the
|
||||
// tray therefore needs Rosetta 2 and dies with EBADARCH without it. We overlay
|
||||
// our own arm64 build of the same upstream source on top — see ensureArm64TrayBin.
|
||||
const { spawnSync } = require("child_process");
|
||||
const crypto = require("crypto");
|
||||
const fs = require("fs");
|
||||
const os = require("os");
|
||||
const path = require("path");
|
||||
const { getRuntimeDir, getRuntimeNodeModules, runNpmInstall, summarizeNpmError } = require("./sqliteRuntime");
|
||||
|
||||
@@ -15,6 +23,19 @@ const SYSTRAY_PKG = "systray2";
|
||||
const SYSTRAY_VERSION = "2.1.4";
|
||||
const LEGACY_SYSTRAY_PKG = "systray";
|
||||
|
||||
// Pinned `tray-binaries` release rather than `latest`, so the URL is stable and
|
||||
// the artifact can only change by a deliberate re-upload. The workflow's publish
|
||||
// step re-derives this repo from the literal below and refuses to upload
|
||||
// anywhere else, so the integrity gate can't drift from what clients fetch.
|
||||
//
|
||||
// The asset is built by .github/workflows/tray-binaries.yml on a macos-15 runner.
|
||||
// cgo compiles AppKit against the runner's SDK, so this value tracks that image:
|
||||
// when GitHub updates it the sha changes, the workflow refuses to publish, and
|
||||
// this constant must be bumped in the same change as the re-upload.
|
||||
const ARM64_TRAY_URL = "https://github.com/decolua/9router/releases/download/tray-binaries/tray_darwin_arm64";
|
||||
const ARM64_TRAY_SHA256 = "487e3c365aaa1eb6ad295bf3989711e975b52cee07505bf641c8559954881c81";
|
||||
const ARM64_RETRY_COOLDOWN_MS = 24 * 60 * 60 * 1000;
|
||||
|
||||
function hasSystray() {
|
||||
return fs.existsSync(path.join(getRuntimeNodeModules(), SYSTRAY_PKG, "package.json"));
|
||||
}
|
||||
@@ -71,6 +92,124 @@ function ensureRuntimeDir() {
|
||||
return dir;
|
||||
}
|
||||
|
||||
// A thin (non-fat) 64-bit Mach-O stores its magic then cputype, both LE.
|
||||
// CPU_TYPE_ARM64 is CPU_TYPE_ARM | CPU_ARCH_ABI64. Fat/universal binaries use a
|
||||
// different magic and are reported as "not arm64" here, which is fine: we only
|
||||
// ever overlay a thin arm64 build and only need to tell it apart from x86_64.
|
||||
function isArm64MachO(file) {
|
||||
let fd = null;
|
||||
try {
|
||||
fd = fs.openSync(file, "r");
|
||||
const buf = Buffer.alloc(8);
|
||||
fs.readSync(fd, buf, 0, 8, 0);
|
||||
if (buf.readUInt32LE(0) !== 0xfeedfacf) return false;
|
||||
return buf.readUInt32LE(4) === 0x0100000c;
|
||||
} catch {
|
||||
return false;
|
||||
} finally {
|
||||
if (fd !== null) try { fs.closeSync(fd); } catch {}
|
||||
}
|
||||
}
|
||||
|
||||
// systray2 is constructed with copyDir:true, so what actually executes is
|
||||
// ~/.cache/node-systray/<version>/tray_darwin_release, and index.js only re-copies
|
||||
// when that path is absent. An overlaid binary stays invisible until this is cleared.
|
||||
// Scoped to our systray2 version: the parent dir is machine-global and shared
|
||||
// with any other node-systray consumer.
|
||||
function bustSystrayCopyCache() {
|
||||
try {
|
||||
fs.rmSync(path.join(os.homedir(), ".cache", "node-systray", SYSTRAY_VERSION), { recursive: true, force: true });
|
||||
} catch {}
|
||||
}
|
||||
|
||||
function arm64AttemptMarker() {
|
||||
return path.join(getRuntimeDir(), ".tray-arm64-attempt");
|
||||
}
|
||||
|
||||
// ensureTrayRuntime runs synchronously on every `9router` start (cli.js), so a
|
||||
// failed download must not re-block the next launch. Retry at most daily.
|
||||
function recentlyAttemptedArm64() {
|
||||
try {
|
||||
const at = Number(fs.readFileSync(arm64AttemptMarker(), "utf8").trim());
|
||||
return Number.isFinite(at) && Date.now() - at < ARM64_RETRY_COOLDOWN_MS;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function markArm64Attempt() {
|
||||
try { fs.writeFileSync(arm64AttemptMarker(), String(Date.now())); } catch {}
|
||||
}
|
||||
|
||||
// Cleared on success so the cooldown only ever throttles *failures*. Without
|
||||
// this, anything that restores systray2's x86_64 binary later — notably a
|
||||
// globally installed 9router older than this change, which shares the same
|
||||
// ~/.9router/runtime — would leave the user waiting out the cooldown.
|
||||
function clearArm64Attempt() {
|
||||
try { fs.rmSync(arm64AttemptMarker(), { force: true }); } catch {}
|
||||
}
|
||||
|
||||
// Throws on any failure so the caller has a single error path.
|
||||
function downloadFile(url, dest, timeoutSec) {
|
||||
// darwin-only path, and curl ships with macOS, so this needs no extra dep and
|
||||
// keeps the caller synchronous.
|
||||
const res = spawnSync("curl", ["-fsSL", "--max-time", String(timeoutSec), "-o", dest, url], {
|
||||
encoding: "utf8",
|
||||
timeout: (timeoutSec + 5) * 1000
|
||||
});
|
||||
if (res.status === 0 && fs.existsSync(dest)) return;
|
||||
const detail = (res.stderr || res.error?.message || `curl exit ${res.status}`).trim().split("\n").pop();
|
||||
throw new Error(detail || "download failed");
|
||||
}
|
||||
|
||||
function sha256File(file) {
|
||||
return crypto.createHash("sha256").update(fs.readFileSync(file)).digest("hex");
|
||||
}
|
||||
|
||||
// Replace systray2's x86_64 macOS binary with a native arm64 build so Apple
|
||||
// Silicon users get a tray without installing Rosetta 2. Any failure leaves the
|
||||
// Intel binary untouched, which still works under Rosetta.
|
||||
//
|
||||
// Takes no `silent` flag on purpose: cli.js calls ensureTrayRuntime({silent:true})
|
||||
// synchronously on every start, and a stalled curl would otherwise freeze the
|
||||
// launch for up to 30s with no output at all. These lines print at most once per
|
||||
// 24h on failure and once ever on success, so they are worth more than the quiet.
|
||||
function ensureArm64TrayBin() {
|
||||
if (process.platform !== "darwin" || process.arch !== "arm64") return { skipped: true };
|
||||
|
||||
const binPath = path.join(getRuntimeNodeModules(), SYSTRAY_PKG, "traybin", "tray_darwin_release");
|
||||
if (!fs.existsSync(binPath)) return { skipped: true };
|
||||
if (isArm64MachO(binPath)) return { native: true };
|
||||
if (recentlyAttemptedArm64()) return { deferred: true };
|
||||
|
||||
markArm64Attempt();
|
||||
console.log("⏳ Downloading native Apple Silicon tray binary...");
|
||||
// pid-scoped: two concurrent starts (postinstall racing cli.js, or two
|
||||
// terminals) would otherwise interleave writes to one file, fail each other's
|
||||
// checksum, and delete each other's in-flight download from the catch below.
|
||||
const tmp = `${binPath}.arm64.${process.pid}.tmp`;
|
||||
try {
|
||||
downloadFile(ARM64_TRAY_URL, tmp, 30);
|
||||
const sum = sha256File(tmp);
|
||||
// Integrity matters more than usual: this is an executable that runs on
|
||||
// every Apple Silicon user's machine.
|
||||
if (sum !== ARM64_TRAY_SHA256) throw new Error(`checksum mismatch (got ${sum.slice(0, 12)}…)`);
|
||||
if (!isArm64MachO(tmp)) throw new Error("downloaded file is not an arm64 Mach-O");
|
||||
fs.chmodSync(tmp, 0o755);
|
||||
fs.renameSync(tmp, binPath);
|
||||
bustSystrayCopyCache();
|
||||
clearArm64Attempt();
|
||||
console.log("✅ Native Apple Silicon tray installed");
|
||||
return { native: true, installed: true };
|
||||
} catch (e) {
|
||||
try { fs.rmSync(tmp, { force: true }); } catch {}
|
||||
console.warn("⚠️ Native tray download failed — falling back to the Intel binary");
|
||||
console.warn(` Reason: ${e.message}`);
|
||||
console.warn(" The Intel tray needs Rosetta 2: softwareupdate --install-rosetta --agree-to-license");
|
||||
return { native: false, error: e.message };
|
||||
}
|
||||
}
|
||||
|
||||
function npmInstall(pkgs, { silent = false } = {}) {
|
||||
const cwd = ensureRuntimeDir();
|
||||
if (!silent) console.log("⏳ Installing system tray (first run)...");
|
||||
@@ -94,14 +233,20 @@ function ensureTrayRuntime({ silent = false } = {}) {
|
||||
if (process.platform === "win32") {
|
||||
return { systray: false, skipped: true };
|
||||
}
|
||||
if (hasSystray()) {
|
||||
|
||||
let ready = hasSystray();
|
||||
if (!ready) {
|
||||
ready = npmInstall([`${SYSTRAY_PKG}@${SYSTRAY_VERSION}`], { silent }) && hasSystray();
|
||||
}
|
||||
if (ready) {
|
||||
chmodSystrayBin({ silent });
|
||||
if (!silent) console.log("✅ System tray ready");
|
||||
return { systray: true };
|
||||
}
|
||||
const ok = npmInstall([`${SYSTRAY_PKG}@${SYSTRAY_VERSION}`], { silent });
|
||||
if (ok) chmodSystrayBin({ silent });
|
||||
return { systray: ok && hasSystray() };
|
||||
|
||||
// Runs after the ready log so a download failure doesn't read as a broken
|
||||
// tray — the Intel binary still works under Rosetta.
|
||||
const arm64 = ready ? ensureArm64TrayBin() : { skipped: true };
|
||||
return { systray: ready, arm64 };
|
||||
}
|
||||
|
||||
module.exports = { ensureTrayRuntime };
|
||||
module.exports = { ensureTrayRuntime, ensureArm64TrayBin };
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "9router",
|
||||
"version": "0.5.45",
|
||||
"version": "0.5.91",
|
||||
"description": "9Router CLI - Start and manage 9Router server",
|
||||
"bin": {
|
||||
"9router": "./cli.js"
|
||||
@@ -16,7 +16,8 @@
|
||||
"scripts": {
|
||||
"dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js",
|
||||
"build": "node scripts/build-cli.js",
|
||||
"pack:cli": "npm run build && npm pack --pack-destination ../..",
|
||||
"build:tray-arm64": "node scripts/buildTrayArm64.js",
|
||||
"pack:cli": "npm run build && npm pack --pack-destination ..",
|
||||
"publish:cli": "npm run build && npm publish",
|
||||
"postinstall": "node hooks/postinstall.js",
|
||||
"prepublishOnly": "npm run build"
|
||||
@@ -29,7 +30,8 @@
|
||||
"react-dom": "19.2.1"
|
||||
},
|
||||
"comment_sqlite": "sql.js + better-sqlite3 are NOT bundled here. They are installed into ~/.9router/runtime/node_modules by hooks/postinstall.js (and re-checked at runtime by cli.js). This avoids Windows EBUSY errors when updating the global CLI, since native .node files no longer live under the locked install dir.",
|
||||
"comment_systray": "systray2 is NOT bundled here. It is lazy-installed into ~/.9router/runtime/node_modules by hooks/postinstall.js on macOS/Linux only. Windows uses PowerShell NotifyIcon (zero binary). This avoids shipping unsigned Go binaries that trigger antivirus false positives (Kaspersky). We use the systray2 fork because the legacy systray@1.0.5 ships a 2017 x86_64 binary that fails on modern macOS dyld.",
|
||||
"comment_systray": "systray2 is NOT bundled here. It is lazy-installed into ~/.9router/runtime/node_modules by hooks/postinstall.js on macOS/Linux only. Windows uses PowerShell NotifyIcon (zero binary). This avoids shipping unsigned Go binaries that trigger antivirus false positives (Kaspersky). We use the systray2 fork because the legacy systray@1.0.5 ships a 2017 x86_64 binary that fails to load on modern macOS dyld. Neither package ships an arm64 macOS binary, so on Apple Silicon hooks/trayRuntime.js overlays our own arm64 build from the tray-binaries GitHub release; without it the tray requires Rosetta 2.",
|
||||
"comment_tray_arm64": "tray_darwin_arm64 is built by scripts/buildTrayArm64.js (npm run build:tray-arm64) from felixhao28/systray-portable — the same source systray2's binary comes from — and uploaded to the pinned 'tray-binaries' GitHub release. Its sha256 is pinned as ARM64_TRAY_SHA256 in hooks/trayRuntime.js and verified after every download; rebuild and update both together. .github/workflows/tray-binaries.yml builds the same artifact in CI but refuses to publish when the sha diverges from the pin.",
|
||||
"engines": {
|
||||
"node": ">=18.0.0"
|
||||
},
|
||||
|
||||
@@ -81,201 +81,274 @@ function copyRecursive(src, dest) {
|
||||
}
|
||||
}
|
||||
|
||||
console.log("📦 Building 9Router CLI package with Next.js...\n");
|
||||
function resolveStandaloneBuild(appDir, buildDistDir) {
|
||||
const legacyStandaloneRoot = path.join(appDir, ".next", "standalone");
|
||||
const resolvedStandaloneRoot = path.join(buildDistDir, "standalone");
|
||||
let standaloneRoot = fs.existsSync(resolvedStandaloneRoot)
|
||||
? resolvedStandaloneRoot
|
||||
: legacyStandaloneRoot;
|
||||
|
||||
fs.mkdirSync(buildHomeDir, { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Roaming"), { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Local"), { recursive: true });
|
||||
|
||||
// Step 0: Sync version from app/cli/package.json to app/package.json
|
||||
console.log("0️⃣ Syncing version to app/package.json...");
|
||||
const cliPkg = JSON.parse(fs.readFileSync(path.join(cliDir, "package.json"), "utf8"));
|
||||
const appPkgPath = path.join(appDir, "package.json");
|
||||
const appPkg = JSON.parse(fs.readFileSync(appPkgPath, "utf8"));
|
||||
if (appPkg.version !== cliPkg.version) {
|
||||
appPkg.version = cliPkg.version;
|
||||
fs.writeFileSync(appPkgPath, JSON.stringify(appPkg, null, 2) + "\n");
|
||||
console.log(`✅ Version synced: ${cliPkg.version}\n`);
|
||||
} else {
|
||||
console.log(`✅ Version already synced: ${cliPkg.version}\n`);
|
||||
}
|
||||
|
||||
// Step 1: Build app with Next.js (workspace tracing root → traced node_modules in standalone).
|
||||
console.log("1️⃣ Building Next.js app...");
|
||||
try {
|
||||
execSync("npm run build", {
|
||||
stdio: "inherit",
|
||||
cwd: appDir,
|
||||
env: {
|
||||
...process.env,
|
||||
HOME: buildHomeDir,
|
||||
USERPROFILE: buildHomeDir,
|
||||
APPDATA: path.join(buildHomeDir, "AppData", "Roaming"),
|
||||
LOCALAPPDATA: path.join(buildHomeDir, "AppData", "Local"),
|
||||
NEXT_DIST_DIR: buildDistDirName,
|
||||
NEXT_TRACING_ROOT_MODE: "workspace",
|
||||
}
|
||||
});
|
||||
console.log("✅ Next.js build completed\n");
|
||||
} catch (error) {
|
||||
console.error("❌ Next.js build failed");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Step 2: Clean old app/cli/app if exists
|
||||
console.log("2️⃣ Cleaning old app/cli/app...");
|
||||
if (fs.existsSync(cliAppDir)) {
|
||||
fs.rmSync(cliAppDir, { recursive: true, force: true });
|
||||
}
|
||||
console.log("✅ Cleaned\n");
|
||||
|
||||
// Step 3: Copy Next.js standalone build to app/cli/app.
|
||||
// Newer Next.js standalone output writes server.js/package.json plus .next/, src/, and
|
||||
// node_modules/ directly under .next/standalone. Older builds may still use a nested app/.
|
||||
console.log("3️⃣ Copying Next.js standalone build to app/cli/app...");
|
||||
const standaloneRoot = path.join(appDir, ".next", "standalone");
|
||||
const standaloneRootResolved = path.join(buildDistDir, "standalone");
|
||||
let standaloneRootToUse = fs.existsSync(standaloneRootResolved) ? standaloneRootResolved : standaloneRoot;
|
||||
// Next.js 16 nests standalone output under the project name when NEXT_TRACING_ROOT_MODE=workspace
|
||||
// e.g. .next-cli-build/standalone/9router/server.js
|
||||
const pkgName = path.basename(appDir);
|
||||
const nestedRoot = path.join(standaloneRootToUse, pkgName);
|
||||
if (fs.existsSync(path.join(nestedRoot, "server.js")) && !fs.existsSync(path.join(standaloneRootToUse, "server.js"))) {
|
||||
console.log(`ℹ️ Detected nested standalone output: ${pkgName}/`);
|
||||
standaloneRootToUse = nestedRoot;
|
||||
}
|
||||
const standaloneApp = fs.existsSync(path.join(standaloneRootToUse, "server.js"))
|
||||
? standaloneRootToUse
|
||||
: path.join(standaloneRootToUse, "app");
|
||||
if (!fs.existsSync(standaloneApp)) {
|
||||
console.error("❌ Next.js standalone build not found under .next/standalone");
|
||||
console.error("Expected either .next/standalone/server.js or .next/standalone/app/");
|
||||
process.exit(1);
|
||||
}
|
||||
copyRecursive(standaloneApp, cliAppDir);
|
||||
|
||||
// Older nested-app layout stores traced node_modules at standalone root.
|
||||
const standaloneNodeModules = path.join(standaloneRootToUse, "node_modules");
|
||||
if (standaloneApp !== standaloneRootToUse && fs.existsSync(standaloneNodeModules)) {
|
||||
copyRecursive(standaloneNodeModules, path.join(cliAppDir, "node_modules"));
|
||||
}
|
||||
console.log("✅ Copied standalone build\n");
|
||||
|
||||
// Step 3a: Copy custom server (injects real socket IP, strips spoofable XFF).
|
||||
const customServerSrc = path.join(appDir, "custom-server.js");
|
||||
if (fs.existsSync(customServerSrc)) {
|
||||
fs.copyFileSync(customServerSrc, path.join(cliAppDir, "custom-server.js"));
|
||||
console.log("✅ Copied custom-server.js\n");
|
||||
} else {
|
||||
console.warn("⚠️ custom-server.js not found — server will run without real-IP injection\n");
|
||||
}
|
||||
|
||||
// Step 3b: Ensure sql.js (pure JS fallback) bundled in app/cli/app/node_modules.
|
||||
// Strip better-sqlite3 (native) — it lives in ~/.9router/runtime to avoid
|
||||
// Windows EBUSY during global CLI updates. node:sqlite (Node ≥22.5) is also
|
||||
// available as a no-install middle tier.
|
||||
console.log("3️⃣ b Configuring SQLite drivers...");
|
||||
function ensureModuleInBundle(pkg) {
|
||||
const dest = path.join(cliAppDir, "node_modules", pkg);
|
||||
if (fs.existsSync(dest)) {
|
||||
console.log(`✅ ${pkg} already bundled`);
|
||||
return;
|
||||
// Next.js 16 nests standalone output under the project name when
|
||||
// NEXT_TRACING_ROOT_MODE=workspace, e.g. standalone/9router/server.js.
|
||||
const pkgName = path.basename(appDir);
|
||||
const nestedRoot = path.join(standaloneRoot, pkgName);
|
||||
if (fs.existsSync(path.join(nestedRoot, "server.js")) && !fs.existsSync(path.join(standaloneRoot, "server.js"))) {
|
||||
console.log(`ℹ️ Detected nested standalone output: ${pkgName}/`);
|
||||
standaloneRoot = nestedRoot;
|
||||
}
|
||||
const candidates = [
|
||||
path.join(appDir, "node_modules", pkg),
|
||||
path.join(rootDir, "node_modules", pkg),
|
||||
|
||||
const standaloneApp = fs.existsSync(path.join(standaloneRoot, "server.js"))
|
||||
? standaloneRoot
|
||||
: path.join(standaloneRoot, "app");
|
||||
if (!fs.existsSync(standaloneApp)) {
|
||||
throw new Error(
|
||||
"Next.js standalone build not found under .next/standalone; " +
|
||||
"expected either .next/standalone/server.js or .next/standalone/app/",
|
||||
);
|
||||
}
|
||||
|
||||
return { standaloneApp, standaloneRoot };
|
||||
}
|
||||
|
||||
function copyStandaloneBuild(appDir, buildDistDir, cliAppDir) {
|
||||
const { standaloneApp, standaloneRoot } = resolveStandaloneBuild(appDir, buildDistDir);
|
||||
copyRecursive(standaloneApp, cliAppDir);
|
||||
|
||||
// Older nested-app layout stores traced node_modules at standalone root.
|
||||
const standaloneNodeModules = path.join(standaloneRoot, "node_modules");
|
||||
if (standaloneApp !== standaloneRoot && fs.existsSync(standaloneNodeModules)) {
|
||||
copyRecursive(standaloneNodeModules, path.join(cliAppDir, "node_modules"));
|
||||
}
|
||||
}
|
||||
|
||||
function mergeServerArtifacts(buildDistDir, cliAppDir) {
|
||||
const serverSrc = path.join(buildDistDir, "server");
|
||||
const serverDest = path.join(cliAppDir, buildDistDirName, "server");
|
||||
if (!fs.existsSync(serverSrc)) {
|
||||
throw new Error(`Complete Next.js server build not found: ${serverSrc}`);
|
||||
}
|
||||
copyRecursive(serverSrc, serverDest);
|
||||
}
|
||||
|
||||
function assertRequiredApiArtifacts(cliAppDir) {
|
||||
const requiredArtifacts = [
|
||||
"app/api/v1/chat/completions/route.js",
|
||||
"app/api/v1/messages/route.js",
|
||||
];
|
||||
const src = candidates.find((p) => fs.existsSync(p));
|
||||
if (!src) {
|
||||
console.warn(`⚠️ ${pkg} not found locally — bundle will rely on node:sqlite or runtime install`);
|
||||
return;
|
||||
const serverDir = path.join(cliAppDir, buildDistDirName, "server");
|
||||
const missingArtifacts = requiredArtifacts
|
||||
.map((artifact) => path.join(serverDir, artifact))
|
||||
.filter((artifact) => !fs.existsSync(artifact));
|
||||
|
||||
if (missingArtifacts.length > 0) {
|
||||
throw new Error(
|
||||
`Required CLI API route artifact${missingArtifacts.length === 1 ? " is" : "s are"} missing:\n` +
|
||||
missingArtifacts.join("\n"),
|
||||
);
|
||||
}
|
||||
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
||||
copyRecursive(src, dest);
|
||||
console.log(`✅ Bundled ${pkg}`);
|
||||
}
|
||||
ensureModuleInBundle("sql.js");
|
||||
const betterDir = path.join(cliAppDir, "node_modules", "better-sqlite3");
|
||||
if (fs.existsSync(betterDir)) {
|
||||
fs.rmSync(betterDir, { recursive: true, force: true });
|
||||
console.log("✅ Stripped better-sqlite3 (lives in ~/.9router/runtime)");
|
||||
}
|
||||
console.log("");
|
||||
|
||||
// Step 4: Copy static files
|
||||
console.log("4️⃣ Copying static files...");
|
||||
const staticSrc = path.join(appDir, ".next", "static");
|
||||
const staticSrcResolved = path.join(buildDistDir, "static");
|
||||
const staticDest = path.join(cliAppDir, buildDistDirName, "static");
|
||||
if (fs.existsSync(staticSrcResolved) || fs.existsSync(staticSrc)) {
|
||||
copyRecursive(fs.existsSync(staticSrcResolved) ? staticSrcResolved : staticSrc, staticDest);
|
||||
console.log("✅ Copied static files\n");
|
||||
} else {
|
||||
console.log("⏭️ No static files found\n");
|
||||
}
|
||||
|
||||
// Step 5: Copy public folder if exists
|
||||
console.log("5️⃣ Copying public folder...");
|
||||
const publicSrc = path.join(appDir, "public");
|
||||
const publicDest = path.join(cliAppDir, "public");
|
||||
if (fs.existsSync(publicSrc)) {
|
||||
copyRecursive(publicSrc, publicDest);
|
||||
console.log("✅ Copied public folder\n");
|
||||
} else {
|
||||
console.log("⏭️ No public folder found\n");
|
||||
function buildCliPackage() {
|
||||
console.log("📦 Building 9Router CLI package with Next.js...\n");
|
||||
|
||||
fs.mkdirSync(buildHomeDir, { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Roaming"), { recursive: true });
|
||||
fs.mkdirSync(path.join(buildHomeDir, "AppData", "Local"), { recursive: true });
|
||||
|
||||
// Step 0: Sync version from app/cli/package.json to app/package.json
|
||||
console.log("0️⃣ Syncing version to app/package.json...");
|
||||
const cliPkg = JSON.parse(fs.readFileSync(path.join(cliDir, "package.json"), "utf8"));
|
||||
const appPkgPath = path.join(appDir, "package.json");
|
||||
const appPkg = JSON.parse(fs.readFileSync(appPkgPath, "utf8"));
|
||||
if (appPkg.version !== cliPkg.version) {
|
||||
appPkg.version = cliPkg.version;
|
||||
fs.writeFileSync(appPkgPath, JSON.stringify(appPkg, null, 2) + "\n");
|
||||
console.log(`✅ Version synced: ${cliPkg.version}\n`);
|
||||
} else {
|
||||
console.log(`✅ Version already synced: ${cliPkg.version}\n`);
|
||||
}
|
||||
|
||||
// Step 1: Build app with Next.js (workspace tracing root → traced node_modules in standalone).
|
||||
console.log("1️⃣ Building Next.js app...");
|
||||
try {
|
||||
execSync("npm run build", {
|
||||
stdio: "inherit",
|
||||
cwd: appDir,
|
||||
env: {
|
||||
...process.env,
|
||||
HOME: buildHomeDir,
|
||||
USERPROFILE: buildHomeDir,
|
||||
APPDATA: path.join(buildHomeDir, "AppData", "Roaming"),
|
||||
LOCALAPPDATA: path.join(buildHomeDir, "AppData", "Local"),
|
||||
NEXT_DIST_DIR: buildDistDirName,
|
||||
NEXT_TRACING_ROOT_MODE: "workspace",
|
||||
}
|
||||
});
|
||||
console.log("✅ Next.js build completed\n");
|
||||
} catch (error) {
|
||||
console.error("❌ Next.js build failed");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Step 2: Clean old app/cli/app if exists
|
||||
console.log("2️⃣ Cleaning old app/cli/app...");
|
||||
if (fs.existsSync(cliAppDir)) {
|
||||
fs.rmSync(cliAppDir, { recursive: true, force: true });
|
||||
}
|
||||
console.log("✅ Cleaned\n");
|
||||
|
||||
// Step 3: Copy Next.js standalone build to app/cli/app.
|
||||
// Newer Next.js standalone output writes server.js/package.json plus .next/, src/, and
|
||||
// node_modules/ directly under .next/standalone. Older builds may still use a nested app/.
|
||||
console.log("3️⃣ Copying Next.js standalone build to app/cli/app...");
|
||||
try {
|
||||
copyStandaloneBuild(appDir, buildDistDir, cliAppDir);
|
||||
} catch (error) {
|
||||
console.error("❌ Next.js standalone build not found under .next/standalone");
|
||||
console.error("Expected either .next/standalone/server.js or .next/standalone/app/");
|
||||
process.exit(1);
|
||||
}
|
||||
console.log("✅ Copied standalone build\n");
|
||||
|
||||
// Step 3a: Copy custom server (injects real socket IP, strips spoofable XFF).
|
||||
const customServerSrc = path.join(appDir, "custom-server.js");
|
||||
if (fs.existsSync(customServerSrc)) {
|
||||
fs.copyFileSync(customServerSrc, path.join(cliAppDir, "custom-server.js"));
|
||||
console.log("✅ Copied custom-server.js\n");
|
||||
} else {
|
||||
console.error("❌ custom-server.js not found — without it no request can be proven local,");
|
||||
console.error(" so the packaged CLI would demand an API key for its own dashboard and /v1.");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Step 3b: Ensure sql.js (pure JS fallback) bundled in app/cli/app/node_modules.
|
||||
// Strip better-sqlite3 (native) — it lives in ~/.9router/runtime to avoid
|
||||
// Windows EBUSY during global CLI updates. node:sqlite (Node ≥22.5) is also
|
||||
// available as a no-install middle tier.
|
||||
console.log("3️⃣ b Configuring SQLite drivers...");
|
||||
function ensureModuleInBundle(pkg) {
|
||||
const dest = path.join(cliAppDir, "node_modules", pkg);
|
||||
if (fs.existsSync(dest)) {
|
||||
console.log(`✅ ${pkg} already bundled`);
|
||||
return;
|
||||
}
|
||||
const candidates = [
|
||||
path.join(appDir, "node_modules", pkg),
|
||||
path.join(rootDir, "node_modules", pkg),
|
||||
];
|
||||
const src = candidates.find((p) => fs.existsSync(p));
|
||||
if (!src) {
|
||||
console.warn(`⚠️ ${pkg} not found locally — bundle will rely on node:sqlite or runtime install`);
|
||||
return;
|
||||
}
|
||||
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
||||
copyRecursive(src, dest);
|
||||
console.log(`✅ Bundled ${pkg}`);
|
||||
}
|
||||
ensureModuleInBundle("sql.js");
|
||||
// `open` is external (see serverExternalPackages in next.config.mjs), so it must exist in
|
||||
// the bundle's node_modules or every importer throws MODULE_NOT_FOUND at runtime. Output
|
||||
// tracing normally copies it; this is the same belt-and-braces guard used for sql.js.
|
||||
ensureModuleInBundle("open");
|
||||
const betterDir = path.join(cliAppDir, "node_modules", "better-sqlite3");
|
||||
if (fs.existsSync(betterDir)) {
|
||||
fs.rmSync(betterDir, { recursive: true, force: true });
|
||||
console.log("✅ Stripped better-sqlite3 (lives in ~/.9router/runtime)");
|
||||
}
|
||||
console.log("");
|
||||
|
||||
// Step 4: Copy static files
|
||||
console.log("4️⃣ Copying static files...");
|
||||
const staticSrc = path.join(appDir, ".next", "static");
|
||||
const staticSrcResolved = path.join(buildDistDir, "static");
|
||||
const staticDest = path.join(cliAppDir, buildDistDirName, "static");
|
||||
if (fs.existsSync(staticSrcResolved) || fs.existsSync(staticSrc)) {
|
||||
copyRecursive(fs.existsSync(staticSrcResolved) ? staticSrcResolved : staticSrc, staticDest);
|
||||
console.log("✅ Copied static files\n");
|
||||
} else {
|
||||
console.log("⏭️ No static files found\n");
|
||||
}
|
||||
|
||||
// Step 5: Copy public folder if exists
|
||||
console.log("5️⃣ Copying public folder...");
|
||||
const publicSrc = path.join(appDir, "public");
|
||||
const publicDest = path.join(cliAppDir, "public");
|
||||
if (fs.existsSync(publicSrc)) {
|
||||
copyRecursive(publicSrc, publicDest);
|
||||
console.log("✅ Copied public folder\n");
|
||||
} else {
|
||||
console.log("⏭️ No public folder found\n");
|
||||
}
|
||||
|
||||
// Step 6: Copy vendor-chunks (required for production)
|
||||
console.log("6️⃣ Copying vendor-chunks...");
|
||||
const vendorChunksSrc = path.join(appDir, ".next", "server", "vendor-chunks");
|
||||
const vendorChunksSrcResolved = path.join(buildDistDir, "server", "vendor-chunks");
|
||||
const vendorChunksDest = path.join(cliAppDir, buildDistDirName, "server", "vendor-chunks");
|
||||
if (fs.existsSync(vendorChunksSrcResolved) || fs.existsSync(vendorChunksSrc)) {
|
||||
copyRecursive(fs.existsSync(vendorChunksSrcResolved) ? vendorChunksSrcResolved : vendorChunksSrc, vendorChunksDest);
|
||||
console.log("✅ Copied vendor-chunks\n");
|
||||
} else {
|
||||
console.log("⏭️ No vendor-chunks found\n");
|
||||
}
|
||||
|
||||
// Step 6b: Merge the complete generated server tree. Next.js standalone output
|
||||
// is trace-pruned and can omit route modules or chunks loaded dynamically.
|
||||
console.log("6️⃣ b Copying complete server artifacts...");
|
||||
mergeServerArtifacts(buildDistDir, cliAppDir);
|
||||
assertRequiredApiArtifacts(cliAppDir);
|
||||
console.log("✅ Copied complete server artifacts\n");
|
||||
|
||||
// Step 7: Copy MITM server files (not bundled by Next.js standalone)
|
||||
console.log("7️⃣ Copying MITM server files...");
|
||||
const mitmSrc = path.join(appDir, "src", "mitm");
|
||||
const mitmDest = path.join(cliAppDir, "src", "mitm");
|
||||
if (fs.existsSync(mitmSrc)) {
|
||||
copyRecursive(mitmSrc, mitmDest);
|
||||
console.log("✅ Copied MITM files\n");
|
||||
} else {
|
||||
console.log("⏭️ No MITM files found\n");
|
||||
}
|
||||
|
||||
// Step 7b: Copy standalone updater (headless Node process for install progress)
|
||||
console.log("7️⃣ b Copying updater files...");
|
||||
const updaterSrc = path.join(appDir, "src", "lib", "updater");
|
||||
const updaterDest = path.join(cliAppDir, "src", "lib", "updater");
|
||||
if (fs.existsSync(updaterSrc)) {
|
||||
copyRecursive(updaterSrc, updaterDest);
|
||||
console.log("✅ Copied updater files\n");
|
||||
} else {
|
||||
console.log("⏭️ No updater files found\n");
|
||||
}
|
||||
|
||||
// Step 8: Build MITM server (config driven - see app/cli/scripts/buildMitm.js)
|
||||
console.log("8️⃣ Building MITM server...");
|
||||
try {
|
||||
execSync("node scripts/buildMitm.js", { stdio: "inherit", cwd: cliDir });
|
||||
console.log("✅ MITM server build completed\n");
|
||||
} catch (error) {
|
||||
console.error("❌ MITM build failed");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log("✨ CLI package build completed!");
|
||||
console.log(`📁 Output: ${cliAppDir}`);
|
||||
|
||||
try {
|
||||
const { execSync: exec } = require("child_process");
|
||||
const size = exec(`du -sh "${cliAppDir}"`, { encoding: "utf8" }).trim();
|
||||
console.log(`📊 Package size: ${size.split("\t")[0]}`);
|
||||
} catch (e) {
|
||||
// Silent fail on size check
|
||||
}
|
||||
}
|
||||
|
||||
// Step 6: Copy vendor-chunks (required for production)
|
||||
console.log("6️⃣ Copying vendor-chunks...");
|
||||
const vendorChunksSrc = path.join(appDir, ".next", "server", "vendor-chunks");
|
||||
const vendorChunksSrcResolved = path.join(buildDistDir, "server", "vendor-chunks");
|
||||
const vendorChunksDest = path.join(cliAppDir, buildDistDirName, "server", "vendor-chunks");
|
||||
if (fs.existsSync(vendorChunksSrcResolved) || fs.existsSync(vendorChunksSrc)) {
|
||||
copyRecursive(fs.existsSync(vendorChunksSrcResolved) ? vendorChunksSrcResolved : vendorChunksSrc, vendorChunksDest);
|
||||
console.log("✅ Copied vendor-chunks\n");
|
||||
} else {
|
||||
console.log("⏭️ No vendor-chunks found\n");
|
||||
}
|
||||
module.exports = {
|
||||
assertRequiredApiArtifacts,
|
||||
copyStandaloneBuild,
|
||||
mergeServerArtifacts,
|
||||
};
|
||||
|
||||
// Step 7: Copy MITM server files (not bundled by Next.js standalone)
|
||||
console.log("7️⃣ Copying MITM server files...");
|
||||
const mitmSrc = path.join(appDir, "src", "mitm");
|
||||
const mitmDest = path.join(cliAppDir, "src", "mitm");
|
||||
if (fs.existsSync(mitmSrc)) {
|
||||
copyRecursive(mitmSrc, mitmDest);
|
||||
console.log("✅ Copied MITM files\n");
|
||||
} else {
|
||||
console.log("⏭️ No MITM files found\n");
|
||||
}
|
||||
|
||||
// Step 7b: Copy standalone updater (headless Node process for install progress)
|
||||
console.log("7️⃣ b Copying updater files...");
|
||||
const updaterSrc = path.join(appDir, "src", "lib", "updater");
|
||||
const updaterDest = path.join(cliAppDir, "src", "lib", "updater");
|
||||
if (fs.existsSync(updaterSrc)) {
|
||||
copyRecursive(updaterSrc, updaterDest);
|
||||
console.log("✅ Copied updater files\n");
|
||||
} else {
|
||||
console.log("⏭️ No updater files found\n");
|
||||
}
|
||||
|
||||
// Step 8: Build MITM server (config driven - see app/cli/scripts/buildMitm.js)
|
||||
console.log("8️⃣ Building MITM server...");
|
||||
try {
|
||||
execSync("node scripts/buildMitm.js", { stdio: "inherit", cwd: cliDir });
|
||||
console.log("✅ MITM server build completed\n");
|
||||
} catch (error) {
|
||||
console.error("❌ MITM build failed");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log("✨ CLI package build completed!");
|
||||
console.log(`📁 Output: ${cliAppDir}`);
|
||||
|
||||
try {
|
||||
const { execSync: exec } = require("child_process");
|
||||
const size = exec(`du -sh "${cliAppDir}"`, { encoding: "utf8" }).trim();
|
||||
console.log(`📊 Package size: ${size.split("\t")[0]}`);
|
||||
} catch (e) {
|
||||
// Silent fail on size check
|
||||
if (require.main === module) {
|
||||
buildCliPackage();
|
||||
}
|
||||
|
||||
108
cli/scripts/buildTrayArm64.js
Normal file
108
cli/scripts/buildTrayArm64.js
Normal file
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
// Rebuilds tray_darwin_arm64, the native Apple Silicon menubar binary that
|
||||
// hooks/trayRuntime.js overlays on top of systray2's x86_64-only build.
|
||||
//
|
||||
// Must run on macOS: getlantern/systray is cgo against AppKit, so the arm64
|
||||
// slice needs a real macOS SDK. Requires Go on PATH (`mise use -g go@latest`).
|
||||
//
|
||||
// Output is NOT reproducible across Go versions even with -s -w, so after a
|
||||
// rebuild you must re-upload the asset and update ARM64_TRAY_SHA256 in
|
||||
// hooks/trayRuntime.js — this script prints both and fails if they diverge.
|
||||
|
||||
const { execFileSync, spawnSync } = require("child_process");
|
||||
const crypto = require("crypto");
|
||||
const fs = require("fs");
|
||||
const os = require("os");
|
||||
const path = require("path");
|
||||
|
||||
const UPSTREAM_REPO = "https://github.com/felixhao28/systray-portable.git";
|
||||
// master as of 2021-09-15, the commit systray2@2.1.4's own binary was built from.
|
||||
const UPSTREAM_COMMIT = "6eddc917bf39fcc0d95b57a0741d0c065fbd1e23";
|
||||
|
||||
const outDir = path.join(__dirname, "..", ".tray-build");
|
||||
const outFile = path.join(outDir, "tray_darwin_arm64");
|
||||
const trayRuntimePath = path.join(__dirname, "..", "hooks", "trayRuntime.js");
|
||||
|
||||
function fail(msg) {
|
||||
console.error(`\n❌ ${msg}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
function run(cmd, args, opts = {}) {
|
||||
const res = spawnSync(cmd, args, { stdio: "inherit", ...opts });
|
||||
if (res.status !== 0) fail(`${cmd} ${args.join(" ")} exited with ${res.status}`);
|
||||
}
|
||||
|
||||
if (process.platform !== "darwin") fail("must be run on macOS (cgo needs the AppKit SDK)");
|
||||
|
||||
const go = spawnSync("go", ["version"], { encoding: "utf8" });
|
||||
if (go.status !== 0) {
|
||||
fail("Go toolchain not found on PATH. Install it with: mise use -g go@latest");
|
||||
}
|
||||
console.log(`Go: ${go.stdout.trim()}`);
|
||||
|
||||
const srcDir = fs.mkdtempSync(path.join(os.tmpdir(), "systray-portable-"));
|
||||
// fail() exits through process.exit(), which does not unwind the stack, so a
|
||||
// try/finally here would leak the clone on every failed build. An exit handler
|
||||
// covers normal completion, fail(), and uncaught exceptions alike.
|
||||
process.on("exit", () => {
|
||||
try { fs.rmSync(srcDir, { recursive: true, force: true }); } catch {}
|
||||
});
|
||||
|
||||
console.log(`\nCloning ${UPSTREAM_REPO} @ ${UPSTREAM_COMMIT.slice(0, 7)}`);
|
||||
run("git", ["clone", "--quiet", UPSTREAM_REPO, srcDir]);
|
||||
run("git", ["-C", srcDir, "checkout", "--quiet", UPSTREAM_COMMIT]);
|
||||
|
||||
const head = execFileSync("git", ["-C", srcDir, "log", "-1", "--format=%H %ad %s", "--date=short"], {
|
||||
encoding: "utf8"
|
||||
}).trim();
|
||||
console.log(`HEAD: ${head}`);
|
||||
|
||||
run("go", ["mod", "download"], { cwd: srcDir });
|
||||
|
||||
console.log("\nBuilding darwin/arm64...");
|
||||
// -trimpath strips the local build directory from the binary, so two builds
|
||||
// from the same commit + Go version hash identically regardless of where they
|
||||
// ran. Without it the pinned sha256 could never be regenerated.
|
||||
run("go", ["build", "-trimpath", "-ldflags", "-s -w", "-o", outFile, "tray.go"], {
|
||||
cwd: srcDir,
|
||||
env: { ...process.env, CGO_ENABLED: "1", GOOS: "darwin", GOARCH: "arm64" }
|
||||
});
|
||||
|
||||
// ── Verify ────────────────────────────────────────────────────────────────
|
||||
const buf = Buffer.alloc(8);
|
||||
const fd = fs.openSync(outFile, "r");
|
||||
fs.readSync(fd, buf, 0, 8, 0);
|
||||
fs.closeSync(fd);
|
||||
if (buf.readUInt32LE(0) !== 0xfeedfacf || buf.readUInt32LE(4) !== 0x0100000c) {
|
||||
fail("output is not a thin arm64 Mach-O");
|
||||
}
|
||||
|
||||
// Apple Silicon refuses to execute an unsigned binary. Go's linker applies an
|
||||
// ad-hoc signature automatically; confirm it survived.
|
||||
const sig = spawnSync("codesign", ["--verify", outFile], { encoding: "utf8" });
|
||||
if (sig.status !== 0) fail(`ad-hoc signature invalid: ${(sig.stderr || "").trim()}`);
|
||||
|
||||
const sha256 = crypto.createHash("sha256").update(fs.readFileSync(outFile)).digest("hex");
|
||||
const sizeMb = (fs.statSync(outFile).size / 1024 / 1024).toFixed(2);
|
||||
|
||||
console.log(`\n✅ ${outFile}`);
|
||||
console.log(` arch: arm64 (ad-hoc signed, verified)`);
|
||||
console.log(` size: ${sizeMb} MB`);
|
||||
console.log(` sha256: ${sha256}`);
|
||||
|
||||
// Whitespace-tolerant: a formatter could wrap the assignment across lines, and
|
||||
// a null match must fail loudly rather than silently read as "no pin".
|
||||
const pinMatch = fs.readFileSync(trayRuntimePath, "utf8").match(/ARM64_TRAY_SHA256\s*=\s*"([0-9a-f]{64})"/);
|
||||
if (!pinMatch) fail(`could not find ARM64_TRAY_SHA256 in ${trayRuntimePath}`);
|
||||
const pinned = pinMatch[1];
|
||||
if (pinned === sha256) {
|
||||
console.log(`\n Matches ARM64_TRAY_SHA256 in hooks/trayRuntime.js — no code change needed.`);
|
||||
} else {
|
||||
console.log(`\n⚠️ Differs from ARM64_TRAY_SHA256 in hooks/trayRuntime.js (${pinned}).`);
|
||||
console.log(` Re-upload the release asset, then update that constant to the sha256 above.`);
|
||||
}
|
||||
|
||||
console.log(`\nNext: upload to the pinned release tag, keeping the asset name stable:`);
|
||||
console.log(` gh release upload tray-binaries "${outFile}" --clobber`);
|
||||
@@ -53,6 +53,12 @@ const PROVIDER_MODELS = {
|
||||
{ id: "glm-4.7" },
|
||||
],
|
||||
ag: [
|
||||
{ id: "gemini-3.8-flash-high" },
|
||||
{ id: "gemini-3.8-flash-medium" },
|
||||
{ id: "gemini-3.8-flash-low" },
|
||||
{ id: "gemini-3.7-flash-high" },
|
||||
{ id: "gemini-3.7-flash-medium" },
|
||||
{ id: "gemini-3.7-flash-low" },
|
||||
{ id: "gemini-3.6-flash-high" },
|
||||
{ id: "gemini-3.6-flash-medium" },
|
||||
{ id: "gemini-3.6-flash-low" },
|
||||
@@ -98,6 +104,8 @@ const PROVIDER_MODELS = {
|
||||
{ id: "claude-3-5-sonnet-20241022" },
|
||||
],
|
||||
gemini: [
|
||||
{ id: "gemini-3.8-flash" },
|
||||
{ id: "gemini-3.7-flash" },
|
||||
{ id: "gemini-3.6-flash" },
|
||||
{ id: "gemini-3.5-flash-lite" },
|
||||
{ id: "gemini-3-pro-preview" },
|
||||
|
||||
@@ -141,11 +141,14 @@ function initWindowsTray(options) {
|
||||
/**
|
||||
* macOS/Linux tray via systray binary
|
||||
*
|
||||
* Prefers `systray2` (active fork of `systray`, ships newer
|
||||
* getlantern/systray-portable binaries that work on macOS 14+ and Apple
|
||||
* Silicon under Rosetta). Falls back to legacy `systray@1.0.5` if systray2
|
||||
* is not available, though that binary's Mach-O headers are rejected by
|
||||
* modern dyld and the icon will not appear.
|
||||
* Prefers `systray2`, the active fork of `systray`. Both ship only an x86_64
|
||||
* `tray_darwin_release` and select it by process.platform alone, so on Apple
|
||||
* Silicon the tray runs under Rosetta 2 and fails with EBADARCH when Rosetta is
|
||||
* absent. hooks/trayRuntime.js overlays a native arm64 build over that file to
|
||||
* avoid the dependency; the fallbacks below are Intel-only.
|
||||
*
|
||||
* Falls back to legacy `systray@1.0.5` if systray2 is unavailable, though that
|
||||
* binary's Mach-O headers are rejected by modern dyld and no icon will appear.
|
||||
*/
|
||||
function resolveSystray() {
|
||||
let runtimeDir = null;
|
||||
|
||||
@@ -2,22 +2,24 @@ const api = require("../api/client");
|
||||
const { prompt } = require("./input");
|
||||
const { clearScreen } = require("./display");
|
||||
|
||||
// Provider alias order: OAuth first, then API Key (matches ModelSelectModal)
|
||||
// Provider alias order: OAuth first, then Free, then API Key
|
||||
const PROVIDER_ALIAS_ORDER = [
|
||||
"cc", "ag", "cx", "if", "qw", "gc", "gh", "kr",
|
||||
"cc", "ag", "cx", "if", "qw", "gc", "gh", "kr", "oc",
|
||||
"openrouter", "glm", "kimi", "minimax", "openai", "anthropic", "gemini"
|
||||
];
|
||||
|
||||
// Alias to display name mapping
|
||||
const PROVIDER_ALIAS_NAMES = {
|
||||
cc: "Claude Code",
|
||||
ag: "Antigravity",
|
||||
ag: "Antigravity",
|
||||
cx: "OpenAI Codex",
|
||||
if: "iFlow AI",
|
||||
qw: "Qwen Code",
|
||||
gc: "Gemini CLI",
|
||||
gh: "GitHub Copilot",
|
||||
kr: "Kiro AI",
|
||||
oc: "OpenCode Free",
|
||||
opencode: "OpenCode Free",
|
||||
openrouter: "OpenRouter",
|
||||
glm: "GLM Coding",
|
||||
kimi: "Kimi Coding",
|
||||
@@ -27,35 +29,83 @@ const PROVIDER_ALIAS_NAMES = {
|
||||
gemini: "Gemini"
|
||||
};
|
||||
|
||||
const PROVIDER_ID_TO_ALIAS = {
|
||||
claude: "cc",
|
||||
codex: "cx",
|
||||
"gemini-cli": "gc",
|
||||
github: "gh",
|
||||
antigravity: "ag",
|
||||
iflow: "if",
|
||||
qwen: "qw",
|
||||
kiro: "kr",
|
||||
cursor: "cu",
|
||||
cline: "cline",
|
||||
clinepass: "clinepass",
|
||||
qoder: "qd",
|
||||
"qoder-cn": "qd",
|
||||
gitlab: "gitlab",
|
||||
"codebuddy-cn": "cb",
|
||||
"codebuddy-intl": "cbai",
|
||||
kimchi: "kimchi",
|
||||
"grok-cli": "grok-cli",
|
||||
trae: "trae",
|
||||
windsurf: "windsurf",
|
||||
zed: "zed",
|
||||
opencode: "oc",
|
||||
"opencode-go": "ocg",
|
||||
"opencode-zen": "ocz",
|
||||
};
|
||||
|
||||
// Providers usable without stored credentials
|
||||
const NO_AUTH_PROVIDERS = new Set(["opencode", "oc"]);
|
||||
|
||||
/**
|
||||
* Get all available models grouped by provider + combos
|
||||
* Get all available models grouped by provider + combos (filtered by active connections)
|
||||
* @returns {Promise<{combos: Array, groups: Object}>}
|
||||
*/
|
||||
async function getAvailableModelsGrouped() {
|
||||
const result = await api.getAvailableModels();
|
||||
if (!result.success) return { combos: [], groups: {} };
|
||||
|
||||
const models = result.data?.data || [];
|
||||
const [modelsResult, providersResult] = await Promise.all([
|
||||
api.getAvailableModels(),
|
||||
api.getProviders()
|
||||
]);
|
||||
|
||||
if (!modelsResult.success) return { combos: [], groups: {} };
|
||||
|
||||
const connections = providersResult.success ? (providersResult.data?.connections || []) : [];
|
||||
const activeAliases = new Set(NO_AUTH_PROVIDERS);
|
||||
|
||||
connections.forEach(conn => {
|
||||
if (conn.isActive === false) return;
|
||||
const p = conn.provider;
|
||||
if (!p) return;
|
||||
activeAliases.add(p);
|
||||
const alias = conn.providerSpecificData?.prefix || PROVIDER_ID_TO_ALIAS[p] || p;
|
||||
activeAliases.add(alias);
|
||||
});
|
||||
|
||||
const models = modelsResult.data?.data || [];
|
||||
const combos = [];
|
||||
const groups = {};
|
||||
|
||||
|
||||
models.forEach(m => {
|
||||
if (m.owned_by === "combo") {
|
||||
combos.push(m.id);
|
||||
} else {
|
||||
const provider = m.owned_by;
|
||||
// Only keep connected providers or noAuth providers
|
||||
if (!activeAliases.has(provider)) return;
|
||||
if (!groups[provider]) {
|
||||
groups[provider] = [];
|
||||
}
|
||||
groups[provider].push(m.id);
|
||||
}
|
||||
});
|
||||
|
||||
|
||||
return { combos, groups };
|
||||
}
|
||||
|
||||
/**
|
||||
* Display model list and prompt for selection
|
||||
* Display model list and prompt for selection with provider grouping & search
|
||||
* @param {string} title - Title to display
|
||||
* @param {string} currentValue - Current selected value (optional)
|
||||
* @param {Object} options - { excludeCombos?: boolean }
|
||||
@@ -68,64 +118,214 @@ async function selectModelFromList(title, currentValue = "", options = {}) {
|
||||
|
||||
const totalModels = combos.length + Object.values(groups).flat().length;
|
||||
if (totalModels === 0) {
|
||||
clearScreen();
|
||||
console.log(`\n🎯 ${title}`);
|
||||
console.log("=".repeat(50));
|
||||
console.log("\n No connected providers found.");
|
||||
console.log(" Please connect a provider in Providers menu first.\n");
|
||||
console.log(" m. ✍️ Enter custom model ID");
|
||||
console.log(" 0. Cancel\n");
|
||||
const act = await prompt("Select option (m/0): ");
|
||||
const trimmed = act.trim();
|
||||
if (trimmed.toLowerCase() === "m") {
|
||||
const custom = await prompt("Enter custom model ID: ");
|
||||
return custom.trim() || null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Build flat list for selection
|
||||
const allModels = [];
|
||||
|
||||
// Display
|
||||
clearScreen();
|
||||
console.log(`\n🎯 ${title}`);
|
||||
console.log("=".repeat(50));
|
||||
if (currentValue) {
|
||||
console.log(`Current: ${currentValue}\n`);
|
||||
} else {
|
||||
console.log();
|
||||
}
|
||||
|
||||
let idx = 1;
|
||||
|
||||
// Combos first (skipped when excludeCombos is true)
|
||||
|
||||
// All models for flat search
|
||||
const allModelsList = [
|
||||
...combos,
|
||||
...Object.values(groups).flat()
|
||||
];
|
||||
|
||||
// Build category list
|
||||
const categories = [];
|
||||
if (combos.length > 0) {
|
||||
console.log("[Combos]");
|
||||
combos.forEach(combo => {
|
||||
console.log(` ${idx}. ${combo}`);
|
||||
allModels.push(combo);
|
||||
idx++;
|
||||
categories.push({
|
||||
id: "combos",
|
||||
name: "[Combos]",
|
||||
models: combos
|
||||
});
|
||||
console.log();
|
||||
}
|
||||
|
||||
// Provider groups in order (by alias)
|
||||
|
||||
const sortedProviders = Object.keys(groups).sort((a, b) => {
|
||||
const idxA = PROVIDER_ALIAS_ORDER.indexOf(a);
|
||||
const idxB = PROVIDER_ALIAS_ORDER.indexOf(b);
|
||||
return (idxA === -1 ? 999 : idxA) - (idxB === -1 ? 999 : idxB);
|
||||
});
|
||||
|
||||
sortedProviders.forEach(provider => {
|
||||
|
||||
sortedProviders.forEach((provider) => {
|
||||
const providerName = PROVIDER_ALIAS_NAMES[provider] || provider;
|
||||
console.log(`[${providerName}]`);
|
||||
groups[provider].forEach(model => {
|
||||
console.log(` ${idx}. ${model}`);
|
||||
allModels.push(model);
|
||||
idx++;
|
||||
categories.push({
|
||||
id: provider,
|
||||
name: providerName,
|
||||
models: groups[provider]
|
||||
});
|
||||
console.log();
|
||||
});
|
||||
|
||||
console.log(" 0. Cancel\n");
|
||||
|
||||
// Prompt for number input
|
||||
const input = await prompt("Enter number: ");
|
||||
const num = parseInt(input, 10);
|
||||
|
||||
if (isNaN(num) || num === 0 || num < 0 || num > allModels.length) {
|
||||
return null;
|
||||
|
||||
let filterQuery = null;
|
||||
|
||||
while (true) {
|
||||
clearScreen();
|
||||
console.log(`\n🎯 ${title}`);
|
||||
console.log("=".repeat(50));
|
||||
if (currentValue) {
|
||||
console.log(`Current: ${currentValue}\n`);
|
||||
} else {
|
||||
console.log();
|
||||
}
|
||||
|
||||
// Active search view
|
||||
if (filterQuery !== null) {
|
||||
const q = filterQuery.toLowerCase().trim();
|
||||
const matched = allModelsList.filter((m) => m.toLowerCase().includes(q));
|
||||
|
||||
console.log(`🔍 Search results for "${filterQuery}": (${matched.length} found)\n`);
|
||||
if (matched.length === 0) {
|
||||
console.log(" No matching models found.\n");
|
||||
console.log(" 0. ← Back to providers");
|
||||
console.log(" s. Search again\n");
|
||||
const act = await prompt("Select option: ");
|
||||
if (act.toLowerCase() === "s") {
|
||||
const newQ = await prompt("Enter search keyword: ");
|
||||
filterQuery = newQ.trim() || null;
|
||||
} else {
|
||||
filterQuery = null;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
matched.forEach((m, i) => {
|
||||
console.log(` ${i + 1}. ${m}`);
|
||||
});
|
||||
console.log("\n 0. ← Back to providers");
|
||||
console.log(" s. Search again\n");
|
||||
|
||||
const input = await prompt("Enter number to select (or 0/s): ");
|
||||
if (input.toLowerCase() === "s") {
|
||||
const newQ = await prompt("Enter search keyword: ");
|
||||
filterQuery = newQ.trim() || null;
|
||||
continue;
|
||||
}
|
||||
const num = parseInt(input, 10);
|
||||
if (isNaN(num) || num === 0) {
|
||||
filterQuery = null;
|
||||
continue;
|
||||
}
|
||||
if (num > 0 && num <= matched.length) {
|
||||
return matched[num - 1];
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// If only 1 category exists, jump straight into its model list
|
||||
if (categories.length === 1) {
|
||||
const singleCategory = categories[0];
|
||||
console.log(`[${singleCategory.name}]`);
|
||||
singleCategory.models.forEach((m, i) => {
|
||||
console.log(` ${i + 1}. ${m}`);
|
||||
});
|
||||
console.log();
|
||||
console.log(" s. 🔍 Search models");
|
||||
console.log(" m. ✍️ Enter custom model ID");
|
||||
console.log(" 0. Cancel\n");
|
||||
|
||||
const input = await prompt("Enter choice (number / s / m / 0): ");
|
||||
const trimmed = input.trim();
|
||||
if (!trimmed || trimmed === "0") return null;
|
||||
|
||||
const lower = trimmed.toLowerCase();
|
||||
if (lower === "s") {
|
||||
const q = await prompt("Enter search keyword: ");
|
||||
if (q.trim()) filterQuery = q.trim();
|
||||
continue;
|
||||
}
|
||||
if (lower === "m") {
|
||||
const customModel = await prompt("Enter custom model ID: ");
|
||||
if (customModel.trim()) return customModel.trim();
|
||||
continue;
|
||||
}
|
||||
|
||||
const num = parseInt(trimmed, 10);
|
||||
if (!isNaN(num) && num > 0 && num <= singleCategory.models.length) {
|
||||
return singleCategory.models[num - 1];
|
||||
}
|
||||
filterQuery = trimmed;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Multiple categories view
|
||||
console.log("[Providers & Groups]");
|
||||
categories.forEach((cat, i) => {
|
||||
console.log(` ${i + 1}. ${cat.name} (${cat.models.length} models)`);
|
||||
});
|
||||
|
||||
console.log();
|
||||
console.log(" s. 🔍 Search models");
|
||||
console.log(" m. ✍️ Enter custom model ID");
|
||||
console.log(" 0. Cancel\n");
|
||||
|
||||
const input = await prompt("Enter choice (number / keyword / s / m): ");
|
||||
const trimmed = input.trim();
|
||||
|
||||
if (!trimmed || trimmed === "0") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const lower = trimmed.toLowerCase();
|
||||
if (lower === "s") {
|
||||
const q = await prompt("Enter search keyword: ");
|
||||
if (q.trim()) {
|
||||
filterQuery = q.trim();
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (lower === "m") {
|
||||
const customModel = await prompt("Enter custom model ID: ");
|
||||
if (customModel.trim()) {
|
||||
return customModel.trim();
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
const num = parseInt(trimmed, 10);
|
||||
// Selected a category
|
||||
if (!isNaN(num) && num > 0 && num <= categories.length) {
|
||||
const selectedCategory = categories[num - 1];
|
||||
|
||||
while (true) {
|
||||
clearScreen();
|
||||
console.log(`\n🎯 ${title} > ${selectedCategory.name}`);
|
||||
console.log("=".repeat(50));
|
||||
if (currentValue) {
|
||||
console.log(`Current: ${currentValue}\n`);
|
||||
} else {
|
||||
console.log();
|
||||
}
|
||||
|
||||
selectedCategory.models.forEach((m, i) => {
|
||||
console.log(` ${i + 1}. ${m}`);
|
||||
});
|
||||
console.log("\n 0. ← Back\n");
|
||||
|
||||
const modelChoice = await prompt("Enter number to select (0 to back): ");
|
||||
const modelNum = parseInt(modelChoice, 10);
|
||||
if (isNaN(modelNum) || modelNum === 0) {
|
||||
break;
|
||||
}
|
||||
if (modelNum > 0 && modelNum <= selectedCategory.models.length) {
|
||||
return selectedCategory.models[modelNum - 1];
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// User typed text directly -> treat as search query
|
||||
filterQuery = trimmed;
|
||||
}
|
||||
|
||||
return allModels[num - 1];
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
|
||||
111
custom-server.js
111
custom-server.js
@@ -1,7 +1,51 @@
|
||||
const http = require("http");
|
||||
const path = require("path");
|
||||
const fs = require("fs");
|
||||
const crypto = require("crypto");
|
||||
const { pathToFileURL } = require("url");
|
||||
|
||||
const origCreate = http.createServer.bind(http);
|
||||
|
||||
// Per-process secret proving x-9r-real-ip was stamped below rather than sent by the client.
|
||||
// A bare `next start` / `next dev` never loads this file, so it cannot produce a matching
|
||||
// header even though the env var is inherited by child processes. Named like x-9r-cli-token
|
||||
// so the request-detail header sanitizer redacts it too.
|
||||
const PEER_TOKEN = crypto.randomBytes(24).toString("hex");
|
||||
process.env.NINEROUTER_PEER_TOKEN = PEER_TOKEN;
|
||||
|
||||
let backgroundRefreshStarted = false;
|
||||
|
||||
function startBackgroundTokenRefreshFromCustomServer() {
|
||||
if (backgroundRefreshStarted) return;
|
||||
backgroundRefreshStarted = true;
|
||||
// Prefer source path (repo / standalone that still has src). Fail-open if missing
|
||||
// — initializeApp also starts the same scheduler when the Next app boots.
|
||||
const modPath = path.join(__dirname, "src", "sse", "services", "backgroundTokenRefresh.js");
|
||||
import(pathToFileURL(modPath).href)
|
||||
.then((m) => {
|
||||
try {
|
||||
m.startBackgroundTokenRefresh();
|
||||
} catch (e) {
|
||||
console.error("[BackgroundTokenRefresh] start failed:", e && e.message ? e.message : e);
|
||||
}
|
||||
const stop = () => {
|
||||
try {
|
||||
m.stopBackgroundTokenRefresh();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
};
|
||||
process.once("SIGINT", stop);
|
||||
process.once("SIGTERM", stop);
|
||||
})
|
||||
.catch((e) => {
|
||||
// Expected in published CLI standalone (src/ not on disk). App bootstrap covers it.
|
||||
if (process.env.DEBUG_BACKGROUND_TOKEN_REFRESH) {
|
||||
console.error("[BackgroundTokenRefresh] import failed:", e && e.message ? e.message : e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Wrap Next standalone HTTP server: derive client IP from the TCP socket
|
||||
// (unspoofable) and strip client-supplied forwarding headers so downstream
|
||||
// rate-limiting keys on the real peer address instead of attacker-controlled XFF.
|
||||
@@ -22,11 +66,74 @@ http.createServer = (...args) => {
|
||||
delete req.headers["x-9r-real-ip"];
|
||||
delete req.headers["x-forwarded-for"];
|
||||
delete req.headers["x-9r-via-proxy"];
|
||||
delete req.headers["x-9r-peer-token"];
|
||||
req.headers["x-9r-real-ip"] = ip;
|
||||
req.headers["x-9r-peer-token"] = PEER_TOKEN;
|
||||
if (viaProxy) req.headers["x-9r-via-proxy"] = "1";
|
||||
return handler(req, res);
|
||||
};
|
||||
return origCreate(...rest, wrapped);
|
||||
const server = origCreate(...rest, wrapped);
|
||||
server.once("listening", () => {
|
||||
startBackgroundTokenRefreshFromCustomServer();
|
||||
});
|
||||
const origEmit = server.emit;
|
||||
// JBR 25 sends h2c upgrades that the HTTP/1.1 server would otherwise close.
|
||||
server.emit = function (event, ...eventArgs) {
|
||||
const [req, socket, head] = eventArgs;
|
||||
if (event !== "upgrade" || String(req.headers.upgrade || "").toLowerCase() !== "h2c") {
|
||||
return origEmit.call(this, event, ...eventArgs);
|
||||
}
|
||||
|
||||
const contentLength = Number(req.headers["content-length"] || 0);
|
||||
if (!Number.isSafeInteger(contentLength) || contentLength < 0) {
|
||||
socket.destroy();
|
||||
return true;
|
||||
}
|
||||
const chunks = [head];
|
||||
let received = head.length;
|
||||
const serve = () => {
|
||||
// Replay the upgraded request through the existing HTTP/1.1 handler.
|
||||
const replay = new http.IncomingMessage(socket);
|
||||
Object.assign(replay, { method: req.method, url: req.url, headers: req.headers, complete: true });
|
||||
if (received) replay.push(Buffer.concat(chunks, received).subarray(0, contentLength));
|
||||
replay.push(null);
|
||||
const res = new http.ServerResponse(replay);
|
||||
res.shouldKeepAlive = false;
|
||||
res.assignSocket(socket);
|
||||
res.once("finish", () => socket.end());
|
||||
Promise.resolve().then(() => wrapped(replay, res)).catch((error) => {
|
||||
console.error("Failed to downgrade h2c request", error);
|
||||
socket.destroy();
|
||||
});
|
||||
};
|
||||
if (received >= contentLength) serve();
|
||||
else {
|
||||
socket.on("data", function readBody(chunk) {
|
||||
chunks.push(chunk);
|
||||
received += chunk.length;
|
||||
if (received < contentLength) return;
|
||||
socket.off("data", readBody);
|
||||
serve();
|
||||
});
|
||||
socket.resume();
|
||||
}
|
||||
delete req.headers.upgrade;
|
||||
delete req.headers["http2-settings"];
|
||||
req.headers.connection = "close";
|
||||
return true;
|
||||
};
|
||||
return server;
|
||||
};
|
||||
|
||||
require("./server.js");
|
||||
if (require.main === module) {
|
||||
const standalone = path.join(__dirname, "server.js");
|
||||
if (fs.existsSync(standalone)) {
|
||||
require(standalone);
|
||||
} else {
|
||||
// Repo checkout has no standalone build next to us. `next start` builds its HTTP
|
||||
// server in-process, so the wrapper above still sanitizes every request.
|
||||
const nextBin = require.resolve("next/dist/bin/next");
|
||||
process.argv = [process.argv[0], nextBin, "start", ...process.argv.slice(2)];
|
||||
require(nextBin);
|
||||
}
|
||||
}
|
||||
|
||||
BIN
docs/images/saml-admin-dashboard.png
Normal file
BIN
docs/images/saml-admin-dashboard.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 103 KiB |
BIN
docs/images/saml-login-screen.png
Normal file
BIN
docs/images/saml-login-screen.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 15 KiB |
@@ -0,0 +1,328 @@
|
||||
# GPT-5.6 Codex Reasoning Overrides Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Preserve Codex-advertised Max and Ultra overrides for GPT-5.6 Sol and Terra, preserve Max for Luna, and convert Luna Ultra to Max without changing Kiro or generic OpenAI-format behavior.
|
||||
|
||||
**Architecture:** Keep the supported reasoning matrix in the existing `getThinkingLevels(provider, model)` resolver and reuse that result in both translation and Codex executor normalization. The dashboard already consumes this resolver, so no UI component change is required. Unsupported top-end levels remain safely normalized, with Luna Ultra selecting Luna's supported Max level.
|
||||
|
||||
**Tech Stack:** JavaScript ES modules, Next.js, Vitest, Codex Responses transport.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Apply the new overrides only to the OpenAI Codex provider (`codex`, exposed as `cx/`).
|
||||
- Sol and Terra support `max` and `ultra`; Luna supports `max` but not `ultra`.
|
||||
- Convert Luna `ultra` requests to `max` in both translated and native passthrough request paths.
|
||||
- Preserve existing Kiro and generic OpenAI-compatible normalization.
|
||||
- Do not add runtime model-catalog fetching, dependencies, pricing changes, or unrelated refactors.
|
||||
- Write each behavior test first and observe the expected failure before changing production code.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Provider-scoped GPT-5.6 level matrix
|
||||
|
||||
**Files:**
|
||||
- Modify: `tests/unit/thinking-levels-gpt56-sol.test.js`
|
||||
- Modify: `open-sse/providers/thinkingLevels.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getThinkingLevels(provider, model)` and existing capability metadata.
|
||||
- Produces: `getThinkingLevels(provider, model): string[] | null` with Codex-only GPT-5.6 level overrides.
|
||||
|
||||
- [ ] **Step 1: Replace the Sol-only assertions with the complete behavior matrix**
|
||||
|
||||
Use literal expected arrays so each model/provider contract is independently checked:
|
||||
|
||||
```js
|
||||
it.each([
|
||||
["gpt-5.6-sol", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-terra", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-luna", ["none", "minimal", "low", "medium", "high", "xhigh", "max"]],
|
||||
["gpt-5.6-sol-review", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-terra-review", ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]],
|
||||
["gpt-5.6-luna-review", ["none", "minimal", "low", "medium", "high", "xhigh", "max"]],
|
||||
])("returns Codex levels for %s", (model, expected) => {
|
||||
expect(getThinkingLevels("codex", model)).toEqual(expected);
|
||||
});
|
||||
|
||||
it("does not expose Codex-only GPT-5.6 overrides on Kiro", () => {
|
||||
expect(getThinkingLevels("kiro", "gpt-5.6-sol")).toEqual([
|
||||
"none", "minimal", "low", "medium", "high", "xhigh",
|
||||
]);
|
||||
});
|
||||
```
|
||||
|
||||
Keep the older Codex-model assertion to protect the existing `gpt-5.3-codex` behavior.
|
||||
|
||||
- [ ] **Step 2: Run the level test and verify it fails for the missing matrix/provider scoping**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/thinking-levels-gpt56-sol.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because Sol lacks Ultra, Terra/Luna lack Max, and Kiro currently inherits Sol Max.
|
||||
|
||||
- [ ] **Step 3: Add provider-aware pattern matching and the three Codex model rules**
|
||||
|
||||
Update `PATTERN_THINKING` entries to accept an optional `provider` field and match it in `getThinkingLevels`:
|
||||
|
||||
```js
|
||||
const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
||||
|
||||
const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] },
|
||||
];
|
||||
|
||||
const hit = PATTERN_THINKING.find((entry) =>
|
||||
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
|
||||
);
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Re-run the level test and verify it passes**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/thinking-levels-gpt56-sol.test.js
|
||||
```
|
||||
|
||||
Expected: 1 test file passed with no failures.
|
||||
|
||||
- [ ] **Step 5: Commit the capability matrix**
|
||||
|
||||
```bash
|
||||
git add open-sse/providers/thinkingLevels.js tests/unit/thinking-levels-gpt56-sol.test.js
|
||||
git commit -m "feat(codex): expose GPT-5.6 reasoning overrides"
|
||||
```
|
||||
|
||||
### Task 2: Model-aware shared thinking translation
|
||||
|
||||
**Files:**
|
||||
- Modify: `tests/translator/thinking-unified.test.js`
|
||||
- Modify: `open-sse/translator/concerns/thinkingUnified.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getThinkingLevels(provider, cleanModel): string[] | null` from Task 1.
|
||||
- Produces: `parseSuffix(model)` support for `ultra` and `applyThinking(...)` output that preserves supported Codex levels.
|
||||
|
||||
- [ ] **Step 1: Add failing suffix and translation tests**
|
||||
|
||||
Add a literal parser assertion:
|
||||
|
||||
```js
|
||||
expect(parseSuffix("gpt-5.6-sol(ultra)")).toEqual({
|
||||
cleanModel: "gpt-5.6-sol",
|
||||
override: { mode: "level", level: "ultra" },
|
||||
});
|
||||
```
|
||||
|
||||
Add table-driven Codex assertions using direct request fields:
|
||||
|
||||
```js
|
||||
it.each([
|
||||
["gpt-5.6-sol", "max", "max"],
|
||||
["gpt-5.6-sol", "ultra", "ultra"],
|
||||
["gpt-5.6-terra", "max", "max"],
|
||||
["gpt-5.6-terra", "ultra", "ultra"],
|
||||
["gpt-5.6-luna", "max", "max"],
|
||||
["gpt-5.6-luna", "ultra", "max"],
|
||||
])("normalizes Codex %s effort %s to %s", (model, effort, expected) => {
|
||||
const out = apply("openai-responses", model, { reasoning: { effort } }, "codex");
|
||||
expect(out.reasoning_effort).toBe(expected);
|
||||
});
|
||||
```
|
||||
|
||||
Add a parenthesized override assertion and Kiro isolation assertion:
|
||||
|
||||
```js
|
||||
expect(apply("openai-responses", "gpt-5.6-sol(ultra)", {}, "codex").reasoning_effort).toBe("ultra");
|
||||
expect(apply("openai", "gpt-5.6-sol", { reasoning_effort: "max" }, "kiro").reasoning_effort).toBe("xhigh");
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run the translator test and verify it fails for Ultra parsing and preserved Max/Ultra**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/translator/thinking-unified.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because Ultra suffixes are ignored and OpenAI translation clamps Max to XHigh.
|
||||
|
||||
- [ ] **Step 3: Implement supported-level normalization in the shared translator**
|
||||
|
||||
Import `getThinkingLevels`. Recognize `ultra` explicitly in `parseSuffix` without adding it to the budget map. Resolve supported levels once in `applyThinking` and pass them to `applyFormat`.
|
||||
|
||||
Use this normalization rule for the OpenAI format:
|
||||
|
||||
```js
|
||||
function normalizeOpenAILevel(level, supportedLevels) {
|
||||
if (level !== "max" && level !== "ultra") return level;
|
||||
if (supportedLevels?.includes(level)) return level;
|
||||
if (level === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
return "xhigh";
|
||||
}
|
||||
```
|
||||
|
||||
Keep `none`, automatic effort, budget conversion, and every non-OpenAI format unchanged.
|
||||
|
||||
- [ ] **Step 4: Re-run the translator and generic OpenAI clamp tests**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/translator/thinking-unified.test.js tests/unit/thinking-effort-openai-max-clamp.test.js
|
||||
```
|
||||
|
||||
Expected: 2 test files passed; generic OpenAI Max still becomes XHigh.
|
||||
|
||||
- [ ] **Step 5: Commit shared translation support**
|
||||
|
||||
```bash
|
||||
git add open-sse/translator/concerns/thinkingUnified.js tests/translator/thinking-unified.test.js
|
||||
git commit -m "feat(codex): preserve supported reasoning efforts"
|
||||
```
|
||||
|
||||
### Task 3: Codex native passthrough normalization
|
||||
|
||||
**Files:**
|
||||
- Modify: `tests/unit/codex-fast-capacity.test.js`
|
||||
- Modify: `open-sse/executors/codex.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getThinkingLevels("codex", upstreamModel): string[] | null` from Task 1.
|
||||
- Produces: `CodexExecutor.transformRequest(...)` payloads with model-supported upstream `reasoning.effort` values.
|
||||
|
||||
- [ ] **Step 1: Add failing Codex executor behavior tests**
|
||||
|
||||
Add a separate `describe("Codex reasoning normalization", ...)` block with real `transformRequest` calls:
|
||||
|
||||
```js
|
||||
it.each([
|
||||
["gpt-5.6-sol", "max", "max"],
|
||||
["gpt-5.6-sol", "ultra", "ultra"],
|
||||
["gpt-5.6-terra", "max", "max"],
|
||||
["gpt-5.6-terra", "ultra", "ultra"],
|
||||
["gpt-5.6-luna", "max", "max"],
|
||||
["gpt-5.6-luna", "ultra", "max"],
|
||||
])("normalizes %s effort %s to %s", (model, effort, expected) => {
|
||||
const body = new CodexExecutor().transformRequest(model, {
|
||||
model,
|
||||
input: "hi",
|
||||
reasoning: { effort },
|
||||
}, true, {});
|
||||
expect(body.reasoning.effort).toBe(expected);
|
||||
});
|
||||
|
||||
it("resolves review models before applying the reasoning matrix", () => {
|
||||
const body = new CodexExecutor().transformRequest("gpt-5.6-terra-review", {
|
||||
model: "gpt-5.6-terra-review",
|
||||
input: "hi",
|
||||
reasoning_effort: "ultra",
|
||||
}, true, {});
|
||||
expect(body.model).toBe("gpt-5.6-terra");
|
||||
expect(body.reasoning.effort).toBe("ultra");
|
||||
});
|
||||
```
|
||||
|
||||
Keep the existing GPT-5.5 Max-to-XHigh fast-tier test.
|
||||
|
||||
- [ ] **Step 2: Run the executor test and verify supported values fail by being clamped**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/codex-fast-capacity.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because current normalization maps supported Max to XHigh and does not map Luna Ultra to Max.
|
||||
|
||||
- [ ] **Step 3: Make Codex normalization model-aware**
|
||||
|
||||
Import `getThinkingLevels` and replace the global Max clamp with:
|
||||
|
||||
```js
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
}
|
||||
```
|
||||
|
||||
Call it only after `body.model` has resolved review aliases to their upstream base model. Pass `body.model` for both `reasoning_effort` and existing `reasoning.effort` request shapes.
|
||||
|
||||
- [ ] **Step 4: Re-run the executor and focused feature suites**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/codex-fast-capacity.test.js tests/unit/thinking-levels-gpt56-sol.test.js tests/translator/thinking-unified.test.js tests/unit/thinking-effort-openai-max-clamp.test.js
|
||||
```
|
||||
|
||||
Expected: 4 test files passed with no failures.
|
||||
|
||||
- [ ] **Step 5: Commit native Codex normalization**
|
||||
|
||||
```bash
|
||||
git add open-sse/executors/codex.js tests/unit/codex-fast-capacity.test.js
|
||||
git commit -m "feat(codex): forward GPT-5.6 max and ultra efforts"
|
||||
```
|
||||
|
||||
### Task 4: Full verification and pull request
|
||||
|
||||
**Files:**
|
||||
- Verify all changed production, test, design, and plan files.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: completed Tasks 1-3.
|
||||
- Produces: verified branch pushed to `origin` and a pull request targeting `decolua/9router:master`.
|
||||
|
||||
- [ ] **Step 1: Run all focused regression tests**
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit/thinking-levels-gpt56-sol.test.js tests/translator/thinking-unified.test.js tests/unit/thinking-effort-openai-max-clamp.test.js tests/unit/codex-fast-capacity.test.js
|
||||
```
|
||||
|
||||
Expected: all selected test files and tests pass.
|
||||
|
||||
- [ ] **Step 2: Run the complete unit test suite**
|
||||
|
||||
```bash
|
||||
npx vitest run tests/unit tests/translator
|
||||
```
|
||||
|
||||
Expected: all test files pass with zero failed tests.
|
||||
|
||||
- [ ] **Step 3: Run the production build**
|
||||
|
||||
```bash
|
||||
npm run build
|
||||
```
|
||||
|
||||
Expected: Next.js production build exits with status 0.
|
||||
|
||||
- [ ] **Step 4: Verify repository hygiene and requirement coverage**
|
||||
|
||||
```bash
|
||||
git diff --check upstream/master...HEAD
|
||||
git status --short --branch
|
||||
git log --oneline upstream/master..HEAD
|
||||
```
|
||||
|
||||
Expected: no whitespace errors, no uncommitted source changes, and only scoped feature commits.
|
||||
|
||||
- [ ] **Step 5: Push the feature branch and open the pull request**
|
||||
|
||||
```bash
|
||||
git push -u origin codex/gpt-5-6-reasoning-overrides
|
||||
gh pr create --repo decolua/9router --base master --head seakleangnhak:codex/gpt-5-6-reasoning-overrides --title "feat(codex): support GPT-5.6 Max and Ultra overrides" --body $'## Summary\n- expose Max and Ultra for Codex GPT-5.6 Sol and Terra\n- expose Max for Codex GPT-5.6 Luna and normalize Luna Ultra to Max\n- keep Kiro and generic OpenAI-compatible reasoning behavior unchanged\n\n## Verification\n- `npx vitest run tests/unit tests/translator`\n- `npm run build`'
|
||||
```
|
||||
|
||||
The pull request body must summarize the Codex-only support matrix, Luna Ultra-to-Max fallback, Kiro isolation, and fresh test/build evidence.
|
||||
261
docs/superpowers/plans/2026-09-04-opencode-go-session-header.md
Normal file
261
docs/superpowers/plans/2026-09-04-opencode-go-session-header.md
Normal file
@@ -0,0 +1,261 @@
|
||||
# OpenCode Go Session Header Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Send a stable, conversation-scoped `x-opencode-session` header on every OpenCode Go request and install the patched CLI locally.
|
||||
|
||||
**Architecture:** Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. `chatCore` passes the provider-scoped session resolved from the original request plus the detected client tool; the executor derives a request-local upstream session and delegates all existing transport, authentication, retry, and proxy behavior to `DefaultExecutor`.
|
||||
|
||||
**Tech Stack:** Node.js ESM, Vitest, Next.js, npm CLI packaging, GitHub CLI.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Apply the header to OpenCode Go chat completions, Claude Messages, and OpenAI Responses transports.
|
||||
- Preserve a valid native `x-opencode-session`; hash all translated non-OpenCode identities to `ses_<32 lowercase hex>`.
|
||||
- Namespace translated identities by detected client tool, using `generic` when unknown.
|
||||
- Do not keep mutable per-request session state on the executor singleton or mutate the caller's credentials object.
|
||||
- Do not change OpenCode Go models, routing, reasoning, tool behavior, dependencies, or unrelated providers.
|
||||
- Reuse upstream issue #3759 instead of creating a duplicate issue.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Add Failing OpenCode Go Session Tests
|
||||
|
||||
**Files:**
|
||||
- Create: `tests/unit/opencode-go-session.test.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `getExecutor(provider)` and `DefaultExecutor.buildHeaders(credentials, stream, url, model)`.
|
||||
- Produces: the required public behavior for `OpenCodeGoExecutor.prepareRequestCredentials({ body, credentials, providerSessionId, clientTool })` and `OpenCodeGoExecutor.execute(args)`.
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
Create a Vitest suite that mocks `proxyAwareFetch`, obtains `getExecutor("opencode-go")`, and asserts:
|
||||
|
||||
```js
|
||||
const prepared = executor.prepareRequestCredentials({
|
||||
body: { messages: [{ role: "user", content: "hello" }] },
|
||||
credentials: { apiKey: "test-key", connectionId: "conn-a", rawHeaders: {} },
|
||||
providerSessionId: "conversation-a",
|
||||
clientTool: "claude",
|
||||
});
|
||||
|
||||
expect(prepared).not.toBe(credentials);
|
||||
expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/);
|
||||
expect(credentials).not.toHaveProperty("_opencodeGoSession");
|
||||
```
|
||||
|
||||
Cover native header preservation, stable values across all three runtime transports, different conversation IDs, different client tools using the same ID, connection fallback, no singleton state, no header on `DefaultExecutor("openai")`, and the final fetch headers returned by `execute()`.
|
||||
|
||||
- [ ] **Step 2: Run the focused test and verify RED**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because `getExecutor("opencode-go")` still returns `DefaultExecutor` and `prepareRequestCredentials` does not exist.
|
||||
|
||||
- [ ] **Step 3: Commit the failing test**
|
||||
|
||||
```bash
|
||||
git add tests/unit/opencode-go-session.test.js
|
||||
git commit -m "test: cover OpenCode Go session headers"
|
||||
```
|
||||
|
||||
### Task 2: Implement the Dedicated Executor
|
||||
|
||||
**Files:**
|
||||
- Create: `open-sse/executors/opencode-go.js`
|
||||
- Modify: `open-sse/executors/index.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `DefaultExecutor`, `resolveSessionId()`, request `credentials.rawHeaders`, `providerSessionId`, and `clientTool`.
|
||||
- Produces: `OpenCodeGoExecutor`, `prepareRequestCredentials()`, and an `execute()` override that delegates with cloned credentials.
|
||||
|
||||
- [ ] **Step 1: Add the minimal executor implementation**
|
||||
|
||||
Implement these rules:
|
||||
|
||||
```js
|
||||
function translatedSessionId(sessionId, clientTool) {
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
|
||||
.digest("hex")
|
||||
.slice(0, 32);
|
||||
return `ses_${digest}`;
|
||||
}
|
||||
```
|
||||
|
||||
`prepareRequestCredentials()` must read a case-insensitive native
|
||||
`x-opencode-session` with the same non-empty, 256-character cap used by the
|
||||
session manager. Otherwise it uses `providerSessionId` or calls
|
||||
`resolveSessionId({ headers, body, connectionId, scope: "opencode-go" })`, then
|
||||
returns `{ ...credentials, _opencodeGoSession: value }`.
|
||||
|
||||
`execute(args)` must call `prepareRequestCredentials(args)` and delegate using
|
||||
`super.execute({ ...args, credentials: prepared })`. `buildHeaders()` must call
|
||||
`super.buildHeaders()` and add the prepared session, with a connection-scoped
|
||||
fallback for direct callers.
|
||||
|
||||
Register `new OpenCodeGoExecutor()` under `"opencode-go"` and export the class.
|
||||
|
||||
- [ ] **Step 2: Run the focused test and verify partial GREEN**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
|
||||
```
|
||||
|
||||
Expected: executor-level tests pass; any chatCore-context assertion remains failing until Task 3.
|
||||
|
||||
- [ ] **Step 3: Commit the executor**
|
||||
|
||||
```bash
|
||||
git add open-sse/executors/opencode-go.js open-sse/executors/index.js tests/unit/opencode-go-session.test.js
|
||||
git commit -m "fix(opencode-go): add stable session header executor"
|
||||
```
|
||||
|
||||
### Task 3: Pass Original Request Session Context
|
||||
|
||||
**Files:**
|
||||
- Modify: `open-sse/handlers/chatCore.js`
|
||||
- Modify: `tests/unit/opencode-go-session.test.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: existing `sessionSeed` and `clientTool` variables in `handleChatCore()`.
|
||||
- Produces: `providerSessionId` and `clientTool` fields on both initial and refreshed-credential calls to `executor.execute()`.
|
||||
|
||||
- [ ] **Step 1: Add or enable the failing integration assertion**
|
||||
|
||||
Use a mocked executor or source request containing a body-only `session_id` and
|
||||
assert the executor receives the provider-scoped session resolved before
|
||||
translation.
|
||||
|
||||
- [ ] **Step 2: Run the focused test and verify RED**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
|
||||
```
|
||||
|
||||
Expected: FAIL because `handleChatCore()` does not pass `providerSessionId` or
|
||||
`clientTool` to `executor.execute()`.
|
||||
|
||||
- [ ] **Step 3: Pass the request context**
|
||||
|
||||
Add the same fields to both executor calls:
|
||||
|
||||
```js
|
||||
executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
providerSessionId: sessionSeed,
|
||||
clientTool,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run focused and neighboring tests**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
npx vitest run --config tests/vitest.config.js \
|
||||
tests/unit/opencode-go-session.test.js \
|
||||
tests/unit/opencode-go-models.test.js \
|
||||
tests/unit/session-manager.test.js \
|
||||
tests/unit/executor-const-guard.test.js
|
||||
```
|
||||
|
||||
Expected: PASS with zero failed tests.
|
||||
|
||||
- [ ] **Step 5: Commit the context wiring**
|
||||
|
||||
```bash
|
||||
git add open-sse/handlers/chatCore.js tests/unit/opencode-go-session.test.js
|
||||
git commit -m "fix(chat): forward provider session context"
|
||||
```
|
||||
|
||||
### Task 4: Verify and Install the Local CLI Package
|
||||
|
||||
**Files:**
|
||||
- Generated: `9router-0.5.65.tgz`
|
||||
- Packaged output: `cli/app/server.js`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: completed source changes and existing CLI build scripts.
|
||||
- Produces: a globally installed patched `9router@0.5.65`.
|
||||
|
||||
- [ ] **Step 1: Run source verification**
|
||||
|
||||
```bash
|
||||
git diff --check origin/master...HEAD
|
||||
npx vitest run --config tests/vitest.config.js tests/unit/
|
||||
npm run build
|
||||
```
|
||||
|
||||
Expected: every command exits zero. Record any pre-existing full-suite failures
|
||||
separately rather than hiding them.
|
||||
|
||||
- [ ] **Step 2: Build and package the CLI**
|
||||
|
||||
```bash
|
||||
npm --prefix cli run build
|
||||
npm --prefix cli pack -- --pack-destination ..
|
||||
```
|
||||
|
||||
Expected: `9router-0.5.65.tgz` exists and contains the patched bundled server.
|
||||
|
||||
- [ ] **Step 3: Replace the global npm installation**
|
||||
|
||||
```bash
|
||||
npm install -g ./9router-0.5.65.tgz
|
||||
```
|
||||
|
||||
Expected: `/opt/homebrew/lib/node_modules/9router/package.json` reports `0.5.65`
|
||||
and the installed bundle contains `x-opencode-session` plus the new executor.
|
||||
|
||||
- [ ] **Step 4: Commit any required package-source adjustment**
|
||||
|
||||
Do not commit generated tarballs or CLI build artifacts unless the repository
|
||||
already tracks and requires them.
|
||||
|
||||
### Task 5: Publish the Upstream Pull Request
|
||||
|
||||
**Files:**
|
||||
- No additional source files unless verification finds a required correction.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: verified branch commits and GitHub issue #3759.
|
||||
- Produces: a fork branch and a PR against `decolua/9router:master`.
|
||||
|
||||
- [ ] **Step 1: Create or repair the GitHub fork remote**
|
||||
|
||||
Use `gh repo fork decolua/9router --remote` if the current `fork` remote remains
|
||||
missing, then push `fix/opencode-go-session-header`.
|
||||
|
||||
- [ ] **Step 2: Create the PR**
|
||||
|
||||
Use title:
|
||||
|
||||
```text
|
||||
fix(opencode-go): send stable session header
|
||||
```
|
||||
|
||||
The body must include the root cause, downstream-session translation policy,
|
||||
three covered transports, concurrency behavior, verification evidence,
|
||||
`Fixes #3759`, and a note that this PR is intentionally narrower than #3780.
|
||||
|
||||
- [ ] **Step 3: Verify the published PR**
|
||||
|
||||
Run `gh pr view --json number,title,state,url,headRefName,baseRefName` and report
|
||||
the issue and PR URLs.
|
||||
@@ -0,0 +1,122 @@
|
||||
# GPT-5.6 Codex Reasoning Overrides Design
|
||||
|
||||
## Goal
|
||||
|
||||
Expose and preserve the reasoning levels currently advertised by the OpenAI
|
||||
Codex model catalog for GPT-5.6 Sol, Terra, and Luna when they are routed
|
||||
through the `codex` provider (`cx/`).
|
||||
|
||||
The supported override matrix is:
|
||||
|
||||
| Model family | Max | Ultra |
|
||||
| --- | --- | --- |
|
||||
| GPT-5.6 Sol | Yes | Yes |
|
||||
| GPT-5.6 Terra | Yes | Yes |
|
||||
| GPT-5.6 Luna | Yes | No |
|
||||
|
||||
The same matrix applies to 9router's virtual `-review` variants because they
|
||||
resolve to the corresponding upstream base model.
|
||||
|
||||
## Scope
|
||||
|
||||
This change is limited to OpenAI Codex (`cx/`) routes. Kiro (`kr/`) and other
|
||||
OpenAI-format providers retain their existing reasoning-level behavior even
|
||||
when they expose models with the same GPT-5.6 names.
|
||||
|
||||
The change covers the complete local request path:
|
||||
|
||||
1. The provider page advertises only the levels supported by each Codex model.
|
||||
2. A copied model suffix such as `gpt-5.6-sol(ultra)` is parsed as a reasoning
|
||||
override.
|
||||
3. The shared thinking translator preserves a supported Codex override while
|
||||
retaining the existing `xhigh` fallback for unsupported OpenAI levels.
|
||||
4. The Codex executor sends supported `max` and `ultra` values unchanged to the
|
||||
upstream Codex Responses endpoint.
|
||||
|
||||
## Current Behavior
|
||||
|
||||
`gpt-5.6-luna` and the other GPT-5.6 models already exist in the Codex model
|
||||
registry. The capability picker has a global Sol-only `max` pattern, which also
|
||||
affects providers such as Kiro unintentionally. The shared OpenAI translator
|
||||
and Codex executor then convert `max` to `xhigh`, so the advertised override is
|
||||
not preserved end to end. `ultra` is not recognized as a model suffix.
|
||||
|
||||
## Design
|
||||
|
||||
### Provider-scoped level resolution
|
||||
|
||||
Extend the existing model-pattern overrides in
|
||||
`open-sse/providers/thinkingLevels.js` with an optional provider constraint.
|
||||
Add three Codex-only GPT-5.6 patterns in most-specific order:
|
||||
|
||||
- Sol: existing levels plus `max` and `ultra`.
|
||||
- Terra: existing levels plus `max` and `ultra`.
|
||||
- Luna: existing levels plus `max`.
|
||||
|
||||
Matching remains wildcard-based so virtual `-review` variants inherit the
|
||||
base model's levels. Provider matching prevents these overrides from changing
|
||||
Kiro or other providers.
|
||||
|
||||
### Shared translation
|
||||
|
||||
Teach the suffix parser to recognize `ultra` as a discrete level without
|
||||
assigning it a synthetic token budget. When applying the OpenAI wire format,
|
||||
reuse the resolved per-provider model levels:
|
||||
|
||||
- Preserve `max` or `ultra` when the target provider/model explicitly supports
|
||||
the requested level.
|
||||
- Convert `ultra` to `max` for GPT-5.6 Luna, preserving the highest level Luna
|
||||
supports.
|
||||
- Convert other unsupported `max` or `ultra` requests to `xhigh`, preserving
|
||||
the existing safe fallback for generic OpenAI-compatible providers.
|
||||
- Leave all existing lower levels and `none` handling unchanged.
|
||||
|
||||
This keeps one capability source for the dashboard and translation behavior
|
||||
instead of duplicating the GPT-5.6 matrix.
|
||||
|
||||
### Codex executor
|
||||
|
||||
Make Codex reasoning normalization model-aware. After virtual review models
|
||||
are resolved to their upstream base model, preserve a requested level when
|
||||
the Codex capability resolver lists it. Continue converting unsupported
|
||||
`max` or `ultra` values to `xhigh`, except that Luna converts `ultra` to its
|
||||
supported `max` level.
|
||||
|
||||
Do not add `max` to the executor's legacy hyphen-suffix parser because
|
||||
`gpt-5.1-codex-max` is an actual model identifier. Dashboard overrides use the
|
||||
existing parenthesized suffix and the shared translator removes that suffix
|
||||
before executor dispatch.
|
||||
|
||||
## Error and Compatibility Behavior
|
||||
|
||||
- `cx/gpt-5.6-luna(ultra)` becomes `max` rather than sending an unsupported
|
||||
level upstream.
|
||||
- Non-GPT-5.6 Codex models retain their current supported levels and fallback
|
||||
behavior.
|
||||
- Kiro GPT-5.6 routes no longer inherit the Codex Sol-only picker override and
|
||||
continue using Kiro's existing effort normalization.
|
||||
- Direct request fields and parenthesized model overrides follow the same
|
||||
model-aware rules.
|
||||
|
||||
## Testing
|
||||
|
||||
Use test-driven development with focused unit coverage:
|
||||
|
||||
1. Level resolver tests for Sol, Terra, Luna, their review variants, an older
|
||||
Codex model, and Kiro isolation.
|
||||
2. Shared translator tests proving `max` and `ultra` survive only for supported
|
||||
Codex model/provider combinations, Luna `ultra` becomes `max`, and other
|
||||
unsupported combinations become `xhigh`.
|
||||
3. Codex executor tests proving native and translated request shapes preserve
|
||||
supported values after upstream model resolution.
|
||||
4. Existing thinking translation and Codex executor suites to guard generic
|
||||
OpenAI clamping and fast-tier behavior.
|
||||
5. Project lint/build checks in proportion to the changed JavaScript modules.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Runtime fetching or caching of the Codex model catalog.
|
||||
- Adding these levels to Kiro or another provider.
|
||||
- Changing model pricing, quotas, defaults, or service tiers.
|
||||
- Adding Codex Ultra's multi-agent orchestration behavior inside 9router;
|
||||
9router only forwards the catalog-advertised reasoning override.
|
||||
@@ -0,0 +1,114 @@
|
||||
# OpenCode Go Session Header Design
|
||||
|
||||
## Problem
|
||||
|
||||
OpenCode Go will begin rejecting some requests without an
|
||||
`x-opencode-session` header on September 6, 2026. In 9Router v0.5.65,
|
||||
`opencode-go` uses `DefaultExecutor`, whose generic header builder does not add
|
||||
that header. The specialized OpenCode Free executor already sends it, but that
|
||||
logic does not apply to the paid OpenCode Go provider or its three transports.
|
||||
|
||||
## Goals
|
||||
|
||||
- Add `x-opencode-session` to every OpenCode Go chat, Claude Messages, and
|
||||
OpenAI Responses request.
|
||||
- Translate a downstream conversation identity into a stable upstream identity.
|
||||
- Keep identities isolated across different downstream agents and conversations.
|
||||
- Avoid exposing non-OpenCode downstream session identifiers to OpenCode Go.
|
||||
- Avoid mutable session state on the shared executor singleton.
|
||||
- Leave OpenCode Free and all unrelated providers unchanged.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
- Inferring an exact conversation boundary when a downstream client provides no
|
||||
session or conversation identifier.
|
||||
- Adding or changing OpenCode Go models, routing, reasoning, or tool behavior.
|
||||
- Changing the general session-resolution policy for other providers.
|
||||
|
||||
## Architecture
|
||||
|
||||
Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. The executor
|
||||
keeps the existing generic URL, authentication, translation, retry, and proxy
|
||||
behavior, and overrides only the OpenCode Go session-header concern.
|
||||
|
||||
`handleChatCore` already resolves a provider-scoped session from the original
|
||||
request before translation. It will pass that value and the detected client
|
||||
tool to `executor.execute()` as request context. `OpenCodeGoExecutor.execute()`
|
||||
will create a shallow request-local credentials object containing the resolved
|
||||
OpenCode Go session. It will then delegate to `DefaultExecutor.execute()`.
|
||||
This avoids storing request state on the executor singleton or mutating shared
|
||||
provider credentials.
|
||||
|
||||
## Session Resolution
|
||||
|
||||
The original downstream request remains the source of truth. Existing
|
||||
`resolveSessionId()` behavior recognizes Claude Code, Antigravity, generic
|
||||
session headers, and common body fields before request translation can discard
|
||||
them.
|
||||
|
||||
Resolution rules:
|
||||
|
||||
1. If the downstream request supplies `x-opencode-session`, treat it as an
|
||||
authoritative OpenCode identity after trimming and length validation.
|
||||
2. Otherwise use the provider-scoped session resolved from the original request.
|
||||
3. Namespace the resolved value with the detected downstream agent, falling back
|
||||
to `generic` when the agent is unknown.
|
||||
4. Convert the namespaced value to an opaque deterministic identifier:
|
||||
`ses_` plus the first 32 hexadecimal characters of SHA-256.
|
||||
5. If no explicit downstream identity exists, the existing provider connection
|
||||
fallback guarantees that a header is still sent. It is stable but cannot
|
||||
distinguish multiple conversations sharing that connection.
|
||||
|
||||
The same input conversation produces the same upstream identifier for all three
|
||||
OpenCode Go transports. Different agents using the same raw session value
|
||||
produce different identifiers.
|
||||
|
||||
## Header Injection
|
||||
|
||||
`OpenCodeGoExecutor.buildHeaders()` delegates to
|
||||
`DefaultExecutor.buildHeaders()` and adds only:
|
||||
|
||||
```text
|
||||
x-opencode-session: <stable-session-id>
|
||||
```
|
||||
|
||||
The implementation applies to:
|
||||
|
||||
- `https://opencode.ai/zen/go/v1/chat/completions`
|
||||
- `https://opencode.ai/zen/go/v1/messages`
|
||||
- `https://opencode.ai/zen/go/v1/responses`
|
||||
|
||||
## Error Handling
|
||||
|
||||
Session derivation must not make requests fail. Invalid or oversized native
|
||||
header values are ignored and the normal resolved-session fallback is used.
|
||||
Hashing uses Node's built-in `crypto` module and requires no new dependency.
|
||||
|
||||
## Testing
|
||||
|
||||
Add a focused unit suite that proves:
|
||||
|
||||
- all three OpenCode Go transports receive the header;
|
||||
- the same conversation remains stable across requests and transports;
|
||||
- different conversations produce different values;
|
||||
- different agents using the same raw ID remain isolated;
|
||||
- non-OpenCode session IDs are represented as opaque `ses_<32 hex>` values;
|
||||
- a valid native `x-opencode-session` remains stable;
|
||||
- headerless requests still receive a stable fallback;
|
||||
- OpenCode Free behavior is unchanged;
|
||||
- unrelated `DefaultExecutor` providers do not receive the header;
|
||||
- no request state is retained on the shared executor instance.
|
||||
|
||||
Run the focused unit tests first, then the neighboring executor/session tests,
|
||||
the full offline test suite, the application build, and the CLI package build.
|
||||
|
||||
## Delivery
|
||||
|
||||
Build the CLI with `npm --prefix cli run build`, create a package with
|
||||
`npm --prefix cli pack`, and install the generated tarball globally to replace
|
||||
the current npm-installed `9router@0.5.65`. Verify the installed package version
|
||||
and packaged source contains the new executor.
|
||||
|
||||
Upstream issue #3759 already tracks the problem, so no duplicate issue will be
|
||||
created. The pull request will be narrowly scoped to this fix, reference
|
||||
`Fixes #3759`, and explain how it differs from the broader open PR #3780.
|
||||
@@ -118,6 +118,31 @@ Cursor/Cline/Any tool:
|
||||
|
||||
---
|
||||
|
||||
## Cursor / Claude Default Combos
|
||||
|
||||
Cursor and Claude Code send **unprefixed** model IDs (`composer-2.5`, `claude-opus-5`, `opus`), while 9Router routes with provider prefixes (`cu/composer-2.5`, `cc/claude-opus-5`). Default combo generators bridge that gap.
|
||||
|
||||
On **Dashboard → Combos**:
|
||||
|
||||
1. Click **Cursor Default** or **Claude Default**
|
||||
2. Confirm the preview (new vs already-existing names)
|
||||
3. 9Router creates one combo per client model ID, seeded with the matching prefixed route
|
||||
|
||||
**Examples:**
|
||||
|
||||
| Combo name (what the client sends) | Seeded model (what 9Router routes) |
|
||||
|------------------------------------|------------------------------------|
|
||||
| `composer-2.5` | `cu/composer-2.5` |
|
||||
| `cursor-grok-4.6-high-fast` | `cu/cursor-grok-4.6-high-fast` |
|
||||
| `claude-opus-5` | `cc/claude-opus-5` |
|
||||
| `opus` | `cc/claude-opus-5` |
|
||||
|
||||
Existing combo names are **skipped** (not overwritten). Edit any generated combo afterward to add fallbacks. Click the button again later to pick up new catalog IDs.
|
||||
|
||||
> These combos help when Cursor/Claude already talk to 9Router (`/v1` or `ANTHROPIC_BASE_URL`) and send their native model IDs. They do not change Cursor’s built-in Models tab by themselves.
|
||||
|
||||
---
|
||||
|
||||
## Example Combos
|
||||
|
||||
### Example 1: Premium Coding (Subscription → Cheap → Free)
|
||||
|
||||
@@ -111,6 +111,27 @@ Model: cx/gpt-5.2-codex
|
||||
| `cx/gpt-5.2` | GPT 5.2 | General tasks |
|
||||
| `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding |
|
||||
|
||||
### Image Generation
|
||||
|
||||
The Codex image catalog includes `cx/gpt-5.6-sol-image`,
|
||||
`cx/gpt-5.6-terra-image`, and `cx/gpt-5.6-luna-image`, alongside the existing
|
||||
GPT 5.5, 5.4, and 5.3 image aliases. Select them under **Image → OpenAI Codex**
|
||||
in the dashboard, or discover them with `GET /v1/models/image` after connecting
|
||||
a Codex account.
|
||||
|
||||
```bash
|
||||
curl http://localhost:20128/v1/images/generations \
|
||||
-H "Authorization: Bearer $NINE_ROUTER_API_KEY" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"model":"cx/gpt-5.6-sol-image","prompt":"A blue square","size":"1024x1024"}'
|
||||
```
|
||||
|
||||
These are 9Router aliases: the image adapter removes `-image` and sends the
|
||||
underlying model an `image_generation` tool through the Codex Responses API.
|
||||
The same endpoint accepts an `image` reference for edits. Image generation
|
||||
requires an eligible ChatGPT Plus or higher account; availability of each
|
||||
underlying model and its image tool depends on the connected account.
|
||||
|
||||
### Pro Tips
|
||||
|
||||
- **5-hour rolling quota** - Fresh quota every 5 hours
|
||||
|
||||
1445
i18n/README.es.md
Normal file
1445
i18n/README.es.md
Normal file
File diff suppressed because it is too large
Load Diff
1445
i18n/README.fr.md
Normal file
1445
i18n/README.fr.md
Normal file
File diff suppressed because it is too large
Load Diff
1526
i18n/README.pt-BR.md
Normal file
1526
i18n/README.pt-BR.md
Normal file
File diff suppressed because it is too large
Load Diff
@@ -1,21 +1,15 @@
|
||||
Dưới đây là bản dịch tiếng Việt của tài liệu Markdown, giữ nguyên toàn bộ cú pháp và cấu trúc kỹ thuật.
|
||||
|
||||
<div align="center">
|
||||
<img src="../images/9router.png?1" alt="Bảng điều khiển 9Router" width="800"/>
|
||||
|
||||
# 9Router - Free AI Router
|
||||
# 9Router - Free AI Router & Token Saver
|
||||
|
||||
**Không bao giờ ngừng code. Tự động định tuyến tới các mô hình AI MIỄN PHÍ & giá rẻ với cơ chế dự phòng thông minh.**
|
||||
**Không bao giờ ngừng code. Tiết kiệm 20-40% token với RTK + tự động dự phòng sang các mô hình AI MIỄN PHÍ & giá rẻ.**
|
||||
|
||||
**Nhà cung cấp AI Miễn cho OpenClaw.**
|
||||
|
||||
<p align="center">
|
||||
<img src="../public/providers/openclaw.png" alt="OpenClaw" width="80"/>
|
||||
</p>
|
||||
**Kết nối tất cả công cụ AI Code (Claude Code, Codex, Cursor, Cline, Copilot, Antigravity...) tới 40+ Nhà cung cấp AI & 100+ Mô hình.**
|
||||
|
||||
[](https://www.npmjs.com/package/9router)
|
||||
[](https://www.npmjs.com/package/9router)
|
||||
[](https://github.com/decolua/9router/blob/main/LICENSE)
|
||||
[](https://github.com/decolua/9router/blob/main/LICENSE)
|
||||
|
||||
[🚀 Bắt đầu nhanh](#-quick-start) • [💡 Tính năng](#-key-features) • [📖 Cài đặt](#-setup-guide) • [🌐 Website](https://9router.com)
|
||||
</div>
|
||||
@@ -24,19 +18,21 @@ Dưới đây là bản dịch tiếng Việt của tài liệu Markdown, giữ
|
||||
|
||||
## 🤔 Tại sao chọn 9Router?
|
||||
|
||||
**Ngừng lãng phí tiền bạc và gặp phải giới hạn:**
|
||||
**Ngừng lãng phí tiền bạc, token và không bao giờ lo chạm giới hạn (rate limit):**
|
||||
|
||||
- ❌ Hạn mức gói đăng ký hết hạn mỗi tháng mà không dùng hết
|
||||
- ❌ Giới hạn tốc độ (rate limit) ngăn bạn giữaừng khi code
|
||||
- ❌ Các API đắt đỏ ($20-50/tháng cho mỗi nhà cung cấp)
|
||||
- ❌ Phải chuyển đổi thủ công giữa các nhà cung cấp
|
||||
- ❌ Giới hạn tốc độ (rate limit) làm gián đoạn công việc mid-coding
|
||||
- ❌ Kết quả của công cụ (git diff, grep, ls...) ngốn rất nhiều token
|
||||
- ❌ Chi phí API đắt đỏ ($20-50/tháng cho từng nhà cung cấp)
|
||||
- ❌ Phải chuyển đổi thủ công giữa các nhà cung cấp AI
|
||||
|
||||
**9Router giải quyết vấn đề này:**
|
||||
|
||||
- ✅ **Tối đa hóa gói đăng ký** - Theo dõi hạn mức, sử dụng từng bit trước khi reset
|
||||
- ✅ **Tự động dự phòng** - Gói đăng ký → Giá rẻ → Miễn phí, thời gian chết bằng không
|
||||
- ✅ **Đa tài khoản** - Vòng tròn (round-robin) các tài khoản của mỗi nhà cung cấp
|
||||
- ✅ **Phổ quát** - Hoạt động với Claude Code, Codex, Gemini CLI, Cursor, Cline, bất kỳ công cụ CLI nào
|
||||
- ✅ **RTK Token Saver** - Tự động nén nội dung `tool_result`, tiết kiệm 20-40% token trên mỗi request
|
||||
- ✅ **Tối đa hóa gói đăng ký** - Theo dõi hạn mức, tận dụng triệt để trước khi reset
|
||||
- ✅ **Tự động dự phòng (Auto Fallback)** - Gói đăng ký → Giá rẻ → Miễn phí, không lo downtime
|
||||
- ✅ **Đa tài khoản (Multi-account)** - Xoay vòng (round-robin) các tài khoản cho mỗi nhà cung cấp
|
||||
- ✅ **Phổ quát (Universal)** - Hoạt động với Claude Code, Codex, Cursor, Cline, Antigravity và mọi công cụ CLI
|
||||
|
||||
---
|
||||
|
||||
@@ -44,25 +40,26 @@ Dưới đây là bản dịch tiếng Việt của tài liệu Markdown, giữ
|
||||
|
||||
```
|
||||
┌─────────────┐
|
||||
│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
|
||||
│ Tool │
|
||||
│ Công cụ │ (Claude Code, Codex, OpenClaw, Cursor, Cline, Antigravity...)
|
||||
│ CLI AI │
|
||||
└──────┬──────┘
|
||||
│ http://localhost:20128/v1
|
||||
↓
|
||||
┌────────────────────────────────────────┐
|
||||
│ 9Router (Smart Router) │
|
||||
│ • Format translation (OpenAI ↔ Claude) │
|
||||
│ • Quota tracking │
|
||||
│ • Auto token refresh │
|
||||
└──────┬──────────────────────────────────┘
|
||||
┌─────────────────────────────────────────────┐
|
||||
│ 9Router (Smart Router) │
|
||||
│ • RTK Token Saver (nén tool_result token) │
|
||||
│ • Dịch chuyển định dạng (OpenAI ↔ Claude) │
|
||||
│ • Quota tracking (theo dõi hạn mức) │
|
||||
│ • Tự động làm mới OAuth Token │
|
||||
└──────┬──────────────────────────────────────┘
|
||||
│
|
||||
├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
|
||||
│ ↓ quota exhausted
|
||||
├─→ [Tier 2: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ budget limit
|
||||
└─→ [Tier 3: FREE] iFlow, Qwen, Kiro (unlimited)
|
||||
├─→ [Tier 1: GÓI ĐĂNG KÝ] Claude Code, Codex, GitHub Copilot
|
||||
│ ↓ hết hạn mức quota
|
||||
├─→ [Tier 2: GIÁ RẺ] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ ↓ chạm ngân sách
|
||||
└─→ [Tier 3: MIỄN PHÍ] Kiro AI, OpenCode Free, Vertex AI ($300 credits)
|
||||
|
||||
Result: Never stop coding, minimal cost
|
||||
Kết quả: Không bao giờ ngừng code, chi phí tối thiểu + tiết kiệm 20-40% token qua RTK
|
||||
```
|
||||
|
||||
---
|
||||
@@ -76,26 +73,26 @@ npm install -g 9router
|
||||
9router
|
||||
```
|
||||
|
||||
🎉 Bảng điều khiển mở tại `http://localhost:20128`
|
||||
🎉 Bảng điều khiển (Dashboard) sẽ tự động mở tại `http://localhost:20128`
|
||||
|
||||
**2. Kết nối nhà cung cấp MIỄN PHÍ (không cần đăng ký):**
|
||||
|
||||
Bảng điều khiển → Providers -> Kết nối **ude Code** hoặc **Antigravity** -> Đăng nhập OAuth -> Xong!
|
||||
Bảng điều khiển → Providers → Kết nối **Kiro AI** (~50 credits/tháng miễn phí: Claude 4.5 + GLM-5 + MiniMax) hoặc **OpenCode Free** (không cần auth) → Xong!
|
||||
|
||||
**3. Sử dụng trong công cụ CLI của bạn:**
|
||||
|
||||
```
|
||||
Cài đặt Claude Code/Codex/Gemini CLI/OpenClaw/Cursor/Cline:
|
||||
Cài đặt Claude Code/Codex/OpenClaw/Cursor/Cline/Antigravity:
|
||||
Endpoint: http://localhost:20128/v1
|
||||
API Key: [sao chép từ bảng điều khiển]
|
||||
Model: if/kimi-k2-thinking
|
||||
Model: kr/claude-sonnet-4.5
|
||||
```
|
||||
|
||||
**Xong rồi!** Bắt đầu code với các mô hình AI MIỄN PHÍ.
|
||||
**Thế là xong!** Bắt đầu code ngay với các mô hình AI MIỄN PHÍ.
|
||||
|
||||
**Phương án khác: chạy từ nguồn (k lưu trữ này):**
|
||||
**Phương án khác: chạy từ mã nguồn (repository này):**
|
||||
|
||||
Gói kho lưu trữ này là riêng tư (`9router-app`), vì vậy việc thực thi nguồn/Docker là đường dẫn phát triển cục bộ dự kiến.
|
||||
Gói kho lưu trữ này là riêng tư (`9router-app`), vì vậy việc chạy từ nguồn/Docker là cách phát triển cục bộ mặc định.
|
||||
|
||||
```bash
|
||||
cp .env.example .env
|
||||
@@ -111,11 +108,12 @@ PORT=20128 HOSTNAME=0.0.0.0 NEXT_PUBLIC_BASE_URL=http://localhost:20128 npm run
|
||||
```
|
||||
|
||||
URL mặc định:
|
||||
- Bảng điều khiển: `http://localhost:20128/dashboard`
|
||||
- Bảng điều khiển Dashboard: `http://localhost:20128/dashboard`
|
||||
- API tương thích OpenAI: `http://localhost:20128/v1`
|
||||
|
||||
---
|
||||
|
||||
|
||||
## 🎥 Hướng dẫn Video
|
||||
|
||||
<div align="center">
|
||||
|
||||
@@ -42,25 +42,26 @@
|
||||
|
||||
```
|
||||
┌─────────────┐
|
||||
│ Your CLI │ (Claude Code, Codex, Gemini CLI, OpenClaw, Cursor, Cline...)
|
||||
│ Your CLI │ (Claude Code, Codex, OpenClaw, Cursor, Cline, Antigravity...)
|
||||
│ Tool │
|
||||
└──────┬──────┘
|
||||
│ http://localhost:201281
|
||||
│ http://localhost:20128/v1
|
||||
↓
|
||||
┌─────────────────────────────────────────┐
|
||||
│ 9Router (Smart Router) │
|
||||
│ • Format translation (OpenAI ↔ Claude) │
|
||||
│ • Quota tracking │
|
||||
│ • Auto token refresh │
|
||||
└──────┬──────────────────────────────────┘
|
||||
┌─────────────────────────────────────────────┐
|
||||
│ 9Router (Smart Router) │
|
||||
│ • RTK Token Saver (节省 20-40% Token) │
|
||||
│ • 格式转换 (OpenAI ↔ Claude) │
|
||||
│ • 配额追踪 (Quota tracking) │
|
||||
│ • 自动刷新 OAuth Token │
|
||||
└──────┬──────────────────────────────────────┘
|
||||
│
|
||||
├─→ [Tier 1: SUBSCRIPTION] Claude Code, Codex, Gemini CLI
|
||||
│ ↓ quota exhausted
|
||||
├─→ [Tier 2: CHEAP] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ ↓ budget limit
|
||||
└─→ [Tier 3: FREE] iFlow, Qwen, Kiro (unlimited)
|
||||
├─→ [Tier 1: 订阅] Claude Code, Codex, GitHub Copilot
|
||||
│ ↓ 配额用尽
|
||||
├─→ [Tier 2: 低价] GLM ($0.6/1M), MiniMax ($0.2/1M)
|
||||
│ ↓ 触及预算上限
|
||||
└─→ [Tier 3: 免费] Kiro AI, OpenCode Free, Vertex AI ($300 credits)
|
||||
|
||||
Result: Never stop coding, minimal cost
|
||||
结果:永不停歇的编程体验,最低成本 + 通过 RTK 节省 20-40% Token
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
@@ -13,7 +13,14 @@ const proxyClientMaxBodySize = process.env.NINEROUTER_PROXY_CLIENT_MAX_BODY_SIZE
|
||||
const nextConfig = {
|
||||
distDir: process.env.NEXT_DIST_DIR || ".next",
|
||||
output: "standalone",
|
||||
serverExternalPackages: ["better-sqlite3", "sql.js", "node:sqlite", "bun:sqlite"],
|
||||
// `open` must stay external. It derives its own directory from `import.meta.url`, and
|
||||
// webpack replaces that with the absolute path of the BUILD machine as a string literal.
|
||||
// A release built on macOS therefore ships `file:///Users/.../open/index.js`, which
|
||||
// `fileURLToPath` rejects on Windows ("File URL path must be absolute" — no drive
|
||||
// letter). That throw happens at module scope, so every consumer of `open` dies on
|
||||
// import — including xAI/Grok token refresh, which loads the OAuth service that imports
|
||||
// it. Keeping it external preserves the real `import.meta.url` at runtime.
|
||||
serverExternalPackages: ["better-sqlite3", "sql.js", "node:sqlite", "bun:sqlite", "open"],
|
||||
turbopack: {
|
||||
root: tracingRoot
|
||||
},
|
||||
@@ -68,6 +75,10 @@ const nextConfig = {
|
||||
source: "/responses",
|
||||
destination: "/api/v1/responses"
|
||||
},
|
||||
{
|
||||
source: "/systemone",
|
||||
destination: "/api/v1/systemone"
|
||||
},
|
||||
{
|
||||
source: "/v1beta/:path*",
|
||||
destination: "/api/v1beta/:path*"
|
||||
|
||||
@@ -4,7 +4,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
|
||||
|
||||
## Request lifecycle (chat)
|
||||
|
||||
`handlers/chatCore.js` → `services/model.js` `parseModel` (resolve `provider/model`) → **pre-translate hooks** (`rtk/` tool_result compress, `rtk/headroom.js` proxy compress, `rtk/caveman.js` system inject — all fail-open) → `executors/index.js` `getExecutor(provider)` → `translator/index.js` `translateRequest` (client format → provider format) → `executor.execute()` (streams upstream) → `translateResponse` (provider chunks → client format) → SSE out.
|
||||
`handlers/chatCore.js` → `services/model.js` `parseModel` (resolve `provider/model`) → **RTK for `cursor`** (`rtk/` compresses the source-format `tool_result` / `role:tool` in-place — its translator rewrites those shapes, so this one provider must run **before** translate) → `translator/index.js` `translateRequest` (client format → provider format) → **post-translate savers** (`rtk/` compress for every other provider, `rtk/headroom.js` proxy compress, `rtk/caveman.js` / `rtk/ponytail.js` system inject — all fail-open) → `executors/index.js` `getExecutor(provider)` → `executor.execute()` (streams upstream) → `translateResponse` (provider chunks → client format) → SSE out.
|
||||
|
||||
## Directory map
|
||||
|
||||
@@ -16,7 +16,7 @@ Provider-agnostic SSE engine: one OpenAI-style request → any provider (LLM cha
|
||||
- `rtk/` — request token-killer. `index.js` compresses `tool_result` content in-place (OpenAI/Claude/Kiro shapes); `filters/` per-tool compressors + `autodetect.js`; `headroom.js` external compress proxy; `caveman.js` system-prompt injector.
|
||||
- `transformer/` — `responsesTransformer.js` (Chat Completions SSE → Codex Responses API SSE), `streamToJsonConverter.js`.
|
||||
- `shared/` — cross-provider auth/identity: `clineAuth.js`, `machineId.js`, `qoder/`.
|
||||
- `services/` — `model.js`, `provider.js`, `accountFallback.js`, `combo.js`, `compact.js`, `tokenRefresh/`+`tokenRefresh.js`, `oauthCredentialManager.js`, `usage/`, `projectId.js`, `kiroModels.js`/`qoderModels.js`.
|
||||
- `services/` — `model.js`, `provider.js`, `accountFallback.js`, `combo.js`, `tokenRefresh/`+`tokenRefresh.js`, `oauthCredentialManager.js`, `usage/`, `projectId.js`, `kiroModels.js`/`qoderModels.js`.
|
||||
- `utils/` — streamHandler, stream, sse, error, sessionManager, claudeCloaking, clientDetector, proxyFetch (patches global fetch), cursorProtobuf/cursorChecksum, ollamaTransform.
|
||||
|
||||
## Conventions
|
||||
|
||||
@@ -7,6 +7,9 @@ import { createRequire } from "module";
|
||||
export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion;
|
||||
export const GEMINI_CLI_API_CLIENT = PROVIDERS["gemini-cli"]?.apiClient;
|
||||
|
||||
// === Codex CLI === derive từ registry codex.transport
|
||||
export const CODEX_CLI_VERSION = PROVIDERS["codex"]?.cliVersion;
|
||||
|
||||
// Map Node arch to Gemini CLI arch string (x64/x86/arm64/...)
|
||||
function geminiCLIArch() {
|
||||
const a = arch();
|
||||
@@ -156,6 +159,13 @@ export const LOAD_CODE_ASSIST_HEADERS = {
|
||||
"Client-Metadata": JSON.stringify({ ideType: IDE_TYPE.ANTIGRAVITY, platform: getPlatformEnum(), pluginType: PLUGIN_TYPE.GEMINI }),
|
||||
};
|
||||
|
||||
// Real Antigravity IDE doesn't send X-Goog-Api-Client/Client-Metadata on loadCodeAssist/onboardUser —
|
||||
// Google's backend fingerprints those and silently refuses to provision a cloudaicompanionProject.
|
||||
export const ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT,
|
||||
};
|
||||
|
||||
export const LOAD_CODE_ASSIST_METADATA = {
|
||||
ideType: IDE_TYPE.ANTIGRAVITY,
|
||||
platform: getPlatformEnum(),
|
||||
@@ -164,6 +174,18 @@ export const LOAD_CODE_ASSIST_METADATA = {
|
||||
|
||||
// System prompts
|
||||
export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official CLI for Claude.";
|
||||
// Rewrite rules applied to Antigravity system prompts: competing-client branding
|
||||
// makes the backend flag the request and answer 429 Quota Exhausted.
|
||||
export const ANTIGRAVITY_PROMPT_REWRITES = [
|
||||
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
|
||||
{ from: /You are Hermes(?: Agent)?(?:,\s*(?:an intelligent AI assistant|an AI assistant|an AI agent))?(?:,?\s*(?:built|created)\s+by\s+Nous Research)?\./gi, to: "You are an AI assistant." },
|
||||
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
|
||||
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
|
||||
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.
|
||||
{ from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" },
|
||||
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
|
||||
];
|
||||
|
||||
export const ANTIGRAVITY_DEFAULT_SYSTEM = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.**Absolute paths only****Proactiveness**";
|
||||
|
||||
// Derive từ registry oauth.refreshLeadMs
|
||||
@@ -176,7 +198,6 @@ export const OAUTH_ENDPOINTS = {
|
||||
google: { token: "https://oauth2.googleapis.com/token", auth: "https://accounts.google.com/o/oauth2/auth" },
|
||||
openai: { token: PROVIDER_OAUTH["codex"]?.tokenUrl, auth: PROVIDER_OAUTH["codex"]?.authorizeUrl },
|
||||
anthropic: { token: PROVIDER_OAUTH["claude"]?.tokenUrl, auth: "https://api.anthropic.com/v1/oauth/authorize" }, // ≠ claude.authorizeUrl (claude.ai login) — keep
|
||||
qwen: { token: PROVIDER_OAUTH["qwen"]?.tokenUrl, auth: PROVIDER_OAUTH["qwen"]?.deviceCodeUrl },
|
||||
iflow: { token: PROVIDER_OAUTH["iflow"]?.tokenUrl, auth: PROVIDER_OAUTH["iflow"]?.authorizeUrl },
|
||||
github: { token: PROVIDER_OAUTH["github"]?.tokenUrl, auth: PROVIDER_OAUTH["github"]?.authorizeUrl, deviceCode: PROVIDER_OAUTH["github"]?.deviceCodeUrl },
|
||||
};
|
||||
|
||||
@@ -73,6 +73,17 @@ export const ERROR_RULES = [
|
||||
{ status: 403, cooldownMs: COOLDOWN.long },
|
||||
{ status: 404, cooldownMs: COOLDOWN.long },
|
||||
{ status: 429, backoff: true },
|
||||
// --- Request-scoped errors: the request itself is broken — retrying the same
|
||||
// body on another account/model can never succeed, and locking the account
|
||||
// would punish a healthy credential for our own bad request. Callers use this
|
||||
// to fail fast (no account rotation, no model lock).
|
||||
{ text: "context_length_exceeded", requestScoped: true },
|
||||
{ text: "context window", requestScoped: true },
|
||||
{ text: "maximum context length", requestScoped: true },
|
||||
{ text: "prompt is too long", requestScoped: true },
|
||||
{ text: "input is too long", requestScoped: true },
|
||||
{ text: "max_tokens exceed", requestScoped: true },
|
||||
{ text: "reduce the length", requestScoped: true },
|
||||
];
|
||||
|
||||
// Backward compat: COOLDOWN_MS object (used by index.js re-export)
|
||||
|
||||
@@ -2,10 +2,14 @@ import { PROVIDERS } from "./providers.js";
|
||||
import REGISTRY from "../providers/registry/index.js";
|
||||
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
|
||||
import { PROVIDER_MODELS } from "../providers/index.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel, opencodeFamilyFormats } from "../providers/models/helpers.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
export { PROVIDER_MODELS };
|
||||
|
||||
// OpenCode providers sharing the endpoint-family fallback for unknown model ids
|
||||
const isOpenCodeAlias = (aliasOrId) => !aliasOrId || ["oc", "opencode", "ocg", "opencode-go", "ocz", "opencode-zen"].includes(aliasOrId);
|
||||
|
||||
|
||||
// Helper functions
|
||||
export function getProviderModels(aliasOrId) {
|
||||
@@ -26,11 +30,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]);
|
||||
// ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only.
|
||||
function findModel(models, modelId, aliasOrId) {
|
||||
if (!models) return undefined;
|
||||
const found = models.find(m => m.id === modelId);
|
||||
const baseModelId = typeof modelId === "string"
|
||||
? modelId.replace(/\([^()]+\)\s*$/, "").trim()
|
||||
: modelId;
|
||||
const found = models.find(m => m.id === modelId || m.id === baseModelId);
|
||||
if (found) return found;
|
||||
if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined;
|
||||
const normalized = normalizeModelId(modelId);
|
||||
if (normalized === modelId) return undefined;
|
||||
const normalized = normalizeModelId(baseModelId);
|
||||
if (normalized === baseModelId) return undefined;
|
||||
return models.find(m => m.id === normalized);
|
||||
}
|
||||
|
||||
@@ -49,9 +56,29 @@ export function findModelName(aliasOrId, modelId) {
|
||||
}
|
||||
|
||||
export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
if (isOpenCodeAlias(aliasOrId) && isMuseSparkModel(modelId)) {
|
||||
return FORMATS.OPENAI_RESPONSES;
|
||||
}
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelTargetFormat(findModel(models, modelId, aliasOrId));
|
||||
const found = findModel(models, modelId, aliasOrId);
|
||||
if (found) return modelTargetFormat(found);
|
||||
// Family fallback keeps modelsFetcher/passthrough ids on their endpoint lane
|
||||
if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.targetFormat || null;
|
||||
return null;
|
||||
}
|
||||
|
||||
// Declared upstream formats for a model (registry `supportedFormats`). Drives the
|
||||
// per-model guard on the sourceFormat-matched transport; null when undeclared.
|
||||
// Unknown OpenCode ids fall back to the family regex (chat lane by default) so
|
||||
// auto-fetched models never wrongly use the sourceFormat-matched transport.
|
||||
export function getModelSupportedFormats(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
const found = findModel(models, modelId, aliasOrId);
|
||||
if (found) return modelSupportedFormats(found);
|
||||
if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.supportedFormats || [FORMATS.OPENAI];
|
||||
return null;
|
||||
}
|
||||
|
||||
export function getModelType(aliasOrId, modelId) {
|
||||
|
||||
@@ -33,6 +33,21 @@ const GEMINI_VOICES = [
|
||||
"Vindemiatrix", "Sadachbia", "Sadaltager", "Sulafat",
|
||||
].map((id) => ({ id, name: id, type: "tts" }));
|
||||
|
||||
// Xiaomi MiMo preset voices (from https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5).
|
||||
// Voice id is passed via `audio.voice`; `mimo_default` = default (冰糖 on CN cluster, Mia elsewhere).
|
||||
// Voices are language-independent — the spoken language is a separate hint, not bound to the voice.
|
||||
const MIMO_VOICES = [
|
||||
{ id: "mimo_default", name: "mimo_default" },
|
||||
{ id: "冰糖", name: "冰糖" },
|
||||
{ id: "茉莉", name: "茉莉" },
|
||||
{ id: "苏打", name: "苏打" },
|
||||
{ id: "白桦", name: "白桦" },
|
||||
{ id: "Mia", name: "Mia" },
|
||||
{ id: "Chloe", name: "Chloe" },
|
||||
{ id: "Milo", name: "Milo" },
|
||||
{ id: "Dean", name: "Dean" },
|
||||
].map((v) => ({ type: "tts", ...v }));
|
||||
|
||||
// ── TTS Config (config-driven, single source of truth) ─────────────────────
|
||||
export const TTS_MODELS_CONFIG = {
|
||||
openai: {
|
||||
@@ -107,6 +122,14 @@ export const TTS_MODELS_CONFIG = {
|
||||
},
|
||||
allVoices: GEMINI_VOICES,
|
||||
},
|
||||
"xiaomi-mimo": {
|
||||
models: [
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", type: "tts" },
|
||||
],
|
||||
voices: {
|
||||
"mimo-v2.5-tts": MIMO_VOICES,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
// ── Helper: get voices for a specific model ────────────────────────────────
|
||||
|
||||
@@ -1,12 +1,13 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
|
||||
import { cleanJSONSchemaForAntigravity, normalizeGeminiContents } from "../translator/formats/gemini.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
|
||||
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
|
||||
|
||||
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
|
||||
function sanitizeFunctionName(name) {
|
||||
@@ -187,9 +188,12 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
};
|
||||
}
|
||||
|
||||
const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" });
|
||||
const sessionId = toNumericSessionId(rawSessionId) || rawSessionId;
|
||||
|
||||
// ─── Standard (non-image) request ───
|
||||
// Fix contents for Claude models via Antigravity
|
||||
const contents = body.request?.contents?.map(c => {
|
||||
const rawContents = (body.request?.contents || []).map(c => {
|
||||
let role = c.role;
|
||||
// functionResponse must be role "user" for Claude models
|
||||
if (c.parts?.some(p => p.functionResponse)) {
|
||||
@@ -202,21 +206,33 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
return true;
|
||||
});
|
||||
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
|
||||
// don't persist thoughtSignature in their history, so backfill the default signature on any
|
||||
// functionCall part that arrives without one.
|
||||
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
|
||||
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
|
||||
return {
|
||||
...c, role,
|
||||
parts: needsBackfill
|
||||
? parts.map(p => (p.functionCall && !p.thoughtSignature)
|
||||
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
|
||||
: p)
|
||||
: parts,
|
||||
};
|
||||
}
|
||||
return c;
|
||||
// don't persist thoughtSignature in their history, so backfill from cache or default signature.
|
||||
// In parallel function calls, only the first call needs a signature; siblings stay unsigned.
|
||||
let firstFunctionCallSeen = false;
|
||||
const modifiedParts = parts?.map(p => {
|
||||
if (!p.functionCall) return p;
|
||||
const callId = p.functionCall.id;
|
||||
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null;
|
||||
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
|
||||
firstFunctionCallSeen = true;
|
||||
if (callSig) {
|
||||
return { ...p, thoughtSignature: callSig };
|
||||
}
|
||||
if (p.thoughtSignature && !cachedSig) {
|
||||
// Unsigned sibling call
|
||||
const { thoughtSignature: _, ...rest } = p;
|
||||
return rest;
|
||||
}
|
||||
return p;
|
||||
});
|
||||
|
||||
return {
|
||||
...c,
|
||||
role,
|
||||
parts: modifiedParts || parts || [],
|
||||
};
|
||||
});
|
||||
const contents = normalizeGeminiContents(rawContents);
|
||||
|
||||
// Sanitize tool schemas and function names before sending to Antigravity.
|
||||
let tools = body.request?.tools;
|
||||
@@ -245,6 +261,18 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
// Strip tools/toolConfig (handled separately) and blacklisted fields that Google rejects
|
||||
const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {};
|
||||
stripBlacklisted(requestWithoutTools);
|
||||
|
||||
// Rewrite competing-client branding in system prompts (e.g. Zed's Claude prompt,
|
||||
// OpenCode naming) so Antigravity doesn't flag the request with a 429 Quota Exhausted.
|
||||
if (requestWithoutTools.systemInstruction?.parts) {
|
||||
for (const part of requestWithoutTools.systemInstruction.parts) {
|
||||
if (typeof part.text !== "string") continue;
|
||||
for (const { from, to } of ANTIGRAVITY_PROMPT_REWRITES) {
|
||||
part.text = part.text.replaceAll(from, to);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const generationConfig = { ...(requestWithoutTools.generationConfig || {}) };
|
||||
if (generationConfig.maxOutputTokens > MAX_ANTIGRAVITY_OUTPUT_TOKENS) {
|
||||
generationConfig.maxOutputTokens = MAX_ANTIGRAVITY_OUTPUT_TOKENS;
|
||||
@@ -255,7 +283,7 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
generationConfig,
|
||||
...(contents && { contents }),
|
||||
...(tools && { tools }),
|
||||
sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }),
|
||||
sessionId,
|
||||
safetySettings: undefined,
|
||||
...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } })
|
||||
};
|
||||
@@ -265,12 +293,19 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
|
||||
this._lastSessionId = transformedRequest.sessionId; // cached for buildHeaders (base.execute order)
|
||||
|
||||
// Official Antigravity client omits `requestType` entirely on the agent
|
||||
// (chat) path. Sending `requestType: "agent"` here (or leaking it through
|
||||
// from an upstream envelope via the ...body spread below) makes Google
|
||||
// bucket the request and return a detail-free 429 RESOURCE_EXHAUSTED even
|
||||
// with quota available. `image_gen` and
|
||||
// `search` buckets are unaffected and keep their own requestType.
|
||||
delete body.requestType;
|
||||
|
||||
return {
|
||||
...body,
|
||||
project: projectId,
|
||||
model: body.model || model,
|
||||
userAgent: "antigravity",
|
||||
requestType: "agent",
|
||||
requestId: buildIdeRequestId({ body, request: transformedRequest, credentials, model, requestType: "agent" }),
|
||||
request: transformedRequest
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
|
||||
/**
|
||||
* BaseExecutor - Base class for provider executors
|
||||
@@ -31,7 +32,7 @@ export class BaseExecutor {
|
||||
if (this.provider?.startsWith?.("openai-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || OPENAI_COMPAT_BASE;
|
||||
const normalized = baseUrl.replace(/\/$/, "");
|
||||
const path = this.provider.includes("responses") ? "/responses" : "/chat/completions";
|
||||
const path = resolveOpenAICompatibleApiType(this.provider, credentials) === "responses" ? "/responses" : "/chat/completions";
|
||||
return `${normalized}${path}`;
|
||||
}
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
@@ -127,7 +128,7 @@ export class BaseExecutor {
|
||||
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
|
||||
const url = this.buildUrl(model, stream, urlIndex, credentials);
|
||||
const transformedBody = this.transformRequest(model, body, stream, credentials);
|
||||
const headers = this.buildHeaders(credentials, stream, url);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model, transformedBody);
|
||||
|
||||
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
|
||||
|
||||
|
||||
@@ -18,6 +18,35 @@ export class CodeBuddyExecutor extends DefaultExecutor {
|
||||
const transformed = super.transformRequest(model, body, stream, credentials);
|
||||
transformed.stream = true;
|
||||
|
||||
// Tencent's content filter flags CLI agent system prompts ("You are Claude
|
||||
// Code, Anthropic's official CLI...") as prompt injection / sensitive content
|
||||
// and rejects the whole request. Detect agent system prompts (length catch-all
|
||||
// + identity-marker regex) and replace them with a neutral one, while leaving
|
||||
// legitimate user system prompts untouched. content may be a string or typed
|
||||
// blocks ([{type:"text",text}]) depending on the incoming client format, so
|
||||
// flatten before matching and preserve the original shape on replacement.
|
||||
const NEUTRAL_PROMPT = "You are a helpful AI assistant that helps with software engineering tasks.";
|
||||
const AGENT_PATTERN = /you are claude code|claude.?code.+official.+cli|anthropic.+official.+cli|anxthxropic.+official.+cli|you are (?:cursor|windsurf|cline|aider|continue|copilot|cody)|you are an? (?:ai )?(?:coding |code )?agent|cc_entrypoint\s*=\s*(?:cli|vscode|jetbrains|gui)|claude.?code.+issues|give feedback.+claude.?code|you are .{0,30}(?:powerful )?ai agent|orchestration capabilities|OhMyOpenCode|<agent-identity>|<Role>|<Behavior_Instructions>/i;
|
||||
const flatten = (content) =>
|
||||
typeof content === "string"
|
||||
? content
|
||||
: Array.isArray(content)
|
||||
? content.map((b) => (b && typeof b.text === "string" ? b.text : "")).join("\n")
|
||||
: "";
|
||||
if (Array.isArray(transformed.messages)) {
|
||||
transformed.messages = transformed.messages.map((message) => {
|
||||
if (!message || message.role !== "system") return message;
|
||||
const text = flatten(message.content);
|
||||
if (!text) return message;
|
||||
if (text.length > 2000 || AGENT_PATTERN.test(text)) {
|
||||
return typeof message.content === "string"
|
||||
? { ...message, content: NEUTRAL_PROMPT }
|
||||
: { ...message, content: [{ type: "text", text: NEUTRAL_PROMPT }] };
|
||||
}
|
||||
return message;
|
||||
});
|
||||
}
|
||||
|
||||
// CodeBuddy only surfaces model reasoning when the request carries the CLI's
|
||||
// OpenAI-style params: reasoning_effort + reasoning_summary:"auto". 9router's
|
||||
// thinking pipeline sets reasoning_effort only when the client asks, and never
|
||||
|
||||
@@ -23,6 +23,20 @@ export class CodeBuddyIntlExecutor extends DefaultExecutor {
|
||||
} else if (eff) {
|
||||
transformed.reasoning_summary = "auto";
|
||||
}
|
||||
|
||||
// CodeBuddy rejects plain OpenAI shape (11101 invalid request): needs a
|
||||
// leading system prompt + user content as typed blocks, not a bare string.
|
||||
const source = Array.isArray(transformed.messages) ? transformed.messages : [];
|
||||
transformed.messages = [{ role: "system", content: "You are CodeBuddy Code." }];
|
||||
for (const message of source) {
|
||||
if (!message || typeof message !== "object" || ["system", "developer"].includes(message.role)) continue;
|
||||
if (message.role === "user" && typeof message.content === "string") {
|
||||
transformed.messages.push({ ...message, content: [{ type: "text", text: message.content }] });
|
||||
} else {
|
||||
transformed.messages.push({ ...message });
|
||||
}
|
||||
}
|
||||
|
||||
return transformed;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,10 +7,12 @@ import {
|
||||
} from "../services/oauthCredentialManager.js";
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { getModelUpstreamId, getProviderModels } from "../config/providerModels.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { stripCodexUnsupportedPatterns } from "../utils/codexToolSchema.js";
|
||||
|
||||
// SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts.
|
||||
const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
|
||||
@@ -23,6 +25,10 @@ const CODEX_SSE_USER_OUTPUT_PATTERNS = [
|
||||
];
|
||||
const CODEX_SSE_PEEK_BYTES = 256 * 1024;
|
||||
const CODEX_MODEL_CAPACITY_MESSAGE = "Selected model is at capacity. Please try a different model.";
|
||||
function isCodexResponsesLiteModel(model) {
|
||||
const baseId = String(model || "").replace(/\([^()]+\)\s*$/, "");
|
||||
return getProviderModels("cx").some((entry) => entry.id === baseId && entry.responsesLite === true);
|
||||
}
|
||||
|
||||
// Server-generated item id prefixes that Codex /responses cannot resolve when store=false
|
||||
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
|
||||
@@ -41,7 +47,7 @@ const CODEX_PASSTHROUGH_TOOL_TYPES = new Set(["custom"]);
|
||||
const RESPONSES_API_ALLOWLIST = new Set([
|
||||
"model", "input", "instructions", "tools", "tool_choice", "stream", "store",
|
||||
"reasoning", "service_tier", "include", "prompt_cache_key", "client_metadata",
|
||||
"text"
|
||||
"text", "parallel_tool_calls"
|
||||
]);
|
||||
|
||||
// Convert role=system → role=developer in body.input (keeps content in cacheable prefix)
|
||||
@@ -55,13 +61,14 @@ function convertSystemToDeveloperRole(body) {
|
||||
}
|
||||
|
||||
// Strip server-generated item IDs (rs_/fc_/resp_/msg_) from input — avoids 404 with store=false
|
||||
function stripStoredItemReferences(body) {
|
||||
function stripStoredItemReferences(body, preserveLitePrefix = false) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (typeof item === "string" && SERVER_ID_PATTERN.test(item)) return false;
|
||||
if (item && typeof item === "object" && !Array.isArray(item)) {
|
||||
if (item.type === "item_reference") return false;
|
||||
if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)) delete item.id;
|
||||
if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)
|
||||
&& !(preserveLitePrefix && item.role === "developer" && item.id.startsWith("msg_"))) delete item.id;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
@@ -71,6 +78,9 @@ function stripStoredItemReferences(body) {
|
||||
function normalizeCodexTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
// Codex's schema validator has no Unicode property escapes; a `pattern`
|
||||
// carrying `\p{...}` 400s the whole request on every account (#3922).
|
||||
const patternStats = { removed: 0 };
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const type = typeof tool.type === "string" ? tool.type : "";
|
||||
@@ -79,6 +89,9 @@ function normalizeCodexTools(body) {
|
||||
for (const st of tool.tools) {
|
||||
const n = typeof st?.name === "string" ? st.name.trim().slice(0, 128) : "";
|
||||
if (n) validNames.add(n);
|
||||
if (st?.parameters && typeof st.parameters === "object") {
|
||||
st.parameters = stripCodexUnsupportedPatterns(st.parameters, patternStats);
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
@@ -100,10 +113,13 @@ function normalizeCodexTools(body) {
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, 128);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
tool.parameters = stripCodexUnsupportedPatterns(parameters, patternStats);
|
||||
validNames.add(name);
|
||||
return true;
|
||||
});
|
||||
if (patternStats.removed > 0) {
|
||||
dbg("CODEX", `stripped ${patternStats.removed} unsupported tool schema pattern(s)`);
|
||||
}
|
||||
// Drop tool_choice if it references an unknown function name
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
@@ -124,8 +140,13 @@ function resolveCacheSessionId(body, credentials) {
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeReasoningEffort(value) {
|
||||
return value === "max" ? "xhigh" : value;
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (isCodexResponsesLiteModel(model) && (value === "none" || value === "minimal")) return "low";
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
}
|
||||
|
||||
function findNestedMessage(value, depth = 0) {
|
||||
@@ -194,8 +215,11 @@ export class CodexExecutor extends BaseExecutor {
|
||||
* Override headers to add codex-specific identity headers.
|
||||
* transformRequest runs BEFORE buildHeaders, sets this._currentSessionId.
|
||||
*/
|
||||
buildHeaders(credentials, stream = true) {
|
||||
buildHeaders(credentials, stream = true, _url = null, model = null) {
|
||||
const headers = super.buildHeaders(credentials, stream);
|
||||
if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model))) {
|
||||
headers["x-openai-internal-codex-responses-lite"] = "true";
|
||||
}
|
||||
headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default";
|
||||
// Identify client type to Codex backend (matches official codex CLI)
|
||||
if (!headers["originator"]) headers["originator"] = "codex_cli_rs";
|
||||
@@ -393,6 +417,8 @@ export class CodexExecutor extends BaseExecutor {
|
||||
// Convert string input to array format (Codex API requires input as array)
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
const upstreamModel = getModelUpstreamId("cx", body.model || model);
|
||||
const responsesLite = isCodexResponsesLiteModel(upstreamModel);
|
||||
|
||||
// Ensure input is present and non-empty (Codex API rejects empty input)
|
||||
if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) {
|
||||
@@ -402,7 +428,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
// Keep system prompts in body.input as role=developer so they stay in the cacheable prefix
|
||||
convertSystemToDeveloperRole(body);
|
||||
// Strip server-generated item IDs (rs_/fc_/resp_/msg_) — Codex /responses can't resolve when store=false
|
||||
stripStoredItemReferences(body);
|
||||
stripStoredItemReferences(body, responsesLite);
|
||||
// Flatten function tools + drop unsupported types
|
||||
normalizeCodexTools(body);
|
||||
|
||||
@@ -410,7 +436,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
body.stream = true;
|
||||
|
||||
// If no instructions provided, inject default Codex instructions
|
||||
if (!body.instructions || body.instructions.trim() === "") {
|
||||
if (!responsesLite && (!body.instructions || body.instructions.trim() === "")) {
|
||||
body.instructions = CODEX_DEFAULT_INSTRUCTIONS;
|
||||
}
|
||||
|
||||
@@ -423,7 +449,29 @@ export class CodexExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
// Map virtual Codex review models to the upstream Codex model before suffix parsing.
|
||||
body.model = getModelUpstreamId("cx", body.model || model);
|
||||
body.model = upstreamModel;
|
||||
|
||||
if (responsesLite) {
|
||||
// Codex 0.155 carries tools and instructions as input prefix items.
|
||||
const input = Array.isArray(body.input) ? body.input : [body.input];
|
||||
const hasLitePrefix = input.some((item) => item?.type === "additional_tools");
|
||||
if (!hasLitePrefix) {
|
||||
const instructions = typeof body.instructions === "string" && body.instructions.trim()
|
||||
? body.instructions : CODEX_DEFAULT_INSTRUCTIONS;
|
||||
const prefix = [{ type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] }];
|
||||
if (instructions) {
|
||||
prefix.push({ type: "message", role: "developer", content: [{ type: "input_text", text: instructions }] });
|
||||
}
|
||||
input.unshift(...prefix);
|
||||
}
|
||||
body.input = input;
|
||||
body.instructions = "";
|
||||
body.tools = null;
|
||||
body.tool_choice ||= "auto";
|
||||
body.parallel_tool_calls = false;
|
||||
} else {
|
||||
delete body.parallel_tool_calls;
|
||||
}
|
||||
|
||||
// Extract thinking level from model name suffix
|
||||
// e.g., gpt-5.3-codex-high → high, gpt-5.3-codex → medium (default)
|
||||
@@ -440,12 +488,13 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = normalizeReasoningEffort(body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || (responsesLite ? 'medium' : 'low'));
|
||||
body.reasoning = responsesLite ? { effort } : { effort, summary: "auto" };
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort);
|
||||
if (!responsesLite && !body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
if (responsesLite) body.reasoning.context = "all_turns";
|
||||
delete body.reasoning_effort;
|
||||
|
||||
// Include reasoning encrypted content (required by Codex backend for reasoning models)
|
||||
|
||||
@@ -46,203 +46,240 @@ export class CommandCodeExecutor extends BaseExecutor {
|
||||
return headers;
|
||||
}
|
||||
|
||||
async execute(opts) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
result.response = await peekForUpstreamError(result.response, opts.model, {
|
||||
signal: opts.signal,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
async execute(opts) {
|
||||
const maxRetries = 2;
|
||||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
|
||||
const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
|
||||
if (!wrappedResponse.ok && attempt < maxRetries) {
|
||||
const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504;
|
||||
if (isRetryableStatus) {
|
||||
opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`);
|
||||
await new Promise(r => setTimeout(r, 1000 * (attempt + 1)));
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
result.response = wrappedResponse;
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
let parsed = null;
|
||||
try {
|
||||
parsed = JSON.parse(bodyText || "{}");
|
||||
} catch {
|
||||
parsed = null;
|
||||
}
|
||||
const errObj = parsed?.error || parsed;
|
||||
const msg = errObj?.message || parsed?.message || bodyText || response.statusText;
|
||||
const status = Number(errObj?.code || errObj?.statusCode || response.status) || response.status;
|
||||
return {
|
||||
status,
|
||||
message: msg || `CommandCode upstream error: ${response.status}`,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// How long to hold the response open while peeking the first upstream events.
|
||||
// An upstream error event ("Network connection lost") is emitted at stream
|
||||
// start, so the peek is fast; the bound just prevents a slow-started stream
|
||||
// from being held hostage. Env: COMMANDCODE_PEEK_TIMEOUT_MS.
|
||||
const PEEK_TIMEOUT_MS = (() => {
|
||||
const raw = process.env.COMMANDCODE_PEEK_TIMEOUT_MS;
|
||||
const n = raw ? parseInt(raw, 10) : NaN;
|
||||
return Number.isFinite(n) && n > 0 ? n : 10 * 1000;
|
||||
})();
|
||||
export function parseCommandCodeError(event) {
|
||||
if (!event || typeof event !== "object") {
|
||||
return {
|
||||
statusCode: 503,
|
||||
message: "CommandCode upstream error",
|
||||
type: "server_error",
|
||||
};
|
||||
}
|
||||
|
||||
// Event types that count as "the stream has started producing". Everything
|
||||
// else (start, start-step, reasoning-start, text-start, ...) is metadata and
|
||||
// does not end the peek.
|
||||
const MEANINGFUL_EVENT_TYPES = new Set([
|
||||
"text-delta",
|
||||
"reasoning-delta",
|
||||
"tool-input-start",
|
||||
"tool-input-delta",
|
||||
"tool-input-end",
|
||||
"tool-call",
|
||||
"finish-step",
|
||||
"finish",
|
||||
]);
|
||||
const errVal = event.error ?? event.message ?? "unknown";
|
||||
let message = "";
|
||||
let statusCode = null;
|
||||
let type = "server_error";
|
||||
|
||||
function makeAbortError(reason) {
|
||||
const error = new Error(reason?.message || reason || "Request aborted");
|
||||
error.name = "AbortError";
|
||||
return error;
|
||||
if (typeof errVal === "object" && errVal !== null) {
|
||||
message = errVal.message || errVal.error || JSON.stringify(errVal);
|
||||
if (errVal.statusCode && Number.isInteger(Number(errVal.statusCode))) {
|
||||
statusCode = Number(errVal.statusCode);
|
||||
} else if (errVal.status && Number.isInteger(Number(errVal.status))) {
|
||||
statusCode = Number(errVal.status);
|
||||
}
|
||||
if (errVal.type) type = errVal.type;
|
||||
} else if (typeof errVal === "string") {
|
||||
message = errVal;
|
||||
} else {
|
||||
message = JSON.stringify(errVal);
|
||||
}
|
||||
|
||||
if (event.statusCode && Number.isInteger(Number(event.statusCode))) {
|
||||
statusCode = Number(event.statusCode);
|
||||
}
|
||||
|
||||
if (!statusCode || statusCode < 400 || statusCode > 599) {
|
||||
const lower = message.toLowerCase();
|
||||
if (lower.includes("rate limit") || lower.includes("too many requests")) {
|
||||
statusCode = 429;
|
||||
type = "rate_limit_error";
|
||||
} else if (lower.includes("unauthorized") || lower.includes("invalid api key") || lower.includes("authentication")) {
|
||||
statusCode = 401;
|
||||
type = "authentication_error";
|
||||
} else if (lower.includes("payment required") || lower.includes("billing")) {
|
||||
statusCode = 402;
|
||||
type = "billing_error";
|
||||
} else if (lower.includes("quota") || lower.includes("forbidden") || lower.includes("permission")) {
|
||||
statusCode = 403;
|
||||
type = "permission_error";
|
||||
} else if (lower.includes("not found")) {
|
||||
statusCode = 404;
|
||||
type = "invalid_request_error";
|
||||
} else if (lower.includes("unavailable") || lower.includes("overloaded") || lower.includes("server error")) {
|
||||
statusCode = 503;
|
||||
type = "server_error";
|
||||
} else {
|
||||
statusCode = 503;
|
||||
}
|
||||
}
|
||||
|
||||
return { statusCode, message, type };
|
||||
}
|
||||
|
||||
function tryParseEvent(line) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) return null;
|
||||
const json = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
if (!json || json === "[DONE]") return null;
|
||||
try {
|
||||
return JSON.parse(json);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
export async function inspectAndWrapCommandCodeResponse(originalResponse, model) {
|
||||
const reader = originalResponse.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
const rawChunks = [];
|
||||
let detectedError = null;
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const { value, done } = await reader.read();
|
||||
if (done) {
|
||||
const trimmed = buffer.trim();
|
||||
if (trimmed) {
|
||||
try {
|
||||
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
const parsed = JSON.parse(jsonStr);
|
||||
if (parsed?.type === "error") {
|
||||
detectedError = parsed;
|
||||
}
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
rawChunks.push(value);
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() || "";
|
||||
|
||||
let stopLoop = false;
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
if (!jsonStr || jsonStr === "[DONE]") {
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
|
||||
let event;
|
||||
try {
|
||||
event = JSON.parse(jsonStr);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (event?.type === "error") {
|
||||
detectedError = event;
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (
|
||||
event?.type === "text-delta" ||
|
||||
event?.type === "reasoning-delta" ||
|
||||
event?.type === "tool-input-start" ||
|
||||
event?.type === "tool-call" ||
|
||||
event?.type === "finish" ||
|
||||
event?.type === "finish-step"
|
||||
) {
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (stopLoop) break;
|
||||
}
|
||||
} catch {
|
||||
try { reader.releaseLock(); } catch { /* ignore */ }
|
||||
return originalResponse;
|
||||
}
|
||||
|
||||
if (detectedError) {
|
||||
try { await reader.cancel(); } catch { /* ignore */ }
|
||||
const { statusCode, message, type } = parseCommandCodeError(detectedError);
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
error: {
|
||||
message: `[CommandCode error: ${message}]`,
|
||||
type,
|
||||
code: statusCode,
|
||||
},
|
||||
}),
|
||||
{
|
||||
status: statusCode,
|
||||
statusText: statusCode === 503 ? "Service Unavailable" : (statusCode === 429 ? "Too Many Requests" : "Bad Gateway"),
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
const combinedStream = createRawReplayedStream(rawChunks, reader);
|
||||
return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse);
|
||||
}
|
||||
|
||||
function formatErrorValue(errVal) {
|
||||
const errStr =
|
||||
typeof errVal === "string"
|
||||
? errVal
|
||||
: typeof errVal?.message === "string"
|
||||
? errVal.message
|
||||
: JSON.stringify(errVal);
|
||||
const errType =
|
||||
typeof errVal === "string"
|
||||
? "upstream_error"
|
||||
: errVal?.type || "upstream_error";
|
||||
return { message: errStr, type: errType };
|
||||
function createRawReplayedStream(rawChunks, reader) {
|
||||
let chunkIndex = 0;
|
||||
|
||||
return new ReadableStream({
|
||||
async pull(controller) {
|
||||
if (chunkIndex < rawChunks.length) {
|
||||
controller.enqueue(rawChunks[chunkIndex++]);
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const { value, done } = await reader.read();
|
||||
if (done) {
|
||||
controller.close();
|
||||
} else {
|
||||
controller.enqueue(value);
|
||||
}
|
||||
} catch (err) {
|
||||
controller.error(err);
|
||||
}
|
||||
},
|
||||
async cancel(reason) {
|
||||
try {
|
||||
await reader.cancel(reason);
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the first upstream events before committing the response.
|
||||
*
|
||||
* - `{"type":"error"}` as the first meaningful event → return a 502 Response so
|
||||
* chatCore's `!response.ok` path parses the error and triggers fallback.
|
||||
* - Otherwise → re-emit the buffered bytes + the rest of the stream through the
|
||||
* normal NDJSON → OpenAI SSE wrapper and return it untouched in spirit.
|
||||
*
|
||||
* Bounded by `timeoutMs` (default PEEK_TIMEOUT_MS): if no meaningful event
|
||||
* arrives in time, or the request signal aborts, we commit whatever we have and
|
||||
* let the regular stream pipeline (stall detection, abort handling) take over.
|
||||
*/
|
||||
export async function peekForUpstreamError(
|
||||
originalResponse,
|
||||
model,
|
||||
{ signal = null, timeoutMs = PEEK_TIMEOUT_MS } = {},
|
||||
) {
|
||||
const reader = originalResponse.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
const abortController = new AbortController();
|
||||
const forwardAbort = () => abortController.abort(signal?.reason);
|
||||
if (signal?.aborted) abortController.abort(signal?.reason);
|
||||
else if (signal)
|
||||
signal.addEventListener("abort", forwardAbort, { once: true });
|
||||
|
||||
// Raw bytes for lossless re-emission; decoded text is only used for line
|
||||
// parsing / error detection. Never re-encode decoded text: TextDecoder
|
||||
// holds a split multi-byte char internally and flush() would replace it
|
||||
// with U+FFFD, corrupting the stream.
|
||||
const rawChunks = [];
|
||||
let peeked = "";
|
||||
let errorEvent = null;
|
||||
let committed = false;
|
||||
|
||||
const readWithTimeout = (ms) => {
|
||||
if (abortController.signal.aborted) {
|
||||
return Promise.reject(makeAbortError(abortController.signal.reason));
|
||||
}
|
||||
const timeoutPromise = new Promise((_, reject) => {
|
||||
const t = setTimeout(() => reject(new Error("peek timeout")), ms);
|
||||
t.unref?.();
|
||||
});
|
||||
const abortPromise = new Promise((_, reject) => {
|
||||
abortController.signal.addEventListener(
|
||||
"abort",
|
||||
() => reject(makeAbortError(abortController.signal.reason)),
|
||||
{ once: true },
|
||||
);
|
||||
});
|
||||
return Promise.race([reader.read(), timeoutPromise, abortPromise]);
|
||||
};
|
||||
|
||||
try {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (!errorEvent && !committed && Date.now() < deadline) {
|
||||
const { done, value } = await readWithTimeout(
|
||||
Math.max(deadline - Date.now(), 1),
|
||||
);
|
||||
if (done) break;
|
||||
rawChunks.push(value);
|
||||
peeked += decoder.decode(value, { stream: true });
|
||||
const lines = peeked.split("\n");
|
||||
// The last segment may be a partial line — only parse complete ones.
|
||||
for (const line of lines.slice(0, -1)) {
|
||||
const event = tryParseEvent(line);
|
||||
if (!event?.type) continue;
|
||||
if (event.type === "error") {
|
||||
errorEvent = event;
|
||||
break;
|
||||
}
|
||||
if (MEANINGFUL_EVENT_TYPES.has(event.type)) {
|
||||
committed = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// timeout / abort / read failure during the peek → commit whatever we have;
|
||||
// the downstream stream pipeline (stall detection, abort handling) takes over.
|
||||
}
|
||||
|
||||
if (signal) signal.removeEventListener("abort", forwardAbort);
|
||||
|
||||
if (errorEvent) {
|
||||
await reader.cancel("commandcode early error detected").catch(() => {});
|
||||
const { message, type } = formatErrorValue(
|
||||
errorEvent.error ?? errorEvent.message ?? "unknown",
|
||||
);
|
||||
return new Response(JSON.stringify({ error: { message, type } }), {
|
||||
status: HTTP_STATUS.BAD_GATEWAY,
|
||||
statusText: message.slice(0, 200),
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
const remaining = new ReadableStream({
|
||||
start(controller) {
|
||||
(async () => {
|
||||
try {
|
||||
// Re-emit RAW bytes (never re-encoded decoded text) so split
|
||||
// multi-byte UTF-8 sequences survive the peek untouched.
|
||||
for (const c of rawChunks) controller.enqueue(c);
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
controller.enqueue(value);
|
||||
}
|
||||
controller.close();
|
||||
} catch (err) {
|
||||
controller.error(err);
|
||||
}
|
||||
})();
|
||||
},
|
||||
cancel() {
|
||||
reader.cancel("commandcode stream cancelled").catch(() => {});
|
||||
},
|
||||
});
|
||||
|
||||
const combined = new Response(remaining, {
|
||||
status: originalResponse.status,
|
||||
statusText: originalResponse.statusText,
|
||||
headers: originalResponse.headers,
|
||||
});
|
||||
return wrapNdjsonAsOpenAISse(combined, model);
|
||||
}
|
||||
|
||||
function wrapNdjsonAsOpenAISse(originalResponse, model) {
|
||||
const decoder = new TextDecoder();
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
const state = { model };
|
||||
function wrapNdjsonAsOpenAISse(streamBody, model, originalResponse = null) {
|
||||
const decoder = new TextDecoder();
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
const state = { model };
|
||||
|
||||
const emitChunks = (chunks, controller) => {
|
||||
if (!chunks) return;
|
||||
@@ -253,33 +290,38 @@ function wrapNdjsonAsOpenAISse(originalResponse, model) {
|
||||
}
|
||||
};
|
||||
|
||||
const transform = new TransformStream({
|
||||
transform(chunk, controller) {
|
||||
buffer += decoder.decode(chunk, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() || "";
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
// Translate AI SDK v5 NDJSON line to one or more OpenAI chunks
|
||||
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
|
||||
}
|
||||
},
|
||||
flush(controller) {
|
||||
const trimmed = buffer.trim();
|
||||
if (trimmed) {
|
||||
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
|
||||
}
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
},
|
||||
});
|
||||
const transform = new TransformStream({
|
||||
transform(chunk, controller) {
|
||||
buffer += decoder.decode(chunk, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() || "";
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
|
||||
}
|
||||
},
|
||||
flush(controller) {
|
||||
const trimmed = buffer.trim();
|
||||
if (trimmed) {
|
||||
emitChunks(commandCodeToOpenAIResponse(trimmed, state), controller);
|
||||
}
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
},
|
||||
});
|
||||
|
||||
const newBody = originalResponse.body.pipeThrough(transform);
|
||||
return new Response(newBody, {
|
||||
status: originalResponse.status,
|
||||
statusText: originalResponse.statusText,
|
||||
headers: originalResponse.headers,
|
||||
});
|
||||
const newBody = streamBody.pipeThrough(transform);
|
||||
return new Response(newBody, {
|
||||
status: originalResponse?.status || 200,
|
||||
statusText: originalResponse?.statusText || "OK",
|
||||
headers: {
|
||||
"Content-Type": "text/event-stream",
|
||||
"Cache-Control": "no-cache",
|
||||
"Connection": "keep-alive",
|
||||
...(originalResponse?.headers ? Object.fromEntries(originalResponse.headers.entries()) : {}),
|
||||
"content-type": "text/event-stream",
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export default CommandCodeExecutor;
|
||||
|
||||
@@ -7,13 +7,16 @@ import {
|
||||
wrapConnectRPCFrame,
|
||||
decodeMessage,
|
||||
parseConnectRPCFrame,
|
||||
extractTextFromResponse
|
||||
extractTextFromResponse,
|
||||
encodeMcpTools,
|
||||
decodeMcpArgs,
|
||||
} from "../utils/cursorProtobuf.js";
|
||||
import { buildCursorHeaders } from "../utils/cursorChecksum.js";
|
||||
import { estimateUsage } from "../utils/usageTracking.js";
|
||||
import { SSE_DONE, SSE_HEADERS } from "../utils/sseConstants.js";
|
||||
import { chatChunkSse, sseChunk } from "../utils/sse.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { ROLE, OPENAI_BLOCK } from "../translator/schema/index.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import zlib from "zlib";
|
||||
import crypto from "crypto";
|
||||
@@ -65,55 +68,74 @@ function textFromContent(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (!Array.isArray(content)) return "";
|
||||
return content
|
||||
.filter((part) => part?.type === "text" && typeof part.text === "string")
|
||||
.filter((part) => part?.type === OPENAI_BLOCK.TEXT && typeof part.text === "string")
|
||||
.map((part) => part.text)
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
function isAgentTextRequest(body) {
|
||||
// Many compatible clients always attach their built-in tool schemas, even
|
||||
// for a normal text turn. Cursor's retired ChatService rejects those
|
||||
// requests; AgentService can still answer the text turn, so ignore schemas
|
||||
// here. A real tool-call/result conversation is kept on the legacy path
|
||||
// until its AgentService tool protocol is implemented.
|
||||
return Array.isArray(body?.messages) && body.messages.every((message) => {
|
||||
if (message?.tool_calls?.length || message?.role === "tool") return false;
|
||||
return typeof message?.content === "string"
|
||||
|| Array.isArray(message?.content) && message.content.every((part) => part?.type === "text");
|
||||
function isTextPart(part) {
|
||||
return !part || part.type === OPENAI_BLOCK.TEXT || typeof part === "string";
|
||||
}
|
||||
|
||||
export function isAgentCapableRequest(body) {
|
||||
// ChatService rejects auto/composer and most thinking variants. AgentService
|
||||
// can answer text turns (including declared tool schemas) and tool-call
|
||||
// history. Image parts still need the legacy protobuf path.
|
||||
if (!Array.isArray(body?.messages) || body.messages.length === 0) return false;
|
||||
return body.messages.every((message) => {
|
||||
if (Array.isArray(message?.content)) return message.content.every(isTextPart);
|
||||
return message?.content == null || typeof message.content === "string";
|
||||
});
|
||||
}
|
||||
|
||||
function encodeHistoryMessage(message) {
|
||||
const content = textFromContent(message?.content);
|
||||
if (!content) return null;
|
||||
const extras = [];
|
||||
if (message?.role === ROLE.ASSISTANT && message.tool_calls?.length) {
|
||||
for (const tc of message.tool_calls) {
|
||||
extras.push(`[tool_call id=${tc.id || ""} name=${tc.function?.name || "tool"} args=${tc.function?.arguments || "{}"}]`);
|
||||
}
|
||||
}
|
||||
if (message?.role === ROLE.TOOL) {
|
||||
extras.push(`[tool_result id=${message.tool_call_id || ""}]`);
|
||||
}
|
||||
const textBody = [content, ...extras].filter(Boolean).join("\n");
|
||||
if (!textBody) return null;
|
||||
|
||||
// ConversationHistoryMessage.user / .assistant -> repeated content -> text.
|
||||
const text = agentString(1, content);
|
||||
if (message.role === "assistant") {
|
||||
const text = agentString(1, textBody);
|
||||
if (message.role === ROLE.ASSISTANT) {
|
||||
return agentMessage(2, agentMessage(1, agentMessage(1, text)));
|
||||
}
|
||||
return agentMessage(1, agentMessage(1, agentMessage(1, text)));
|
||||
}
|
||||
|
||||
function buildAgentRunFrame(messages, model) {
|
||||
export function buildAgentRunFrame(messages, model, tools = []) {
|
||||
// custom_system_prompt (RunRequest field 8) makes AgentService return an
|
||||
// empty turn. Fold system text into the current user message instead.
|
||||
const system = messages
|
||||
.filter((message) => message?.role === "system")
|
||||
.filter((message) => message?.role === ROLE.SYSTEM)
|
||||
.map((message) => textFromContent(message.content))
|
||||
.filter(Boolean)
|
||||
.join("\n\n");
|
||||
const chatMessages = messages.filter((message) => message?.role !== "system");
|
||||
const currentIndex = [...chatMessages].map((message) => message?.role).lastIndexOf("user");
|
||||
const chatMessages = messages.filter((message) => message?.role !== ROLE.SYSTEM);
|
||||
const currentIndex = [...chatMessages].map((message) => message?.role).lastIndexOf(ROLE.USER);
|
||||
const current = currentIndex >= 0 ? chatMessages[currentIndex] : chatMessages.at(-1);
|
||||
const history = chatMessages
|
||||
.slice(0, currentIndex >= 0 ? currentIndex : -1)
|
||||
.map(encodeHistoryMessage)
|
||||
.filter(Boolean);
|
||||
const userText = textFromContent(current?.content) || "Continue.";
|
||||
const rawUser = textFromContent(current?.content) || "Continue.";
|
||||
const userText = system ? `${system}\n\n${rawUser}` : rawUser;
|
||||
|
||||
// agent.v1.UserMessageAction.user_message and its optional history.
|
||||
// selected_context (3) + mode=1 (4) match cursor-agent's wire format; without
|
||||
// them the server may accept the RPC and stream an empty turn.
|
||||
const userMessage = concatBuffers(
|
||||
agentString(1, userText),
|
||||
agentString(2, crypto.randomUUID()),
|
||||
agentMessage(3, new Uint8Array()),
|
||||
encodeField(4, PROTOBUF_VARINT, 1),
|
||||
);
|
||||
const conversationHistory = history.length
|
||||
? concatBuffers(...history.map((entry) => agentMessage(1, entry)))
|
||||
@@ -124,11 +146,20 @@ function buildAgentRunFrame(messages, model) {
|
||||
);
|
||||
const conversationAction = agentMessage(1, userAction);
|
||||
const requestedModel = concatBuffers(agentString(1, model), agentBool(7, true));
|
||||
// ModelDetails (field 3): thinking variants (Composer, Grok, *-thinking)
|
||||
// return an empty turn when only RequestedModel (field 9) is set.
|
||||
const modelDetails = concatBuffers(
|
||||
agentString(1, model),
|
||||
agentString(3, model),
|
||||
agentString(4, model),
|
||||
);
|
||||
const mcpTools = encodeMcpTools(tools);
|
||||
const runRequest = concatBuffers(
|
||||
// An empty ConversationStateStructure starts a fresh local agent session.
|
||||
agentMessage(1, new Uint8Array()),
|
||||
agentMessage(2, conversationAction),
|
||||
...(system ? [agentString(8, system)] : []),
|
||||
agentMessage(3, modelDetails),
|
||||
...(mcpTools.length ? [agentMessage(4, mcpTools)] : []),
|
||||
agentMessage(9, requestedModel),
|
||||
);
|
||||
|
||||
@@ -157,13 +188,51 @@ function decodeAgentFrames(buffer, onFrame) {
|
||||
return pending;
|
||||
}
|
||||
|
||||
function createRequestContextResponse() {
|
||||
// AgentService asks every run for client context. 9router has no IDE file
|
||||
// context, so acknowledge with an empty RequestContext.
|
||||
function execIds(execRequest) {
|
||||
const id = Number(execRequest?.get(1)?.[0]?.value || 0);
|
||||
const execId = extractAgentString(execRequest, 15);
|
||||
return { id, execId };
|
||||
}
|
||||
|
||||
function wrapExecClientMessage(execMsgId, execId, resultField, resultPayload) {
|
||||
const parts = [];
|
||||
if (execMsgId) parts.push(encodeField(1, PROTOBUF_VARINT, execMsgId));
|
||||
parts.push(agentString(15, execId || ""));
|
||||
parts.push(encodeField(resultField, PROTOBUF_LEN, resultPayload || new Uint8Array()));
|
||||
return wrapConnectRPCFrame(agentMessage(2, concatBuffers(...parts)));
|
||||
}
|
||||
|
||||
function createRequestContextResponse(execRequest) {
|
||||
// Tools already go out on AgentRunRequest.mcp_tools. Echoing them again on
|
||||
// this ack makes AgentService stall silently (0 SSE bytes until abort).
|
||||
const { id, execId } = execIds(execRequest);
|
||||
const requestContextSuccess = agentMessage(1, new Uint8Array());
|
||||
const requestContextResult = agentMessage(1, requestContextSuccess);
|
||||
const execClientMessage = agentMessage(10, requestContextResult);
|
||||
return wrapConnectRPCFrame(agentMessage(2, execClientMessage));
|
||||
return wrapExecClientMessage(id, execId, 10, requestContextResult);
|
||||
}
|
||||
|
||||
// ExecServerMessage variant → ExecClientMessage result field (same numbers).
|
||||
const EXEC_RESULT_FIELD = {
|
||||
2: 2, 3: 3, 4: 4, 5: 5, 7: 7, 8: 8, 9: 9, 16: 16, 20: 20, 23: 23,
|
||||
};
|
||||
|
||||
function rejectExecRequest(execRequest) {
|
||||
const { id, execId } = execIds(execRequest);
|
||||
const variant = [...(execRequest?.keys?.() || [])].find((field) => field !== 1 && field !== 15);
|
||||
const resultField = EXEC_RESULT_FIELD[variant];
|
||||
if (!resultField) return null;
|
||||
// Diagnostics has no rejected variant — empty success unblocks the stream.
|
||||
if (variant === 9) return wrapExecClientMessage(id, execId, 9, new Uint8Array());
|
||||
const rejected = agentMessage(2, agentString(2, "Tool not available in this environment. Use the MCP tools provided instead."));
|
||||
return wrapExecClientMessage(id, execId, resultField, rejected);
|
||||
}
|
||||
|
||||
function encodeKvClientMessage(kvId, resultField, resultPayload, metadata) {
|
||||
const parts = [];
|
||||
if (kvId) parts.push(encodeField(1, PROTOBUF_VARINT, kvId));
|
||||
parts.push(encodeField(resultField, PROTOBUF_LEN, resultPayload || new Uint8Array()));
|
||||
if (metadata && metadata.length) parts.push(encodeField(4, PROTOBUF_LEN, metadata));
|
||||
return wrapConnectRPCFrame(agentMessage(3, concatBuffers(...parts)));
|
||||
}
|
||||
|
||||
const CURSOR_STREAM_DEBUG = process.env.CURSOR_STREAM_DEBUG === "1";
|
||||
@@ -479,7 +548,7 @@ export class CursorExecutor extends BaseExecutor {
|
||||
};
|
||||
}
|
||||
|
||||
async executeAgent({ model, body, stream, credentials, signal }) {
|
||||
async executeAgent({ model, body, stream, credentials, signal, log }) {
|
||||
const agentEndpoint = PROVIDER_OAUTH.cursor?.agentEndpoint;
|
||||
if (!agentEndpoint) throw new Error("Cursor AgentService endpoint is not configured");
|
||||
|
||||
@@ -491,9 +560,10 @@ export class CursorExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
let session;
|
||||
const tools = body.tools || [];
|
||||
try {
|
||||
session = this.openAgentHttp2Stream(url, headers, requestController.signal);
|
||||
session.write(buildAgentRunFrame(body.messages || [], model));
|
||||
session.write(buildAgentRunFrame(body.messages || [], model, tools));
|
||||
} catch (error) {
|
||||
throw new Error(`Cursor AgentService request failed: ${error.message}`);
|
||||
}
|
||||
@@ -533,8 +603,23 @@ export class CursorExecutor extends BaseExecutor {
|
||||
// so strict clients such as Claude Code accept the completed stream.
|
||||
const responseId = `chatcmpl-msg_${Date.now()}`;
|
||||
const created = Math.floor(Date.now() / 1000);
|
||||
const composerModel = isComposerModel(model);
|
||||
let pending = Buffer.alloc(0);
|
||||
let finished = false;
|
||||
let thinkingAcc = "";
|
||||
let emittedVisible = 0;
|
||||
let emittedText = false;
|
||||
|
||||
const flushThinkingFallback = (onEvent) => {
|
||||
if (emittedText || !thinkingAcc) return;
|
||||
const fallback = composerModel
|
||||
? visibleComposerContentFromThinking(thinkingAcc)
|
||||
: thinkingAcc.trim();
|
||||
if (fallback) {
|
||||
emittedText = true;
|
||||
onEvent({ type: "text", value: fallback });
|
||||
}
|
||||
};
|
||||
|
||||
const consume = async (onEvent) => {
|
||||
try {
|
||||
@@ -553,32 +638,87 @@ export class CursorExecutor extends BaseExecutor {
|
||||
const update = decodeMessage(serverMessage.get(1)[0].value);
|
||||
if (update.has(1)) {
|
||||
const textDelta = extractAgentString(decodeMessage(update.get(1)[0].value), 1);
|
||||
if (textDelta) onEvent({ type: "text", value: textDelta });
|
||||
if (textDelta) {
|
||||
emittedText = true;
|
||||
onEvent({ type: "text", value: textDelta });
|
||||
}
|
||||
}
|
||||
// Cursor's AgentService emits internal reasoning without the
|
||||
// cryptographic signature required by Anthropic thinking blocks.
|
||||
// Forwarding it makes strict Anthropic clients (Claude Code)
|
||||
// discard or wait on an otherwise complete response. Keep the
|
||||
// reasoning upstream-only and emit the normal answer text.
|
||||
// thinking_delta (field 4). Composer (and some Grok variants) put
|
||||
// the visible answer after </think> here and never send text_delta.
|
||||
if (update.has(4)) {
|
||||
const thinkingDelta = extractAgentString(decodeMessage(update.get(4)[0].value), 1);
|
||||
if (thinkingDelta) {
|
||||
thinkingAcc += thinkingDelta;
|
||||
if (composerModel) {
|
||||
const visible = visibleComposerContentFromThinking(thinkingAcc);
|
||||
if (visible.length > emittedVisible) {
|
||||
const deltaContent = visible.slice(emittedVisible);
|
||||
emittedVisible = visible.length;
|
||||
emittedText = true;
|
||||
onEvent({ type: "text", value: deltaContent });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Keep unsigned reasoning upstream-only for Anthropic clients.
|
||||
if (update.has(14)) {
|
||||
flushThinkingFallback(onEvent);
|
||||
finished = true;
|
||||
onEvent({ type: "done" });
|
||||
}
|
||||
}
|
||||
|
||||
// KvServerMessage (field 4): get/set blob. Ack so the stream proceeds.
|
||||
if (serverMessage.has(4)) {
|
||||
const kv = decodeMessage(serverMessage.get(4)[0].value);
|
||||
const kvId = kv.get(1)?.[0]?.value || 0;
|
||||
const metadata = kv.get(4)?.[0]?.value || null;
|
||||
if (kv.has(2)) {
|
||||
session.write(encodeKvClientMessage(kvId, 2, agentMessage(1, new Uint8Array()), metadata));
|
||||
} else if (kv.has(3)) {
|
||||
session.write(encodeKvClientMessage(kvId, 3, new Uint8Array(), metadata));
|
||||
}
|
||||
}
|
||||
|
||||
// AgentService requests IDE context before producing a response.
|
||||
// Return an empty context; 9router is not coupled to an editor.
|
||||
if (serverMessage.has(2)) {
|
||||
const execRequest = decodeMessage(serverMessage.get(2)[0].value);
|
||||
if (execRequest.has(10)) {
|
||||
session.write(createRequestContextResponse());
|
||||
log?.info?.("CURSOR", "AgentService request_context ack");
|
||||
session.write(createRequestContextResponse(execRequest));
|
||||
} else if (execRequest.has(11)) {
|
||||
const mcp = decodeMcpArgs(execRequest.get(11)[0].value);
|
||||
const name = mcp.toolName || mcp.name;
|
||||
if (name) {
|
||||
log?.info?.("CURSOR", `AgentService MCP tool_call ${name}`);
|
||||
finished = true;
|
||||
onEvent({
|
||||
type: "tool_call",
|
||||
value: {
|
||||
id: mcp.toolCallId || `call_${crypto.randomUUID()}`,
|
||||
name,
|
||||
arguments: JSON.stringify(mcp.args || {}),
|
||||
},
|
||||
});
|
||||
onEvent({ type: "done", finishReason: "tool_calls" });
|
||||
} else {
|
||||
debugLog(`[CURSOR AGENT] Unsupported exec request fields: ${[...execRequest.keys()].join(",")}`);
|
||||
finished = true;
|
||||
onEvent({ type: "error", value: "Cursor AgentService requested an unsupported IDE tool" });
|
||||
}
|
||||
} else {
|
||||
// Every other ExecServerMessage variant is an editor-backed tool
|
||||
// (shell, read, write, …) that 9router cannot service. Fail the
|
||||
// turn rather than narrating protocol state as assistant text.
|
||||
debugLog(`[CURSOR AGENT] Unsupported exec request fields: ${[...execRequest.keys()].join(",")}`);
|
||||
finished = true;
|
||||
onEvent({ type: "error", value: "Cursor AgentService requested an unsupported IDE tool" });
|
||||
// Auto/Composer often probe IDE builtins (shell/read/…). Reject
|
||||
// them so the model can continue with MCP tools or a text answer
|
||||
// instead of stalling the h2 stream.
|
||||
const rejection = rejectExecRequest(execRequest);
|
||||
if (rejection) {
|
||||
log?.info?.("CURSOR", `AgentService rejected IDE exec fields=${[...execRequest.keys()].join(",")}`);
|
||||
session.write(rejection);
|
||||
} else {
|
||||
debugLog(`[CURSOR AGENT] Unsupported exec request fields: ${[...execRequest.keys()].join(",")}`);
|
||||
finished = true;
|
||||
onEvent({ type: "error", value: "Cursor AgentService requested an unsupported IDE tool" });
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -586,7 +726,10 @@ export class CursorExecutor extends BaseExecutor {
|
||||
} finally {
|
||||
try { session.end(); } catch {}
|
||||
try { session.close(); } catch {}
|
||||
if (!finished) onEvent({ type: "done" });
|
||||
if (!finished) {
|
||||
flushThinkingFallback(onEvent);
|
||||
onEvent({ type: "done" });
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -594,10 +737,21 @@ export class CursorExecutor extends BaseExecutor {
|
||||
let content = "";
|
||||
let reasoning = "";
|
||||
let agentError = null;
|
||||
const toolCalls = [];
|
||||
let finishReason = "stop";
|
||||
await consume((event) => {
|
||||
if (event.type === "text") content += event.value;
|
||||
else if (event.type === "thinking") reasoning += event.value;
|
||||
else if (event.type === "tool_call") {
|
||||
toolCalls.push({
|
||||
id: event.value.id,
|
||||
type: "function",
|
||||
function: { name: event.value.name, arguments: event.value.arguments },
|
||||
});
|
||||
finishReason = "tool_calls";
|
||||
}
|
||||
else if (event.type === "error") agentError = event.value;
|
||||
else if (event.type === "done" && event.finishReason) finishReason = event.finishReason;
|
||||
});
|
||||
if (agentError) {
|
||||
return {
|
||||
@@ -611,13 +765,19 @@ export class CursorExecutor extends BaseExecutor {
|
||||
responseFormat: FORMATS.OPENAI,
|
||||
};
|
||||
}
|
||||
const message = {
|
||||
role: "assistant",
|
||||
content: content || null,
|
||||
...(reasoning ? { reasoning_content: reasoning } : {}),
|
||||
...(toolCalls.length ? { tool_calls: toolCalls } : {}),
|
||||
};
|
||||
return {
|
||||
response: new Response(JSON.stringify({
|
||||
id: responseId,
|
||||
object: "chat.completion",
|
||||
created,
|
||||
model,
|
||||
choices: [{ index: 0, message: { role: "assistant", content: content || null, ...(reasoning ? { reasoning_content: reasoning } : {}) }, finish_reason: "stop" }],
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
usage: estimateUsage(body, content.length, FORMATS.OPENAI),
|
||||
}), { headers: { "Content-Type": "application/json" } }),
|
||||
url,
|
||||
@@ -635,6 +795,18 @@ export class CursorExecutor extends BaseExecutor {
|
||||
controller.enqueue(encoder.encode(chatChunkSse({ id: responseId, created, model, delta: { content: event.value } })));
|
||||
} else if (event.type === "thinking") {
|
||||
controller.enqueue(encoder.encode(chatChunkSse({ id: responseId, created, model, delta: { reasoning_content: event.value } })));
|
||||
} else if (event.type === "tool_call") {
|
||||
controller.enqueue(encoder.encode(chatChunkSse({
|
||||
id: responseId, created, model,
|
||||
delta: {
|
||||
tool_calls: [{
|
||||
index: 0,
|
||||
id: event.value.id,
|
||||
type: "function",
|
||||
function: { name: event.value.name, arguments: event.value.arguments },
|
||||
}],
|
||||
},
|
||||
})));
|
||||
} else if (event.type === "error") {
|
||||
// An SSE error frame, not a content delta: a protocol failure must not
|
||||
// be rendered to the user as the assistant's reply, and downstream
|
||||
@@ -643,7 +815,10 @@ export class CursorExecutor extends BaseExecutor {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
controller.close();
|
||||
} else if (event.type === "done") {
|
||||
controller.enqueue(encoder.encode(chatChunkSse({ id: responseId, created, model, delta: {}, finishReason: "stop" })));
|
||||
controller.enqueue(encoder.encode(chatChunkSse({
|
||||
id: responseId, created, model, delta: {},
|
||||
finishReason: event.finishReason || "stop",
|
||||
})));
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
controller.close();
|
||||
}
|
||||
@@ -664,9 +839,9 @@ export class CursorExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
|
||||
if (isAgentTextRequest(body)) {
|
||||
if (isAgentCapableRequest(body)) {
|
||||
try {
|
||||
return await this.executeAgent({ model, body, stream, credentials, signal });
|
||||
return await this.executeAgent({ model, body, stream, credentials, signal, log });
|
||||
} catch (error) {
|
||||
return {
|
||||
response: new Response(JSON.stringify({
|
||||
|
||||
@@ -1,12 +1,13 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS, PROVIDER_OAUTH } from "../config/providers.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta, mergeAnthropicBeta } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
import { OAUTH_ENDPOINTS, buildKimiHeaders } from "../config/appConstants.js";
|
||||
import { buildClineHeaders } from "../shared/clineAuth.js";
|
||||
import { getCachedClaudeHeaders } from "../utils/claudeHeaderCache.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
|
||||
import { extractClaudeSessionIdFromUserId } from "../utils/claudeCloaking.js";
|
||||
|
||||
// Auth header descriptors — derived from registry transport.auth, fallback to hardcoded defaults.
|
||||
const BEARER = { combined: true, header: "Authorization", scheme: "bearer" };
|
||||
@@ -42,21 +43,6 @@ const HEADER_HOOKS = {
|
||||
kimiHeaders: (h, c) => Object.assign(h, buildKimiHeaders(c?.providerSpecificData?.deviceId)),
|
||||
clineHeaders: (h, c) => Object.assign(h, buildClineHeaders(c.apiKey || c.accessToken)),
|
||||
kilocodeOrg: (h, c) => { if (c.providerSpecificData?.orgId) h["X-Kilocode-OrganizationID"] = c.providerSpecificData.orgId; },
|
||||
claudeOverlay: (h) => {
|
||||
const cached = getCachedClaudeHeaders();
|
||||
if (!cached) return;
|
||||
for (const lcKey of Object.keys(cached)) {
|
||||
const titleKey = lcKey.replace(/(^|-)([a-z])/g, (_, sep, ch) => sep + ch.toUpperCase());
|
||||
if (lcKey === "anthropic-beta") {
|
||||
const staticBetaStr = h[titleKey] || h[lcKey] || "";
|
||||
const flags = new Set(staticBetaStr.split(",").map(f => f.trim()).filter(Boolean));
|
||||
for (const f of cached[lcKey].split(",").map(f => f.trim()).filter(Boolean)) flags.add(f);
|
||||
cached[lcKey] = Array.from(flags).join(",");
|
||||
}
|
||||
if (titleKey !== lcKey && h[titleKey] !== undefined) delete h[titleKey];
|
||||
}
|
||||
Object.assign(h, cached);
|
||||
},
|
||||
};
|
||||
|
||||
// Config-driven OAuth refresh grants — derived from registry oauth.refresh.
|
||||
@@ -125,7 +111,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
if (this.provider?.startsWith?.("openai-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || OPENAI_COMPAT_BASE;
|
||||
const normalized = baseUrl.replace(/\/$/, "");
|
||||
const path = this.provider.includes("responses") ? "/responses" : "/chat/completions";
|
||||
const path = resolveOpenAICompatibleApiType(this.provider, credentials) === "responses" ? "/responses" : "/chat/completions";
|
||||
return `${normalized}${path}`;
|
||||
}
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
@@ -161,14 +147,41 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
return BEARER;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
buildHeaders(credentials, stream = true, url, model, body = null) {
|
||||
const rt = credentials?.runtimeTransport;
|
||||
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
|
||||
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
|
||||
// Hooks run BEFORE auth so dynamic overlays (claude cached headers) can't clobber the token.
|
||||
// Hooks run BEFORE auth so dynamic overlays can't clobber the token.
|
||||
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
|
||||
applyAuth(headers, desc, credentials);
|
||||
|
||||
// anthropic-compatible-* nodes serving a real Claude model sit in front of
|
||||
// Anthropic itself (a rotating multi-account proxy, a corporate gateway),
|
||||
// so the request needs the same beta flags the `claude` provider sends:
|
||||
// without `context-management-2025-06-27` upstream rejects the
|
||||
// `context_management` block Claude Code puts in every request with
|
||||
// "context_management: Extra inputs are not permitted" (HTTP 400), and the
|
||||
// combo silently falls through to the next model. The model id gates this:
|
||||
// a node fronting Kimi or GLM answers on its own ids and never matches, so
|
||||
// gateways that would choke on unknown beta flags are left untouched.
|
||||
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
|
||||
const clientBeta = credentials?.rawHeaders?.["anthropic-beta"];
|
||||
if (model && (this.provider === "claude"
|
||||
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
|
||||
headers["Anthropic-Beta"] = mergeAnthropicBeta(selectAnthropicBeta(model, body), clientBeta);
|
||||
} else if (this.provider === "anthropic" && clientBeta) {
|
||||
headers["Anthropic-Beta"] = mergeAnthropicBeta(headers["Anthropic-Beta"], clientBeta);
|
||||
}
|
||||
|
||||
// Claude OAuth: align x-claude-code-session-id with metadata.user_id.session_id if missing
|
||||
if (this.provider === "claude" && !headers["x-claude-code-session-id"]) {
|
||||
const token = credentials?.accessToken || credentials?.apiKey || "";
|
||||
if (token.includes("sk-ant-oat")) {
|
||||
const sid = extractClaudeSessionIdFromUserId(body?.metadata?.user_id);
|
||||
if (sid) headers["x-claude-code-session-id"] = sid;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || "";
|
||||
@@ -222,7 +235,6 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
const refreshers = {
|
||||
claude: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
codex: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
qwen: () => this.refreshWithForm(OAUTH_ENDPOINTS.qwen.token, { grant_type: "refresh_token", refresh_token: credentials.refreshToken, client_id: PROVIDERS.qwen.clientId }, proxyOptions),
|
||||
iflow: () => this.refreshIflow(credentials.refreshToken, proxyOptions),
|
||||
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
|
||||
|
||||
@@ -9,15 +9,16 @@ import { KimchiExecutor } from "./kimchi.js";
|
||||
import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
import { QwenExecutor } from "./qwen.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { OpenCodeZenExecutor } from "./opencode-zen.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
import { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
import { CommandCodeExecutor } from "./commandcode.js";
|
||||
import { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
|
||||
import { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
|
||||
import { MimoFreeExecutor } from "./mimo-free.js";
|
||||
import { CodeBuddyExecutor } from "./codebuddy-cn.js";
|
||||
import { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
|
||||
@@ -34,6 +35,7 @@ const executors = {
|
||||
github: new GithubExecutor(),
|
||||
iflow: new IFlowExecutor(),
|
||||
qoder: new QoderExecutor(),
|
||||
"qoder-cn": new QoderExecutor("qoder-cn"),
|
||||
kiro: new KiroExecutor(),
|
||||
kimchi: new KimchiExecutor(),
|
||||
codex: new CodexExecutor(),
|
||||
@@ -41,9 +43,9 @@ const executors = {
|
||||
cu: new CursorExecutor(), // Alias for cursor
|
||||
vertex: new VertexExecutor("vertex"),
|
||||
"vertex-partner": new VertexExecutor("vertex-partner"),
|
||||
qwen: new QwenExecutor(),
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"opencode-zen": new OpenCodeZenExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
@@ -52,6 +54,7 @@ const executors = {
|
||||
"ollama-local": new OllamaLocalExecutor(),
|
||||
commandcode: new CommandCodeExecutor(),
|
||||
"xiaomi-tokenplan": new XiaomiTokenplanExecutor(),
|
||||
"xiaomi-mimo": new XiaomiMimoExecutor(),
|
||||
"mimo-free": new MimoFreeExecutor(),
|
||||
mmf: new MimoFreeExecutor(), // Alias for mimo-free
|
||||
"codebuddy-cn": new CodeBuddyExecutor(),
|
||||
@@ -87,15 +90,16 @@ export { CodexExecutor } from "./codex.js";
|
||||
export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
export { DefaultExecutor } from "./default.js";
|
||||
export { QwenExecutor } from "./qwen.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { OpenCodeZenExecutor } from "./opencode-zen.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
export { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
export { CommandCodeExecutor } from "./commandcode.js";
|
||||
export { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
|
||||
export { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
|
||||
export { MimoFreeExecutor } from "./mimo-free.js";
|
||||
export { CodeBuddyExecutor } from "./codebuddy-cn.js";
|
||||
export { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
|
||||
|
||||
@@ -127,12 +127,18 @@ async function readResponsePrefix(response, signal, maxBytes, timeoutMs) {
|
||||
return decoder.decode(concatChunks(chunks, totalBytes));
|
||||
}
|
||||
|
||||
// The instruction goes into the current user turn, never into a top-level
|
||||
// `systemPrompt`: kiro.dev answers any body carrying that field with
|
||||
// 400 REQUEST_BODY_INVALID, so writing it here turned every repair retry into
|
||||
// a hard failure.
|
||||
function appendRepairInstruction(body, kind) {
|
||||
const repaired = structuredClone(body || {});
|
||||
const instruction = REPAIR_INSTRUCTIONS[kind] || "Retry the previous incomplete Kiro response.";
|
||||
repaired.systemPrompt = repaired.systemPrompt
|
||||
? `${repaired.systemPrompt}\n\n${instruction}`
|
||||
: instruction;
|
||||
const msg = repaired?.conversationState?.currentMessage?.userInputMessage;
|
||||
if (msg) {
|
||||
const content = typeof msg.content === "string" ? msg.content : "";
|
||||
msg.content = content ? `${content}\n\n${instruction}` : instruction;
|
||||
}
|
||||
return repaired;
|
||||
}
|
||||
|
||||
@@ -144,6 +150,12 @@ function normalizeStopReason(value) {
|
||||
return reason || null;
|
||||
}
|
||||
|
||||
// Of the reasons stopDisposition() folds into "terminal_incomplete", only these
|
||||
// mean "usable as far as it got, then the budget ran out" -- the case
|
||||
// finish_reason "length" exists for. cancelled / pause_turn are abandoned turns
|
||||
// whose partial content must stay private, so they are deliberately absent.
|
||||
const KIRO_TRUNCATION_STOP_REASONS = new Set(["model_context_window_exceeded", "max_tokens"]);
|
||||
|
||||
function stopDisposition(stopReason, hasToolCalls) {
|
||||
if (["malformed_model_output", "invalid_model_output"].includes(stopReason)) return "retryable_protocol_failure";
|
||||
if (["cancelled", "pause_turn", "model_context_window_exceeded"].includes(stopReason)) return "terminal_incomplete";
|
||||
@@ -253,6 +265,19 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
}
|
||||
|
||||
// CLIRO parity for the Amazon surfaces: the Kiro runtime accepts the
|
||||
// SSO bearer header + agent-mode marker. Without these the deprecated
|
||||
// path gateway answers REQUEST_BODY_INVALID for modern payloads.
|
||||
if (credentials?.accessToken) {
|
||||
headers["x-amz-sso-bearer"] = credentials.accessToken;
|
||||
}
|
||||
headers["x-amzn-kiro-agent-mode"] = "spec";
|
||||
headers["x-amzn-codewhisperer-machine-id"] = "kiro-desktop";
|
||||
const profileArn = credentials?.providerSpecificData?.profileArn;
|
||||
if (profileArn) {
|
||||
headers["x-amzn-codewhisperer-profile-arn"] = profileArn;
|
||||
}
|
||||
|
||||
return headers;
|
||||
}
|
||||
|
||||
@@ -279,9 +304,13 @@ export class KiroExecutor extends BaseExecutor {
|
||||
// 403 "bearer token invalid", so they must hit the CodeWhisperer
|
||||
// *.amazonaws.com surface, and in the region the token was minted in
|
||||
// (the baseUrls are hardcoded us-east-1).
|
||||
const isCodeWhispererSurface =
|
||||
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
|
||||
if (!isCodeWhispererSurface) return baseUrls;
|
||||
// Kiro deprecated the legacy path-style GenerateAssistantResponse on
|
||||
// runtime.*.kiro.dev (IDE 1.0.228+ moved to POST / + x-amz-target). The
|
||||
// path gateway now answers valid modern payloads with 400
|
||||
// REQUEST_BODY_INVALID, and 400 is terminal in BaseExecutor, so kiro.dev
|
||||
// must never be the first surface for any auth method. Amazon surfaces
|
||||
// reject foreign tokens with 401/403, which DO fall through, so trying
|
||||
// q/codewhisperer first is safe for every auth method (CLIRO parity).
|
||||
|
||||
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
|
||||
const regionalize = (u) =>
|
||||
@@ -291,20 +320,17 @@ export class KiroExecutor extends BaseExecutor {
|
||||
|
||||
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
|
||||
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
|
||||
if (authMethod === "api_key") {
|
||||
const q = amazon.filter((u) => u.includes("://q."));
|
||||
const remaining = amazon.filter((u) => !u.includes("://q."));
|
||||
return q.length > 0
|
||||
? [...q, ...remaining, ...others]
|
||||
: [...amazon, ...others];
|
||||
}
|
||||
|
||||
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
|
||||
const q = amazon.filter((u) => u.includes("://q."));
|
||||
const remaining = amazon.filter((u) => !u.includes("://q."));
|
||||
return q.length > 0
|
||||
? [...q, ...remaining, ...others]
|
||||
: [...amazon, ...others];
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
const baseUrls = this.getOrderedBaseUrls(credentials);
|
||||
return baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
|
||||
const url = baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
|
||||
return url;
|
||||
}
|
||||
|
||||
// Retry only endpoint/auth-surface failures. Payload-invalid HTTP 400 must be
|
||||
@@ -711,14 +737,25 @@ export class KiroExecutor extends BaseExecutor {
|
||||
};
|
||||
const emitTools = (controller) => {
|
||||
for (const tool of state.tools.values()) {
|
||||
const input = parsedToolInput(tool);
|
||||
if (tool.name === "tool_call") {
|
||||
if (typeof input.name !== "string" || !input.name.trim()) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool name");
|
||||
}
|
||||
if (!Object.prototype.hasOwnProperty.call(input, "arguments")) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool arguments");
|
||||
// Validate per tool, not per turn: one unusable fragment used to throw out
|
||||
// of emitTools and take every other complete tool call in the same turn
|
||||
// with it, which the client saw as a turn that answered nothing.
|
||||
let input;
|
||||
try {
|
||||
input = parsedToolInput(tool);
|
||||
if (tool.name === "tool_call") {
|
||||
if (typeof input.name !== "string" || !input.name.trim()) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool name");
|
||||
}
|
||||
if (!Object.prototype.hasOwnProperty.call(input, "arguments")) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool arguments");
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
state.droppedTools = (state.droppedTools || 0) + 1;
|
||||
state.toolValidationError ||= error.message;
|
||||
console.error(`[Kiro] dropping unusable tool call ${tool.id} (${tool.name}): ${error.message}`);
|
||||
continue;
|
||||
}
|
||||
const index = state.toolCounter++;
|
||||
emitDelta(controller, {
|
||||
@@ -729,14 +766,26 @@ export class KiroExecutor extends BaseExecutor {
|
||||
function: { name: tool.name, arguments: "" }
|
||||
}]
|
||||
});
|
||||
const serializedInput = JSON.stringify(input);
|
||||
emitDelta(controller, {
|
||||
tool_calls: [{ index, function: { arguments: JSON.stringify(input) } }]
|
||||
tool_calls: [{ index, function: { arguments: serializedInput } }]
|
||||
});
|
||||
// Tool arguments are billed output like any other completion bytes. They
|
||||
// were never added to totalContentLength, so the /4 estimator in finish()
|
||||
// reported OUT 0 -- or the Math.max floor of 1 -- for every turn whose
|
||||
// entire answer was a tool call.
|
||||
state.totalContentLength += tool.name.length + serializedInput.length;
|
||||
state.hasToolCalls = true;
|
||||
}
|
||||
state.tools.clear();
|
||||
state.bufferedToolBytes = 0;
|
||||
if (state.stopReason === "tool_use" && !state.hasToolCalls) {
|
||||
// A declared tool turn that emitted no usable call is only fatal when the
|
||||
// turn produced nothing else. Throwing unconditionally here escaped
|
||||
// emitTools() with provenance "invalid_tool_call", which the integrity gate
|
||||
// re-derived into a repair retry -- discarding text the client had already
|
||||
// been promised.
|
||||
if (state.stopReason === "tool_use" && !state.hasToolCalls &&
|
||||
!state.hasText && !state.hasReasoning && !state.hasCode) {
|
||||
throw new Error("Kiro tool_use stop reason did not include a complete tool call");
|
||||
}
|
||||
};
|
||||
@@ -796,7 +845,6 @@ export class KiroExecutor extends BaseExecutor {
|
||||
emitDelta(controller, { content: event.payload.content });
|
||||
} else if (eventType === "toolUseEvent") {
|
||||
state.sawToolUse = true;
|
||||
if (state.toolValidationError) return true;
|
||||
const values = Array.isArray(event.payload) ? event.payload : [event.payload];
|
||||
if (!values[0]) throw new Error("Kiro toolUseEvent is empty");
|
||||
for (const value of values) {
|
||||
@@ -924,9 +972,10 @@ export class KiroExecutor extends BaseExecutor {
|
||||
} catch (error) {
|
||||
const bufferExceeded = error.code === "KIRO_BUFFER_EXCEEDED";
|
||||
if (!bufferExceeded) {
|
||||
// Keep whatever is already buffered: the rejected fragment belongs to
|
||||
// one tool, and clearing the map dropped the complete calls too.
|
||||
state.toolValidationError ||= error.message;
|
||||
state.tools.clear();
|
||||
state.bufferedToolBytes = 0;
|
||||
console.error(`[Kiro] tool fragment rejected, keeping ${state.tools.size} buffered tool(s): ${error.message}`);
|
||||
continue;
|
||||
}
|
||||
fail(
|
||||
@@ -958,7 +1007,16 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
state.transportState = "clean_eof";
|
||||
const declaredDisposition = stopDisposition(state.stopReason, state.sawToolUse);
|
||||
if (["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(declaredDisposition)) {
|
||||
// model_context_window_exceeded / max_tokens map to terminal_incomplete. When
|
||||
// they arrive after the model already streamed content, fail() threw away a
|
||||
// complete-enough answer; a truncated turn is what finish_reason "length" is
|
||||
// for. chunkIndex > 0 means at least one delta already reached the client.
|
||||
const declaredTruncatedAfterOutput = declaredDisposition === "terminal_incomplete" &&
|
||||
KIRO_TRUNCATION_STOP_REASONS.has(state.stopReason) && state.chunkIndex > 0;
|
||||
if (declaredTruncatedAfterOutput) {
|
||||
console.error(`[Kiro] truncated after ${state.chunkIndex} chunk(s) (stop_reason=${state.stopReason}); keeping output`);
|
||||
}
|
||||
if (!declaredTruncatedAfterOutput && ["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(declaredDisposition)) {
|
||||
const code = declaredDisposition === "retryable_protocol_failure"
|
||||
? "kiro_retryable_protocol_failure"
|
||||
: declaredDisposition === "terminal_refusal"
|
||||
@@ -975,16 +1033,6 @@ export class KiroExecutor extends BaseExecutor {
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (state.toolValidationError) {
|
||||
fail(
|
||||
controller,
|
||||
"invalid_tool_call",
|
||||
"invalid_kiro_tool_call",
|
||||
state.toolValidationError,
|
||||
{ transport_state: state.transportState, stop_disposition: "retryable_protocol_failure" }
|
||||
);
|
||||
return;
|
||||
}
|
||||
try {
|
||||
emitTools(controller);
|
||||
} catch (error) {
|
||||
@@ -997,6 +1045,22 @@ export class KiroExecutor extends BaseExecutor {
|
||||
);
|
||||
return;
|
||||
}
|
||||
// Fail only when the turn has nothing usable left. emitTools() validates
|
||||
// per tool and drops just the unusable ones, so this has to run AFTER it:
|
||||
// before, the rejected tool was still buffered and tools.size was never 0.
|
||||
// A turn that also produced text keeps that text -- the dropped call is
|
||||
// logged, not fatal.
|
||||
if (state.toolValidationError && !state.hasToolCalls &&
|
||||
!state.hasText && !state.hasReasoning && !state.hasCode) {
|
||||
fail(
|
||||
controller,
|
||||
"invalid_tool_call",
|
||||
"invalid_kiro_tool_call",
|
||||
state.toolValidationError,
|
||||
{ transport_state: state.transportState, stop_disposition: "retryable_protocol_failure" }
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const hasOutput = state.hasText || state.hasReasoning || state.hasCode || state.hasToolCalls;
|
||||
if (!hasOutput && !state.explicitStop) {
|
||||
@@ -1011,7 +1075,13 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
const disposition = stopDisposition(state.stopReason, state.hasToolCalls);
|
||||
if (["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(disposition)) {
|
||||
// Same reasoning as declaredTruncatedAfterOutput above.
|
||||
const truncatedAfterOutput = disposition === "terminal_incomplete" &&
|
||||
KIRO_TRUNCATION_STOP_REASONS.has(state.stopReason) && state.chunkIndex > 0;
|
||||
if (truncatedAfterOutput) {
|
||||
console.error(`[Kiro] truncated after ${state.chunkIndex} chunk(s) (stop_reason=${state.stopReason}); closing as length`);
|
||||
}
|
||||
if (!truncatedAfterOutput && ["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(disposition)) {
|
||||
const code = disposition === "retryable_protocol_failure"
|
||||
? "kiro_retryable_protocol_failure"
|
||||
: disposition === "terminal_refusal"
|
||||
@@ -1041,18 +1111,24 @@ export class KiroExecutor extends BaseExecutor {
|
||||
total_tokens: prompt + completion
|
||||
};
|
||||
}
|
||||
const finishReason = state.hasToolCalls
|
||||
? "tool_calls"
|
||||
: disposition === "length"
|
||||
? "length"
|
||||
: "stop";
|
||||
const finishReason = truncatedAfterOutput
|
||||
? "length"
|
||||
: state.hasToolCalls
|
||||
? "tool_calls"
|
||||
: disposition === "length"
|
||||
? "length"
|
||||
: "stop";
|
||||
controller.enqueue(sseChunk({}, finishReason, state.usage));
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
state.finished = true;
|
||||
options.onTerminalState?.(diagnostics({
|
||||
terminal_provenance: state.terminalProvenance || "clean_eventstream_eof",
|
||||
transport_state: state.transportState,
|
||||
stop_disposition: disposition
|
||||
// Report what this exit actually did, not the raw disposition. The
|
||||
// integrity gate re-derives its verdict from stop_disposition, so
|
||||
// reporting "terminal_incomplete" for a turn we deliberately kept made
|
||||
// it discard the very bytes we just released to the client.
|
||||
stop_disposition: truncatedAfterOutput ? "length" : disposition
|
||||
}));
|
||||
};
|
||||
|
||||
|
||||
@@ -1,49 +1,186 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
import crypto from "node:crypto";
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { getModelTargetFormat } from "../config/providerModels.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
// Models that use /zen/go/v1/messages (Anthropic/Claude format + x-api-key auth)
|
||||
const MESSAGES_FORMAT_MODELS = new Set([
|
||||
"minimax-m3",
|
||||
"minimax-m2.7",
|
||||
"minimax-m2.5",
|
||||
"qwen3.7-max",
|
||||
"qwen3.7-plus",
|
||||
"qwen3.6-plus",
|
||||
]);
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeGoSession";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
|
||||
const BASE = "https://opencode.ai/zen/go/v1";
|
||||
const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses";
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
|
||||
export class OpenCodeGoExecutor extends BaseExecutor {
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function translatedSession(sessionId, clientTool) {
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
|
||||
.digest("hex")
|
||||
.slice(0, 32);
|
||||
return `ses_${digest}`;
|
||||
}
|
||||
|
||||
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …),
|
||||
// including the family-regex fallback for passthrough ids — never hardcode model ids here.
|
||||
function isResponsesModel(model) {
|
||||
return getModelTargetFormat("opencode-go", model) === FORMATS.OPENAI_RESPONSES;
|
||||
}
|
||||
|
||||
// Flatten Chat Completions tool declarations into the Responses flat shape and
|
||||
// drop hosted/nameless tools the /responses endpoint rejects.
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
// Mirror the request translator: {type:"object"} without properties is rejected
|
||||
// by strict Responses backends, so fill in the empty properties map.
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Last line of defense for native Responses clients (sourceFormat === targetFormat
|
||||
// skips translation): coerce items in place so malformed tool payloads 400 here
|
||||
// with a clear shape instead of upstream as InputValidationError.
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: Muse Spark contributor models route to
|
||||
// an upstream Console backend where encrypted_content cannot be validated across
|
||||
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export class OpenCodeGoExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("opencode-go", PROVIDERS["opencode-go"]);
|
||||
super("opencode-go");
|
||||
}
|
||||
|
||||
// buildUrl runs before buildHeaders in BaseExecutor.execute, cache model here
|
||||
buildUrl(model) {
|
||||
this._lastModel = model;
|
||||
return MESSAGES_FORMAT_MODELS.has(model)
|
||||
? `${BASE}/messages`
|
||||
: `${BASE}/chat/completions`;
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
|
||||
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
|
||||
return super.buildUrl(model, stream, urlIndex, credentials);
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const key = credentials?.apiKey || credentials?.accessToken;
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const native = nativeSession(sourceCredentials.rawHeaders);
|
||||
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
|
||||
headers: sourceCredentials.rawHeaders,
|
||||
body,
|
||||
connectionId: sourceCredentials.connectionId,
|
||||
scope: "opencode-go",
|
||||
});
|
||||
|
||||
if (MESSAGES_FORMAT_MODELS.has(this._lastModel)) {
|
||||
headers["x-api-key"] = key;
|
||||
headers["anthropic-version"] = ANTHROPIC_API_VERSION;
|
||||
} else {
|
||||
headers["Authorization"] = `Bearer ${key}`;
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
|
||||
};
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
const credentials = this.prepareRequestCredentials(args);
|
||||
return super.execute({ ...args, credentials });
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
const headers = super.buildHeaders(credentials || {}, stream, url, model);
|
||||
const prepared = credentials?.[SESSION_FIELD];
|
||||
if (prepared) {
|
||||
headers[SESSION_HEADER] = prepared;
|
||||
return headers;
|
||||
}
|
||||
|
||||
if (stream) headers["Accept"] = "text/event-stream";
|
||||
const fallback = this.prepareRequestCredentials({ credentials });
|
||||
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
|
||||
return headers;
|
||||
}
|
||||
|
||||
transformRequest(model, body) {
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
const out = super.transformRequest(model, body);
|
||||
if (!isResponsesModel(model || body?.model)) return out;
|
||||
const normalized = normalizeResponsesInput(out.input);
|
||||
if (normalized) out.input = normalized;
|
||||
if (!Array.isArray(out.input) || out.input.length === 0) {
|
||||
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses names the output cap max_output_tokens, not max_tokens.
|
||||
if (out.max_output_tokens === undefined) {
|
||||
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
|
||||
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
|
||||
}
|
||||
delete out.max_tokens;
|
||||
delete out.max_completion_tokens;
|
||||
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
|
||||
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
|
||||
}
|
||||
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
|
||||
if (!out.reasoning.summary) out.reasoning.summary = "auto";
|
||||
}
|
||||
delete out.reasoning_effort;
|
||||
out.stream = true;
|
||||
out.store = false;
|
||||
normalizeResponsesTools(out);
|
||||
sanitizeResponsesItems(out);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
|
||||
315
open-sse/executors/opencode-zen.js
Normal file
315
open-sse/executors/opencode-zen.js
Normal file
@@ -0,0 +1,315 @@
|
||||
import crypto from "node:crypto";
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeZenSession";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
|
||||
const RESPONSES_BASE_URL = "https://opencode.ai/zen/v1/responses";
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
const OPENCODE_UA = "opencode/1.18.31";
|
||||
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
|
||||
// Free-tier fingerprint (mirrors opencode executor, PR #4132): upstream 403s
|
||||
// requests without the file-search quartet and without stream:true.
|
||||
const OPENCODE_FINGERPRINT_TOOLS = ["bash", "glob", "grep", "read"];
|
||||
|
||||
function hasValidOpencodeVersion(ua) {
|
||||
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
|
||||
if (!m) return false;
|
||||
const major = parseInt(m[1], 10);
|
||||
const minor = parseInt(m[2], 10);
|
||||
return major > 1 || (major === 1 && minor >= 17);
|
||||
}
|
||||
|
||||
function unstableRandom() {
|
||||
const bytes = crypto.randomBytes(14);
|
||||
let randomPart = "";
|
||||
for (let i = 0; i < 14; i++) {
|
||||
randomPart += BASE62_CHARS[bytes[i] % 62];
|
||||
}
|
||||
return randomPart;
|
||||
}
|
||||
|
||||
export function generateSessionId(timestamp = Date.now()) {
|
||||
const current = BigInt(timestamp) * 0x1000n + 1n;
|
||||
const value = ~current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `ses_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function generateRequestId(timestamp = Date.now()) {
|
||||
const current = BigInt(timestamp) * 0x1000n + 1n;
|
||||
const value = current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `msg_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function translateSessionId(sessionId, clientTool = "") {
|
||||
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
|
||||
return sessionId.trim();
|
||||
}
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
return `ses_${timeHex}${randomPart}`;
|
||||
}
|
||||
|
||||
function toolNameOf(tool) {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return "";
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const raw = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
return raw.trim();
|
||||
}
|
||||
|
||||
function ensureChatFingerprintTools(body) {
|
||||
if (!body || typeof body !== "object") return;
|
||||
const present = new Set();
|
||||
if (Array.isArray(body.tools)) {
|
||||
for (const tool of body.tools) {
|
||||
const name = toolNameOf(tool);
|
||||
if (name) present.add(name);
|
||||
}
|
||||
} else {
|
||||
body.tools = [];
|
||||
}
|
||||
for (const name of OPENCODE_FINGERPRINT_TOOLS) {
|
||||
if (present.has(name)) continue;
|
||||
body.tools.push({
|
||||
type: "function",
|
||||
function: {
|
||||
name,
|
||||
description: `OpenCode built-in ${name} tool`,
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
});
|
||||
present.add(name);
|
||||
}
|
||||
}
|
||||
|
||||
function ensureResponsesFingerprintTools(body) {
|
||||
if (!body || typeof body !== "object") return;
|
||||
const present = new Set();
|
||||
if (Array.isArray(body.tools)) {
|
||||
for (const tool of body.tools) {
|
||||
const name = toolNameOf(tool);
|
||||
if (name) present.add(name);
|
||||
}
|
||||
} else {
|
||||
body.tools = [];
|
||||
}
|
||||
for (const name of OPENCODE_FINGERPRINT_TOOLS) {
|
||||
if (present.has(name)) continue;
|
||||
body.tools.push({
|
||||
type: "function",
|
||||
name,
|
||||
description: `OpenCode built-in ${name} tool`,
|
||||
parameters: { type: "object", properties: {} },
|
||||
});
|
||||
present.add(name);
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
const normalized = normalizeSession(value);
|
||||
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function translatedSession(sessionId, clientTool) {
|
||||
return translateSessionId(sessionId, clientTool);
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so checks hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
function isResponsesModel(model) {
|
||||
return isMuseSparkModel(baseModelId(model));
|
||||
}
|
||||
|
||||
// Flatten Chat Completions tool declarations into the Responses flat shape and
|
||||
// drop hosted/nameless tools the /responses endpoint rejects.
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
// Mirror the request translator: {type:"object"} without properties is rejected
|
||||
// by strict Responses backends, so fill in the empty properties map.
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Last line of defense for native Responses clients (sourceFormat === targetFormat
|
||||
// skips translation): coerce items in place so malformed tool payloads 400 here
|
||||
// with a clear shape instead of upstream as InputValidationError.
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: Muse Spark contributor models route to
|
||||
// an upstream Console backend where encrypted_content cannot be validated across
|
||||
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export class OpenCodeZenExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("opencode-zen");
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
|
||||
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
|
||||
return super.buildUrl(model, stream, urlIndex, credentials);
|
||||
}
|
||||
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const native = nativeSession(sourceCredentials.rawHeaders);
|
||||
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
|
||||
headers: sourceCredentials.rawHeaders,
|
||||
body,
|
||||
connectionId: sourceCredentials.connectionId,
|
||||
scope: "opencode-zen",
|
||||
});
|
||||
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
|
||||
};
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
const credentials = this.prepareRequestCredentials(args);
|
||||
return super.execute({ ...args, credentials });
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
const headers = super.buildHeaders(credentials || {}, stream, url, model);
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const lower = {};
|
||||
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
|
||||
const downstreamUa = lower["user-agent"] || "";
|
||||
// Free-tier gate: spoof the official client UA.
|
||||
headers["User-Agent"] = hasValidOpencodeVersion(downstreamUa) ? downstreamUa : OPENCODE_UA;
|
||||
headers["x-opencode-client"] = lower["x-opencode-client"] || "desktop";
|
||||
const prepared = credentials?.[SESSION_FIELD];
|
||||
if (prepared) {
|
||||
headers[SESSION_HEADER] = prepared;
|
||||
return headers;
|
||||
}
|
||||
|
||||
const fallback = this.prepareRequestCredentials({ credentials });
|
||||
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
|
||||
return headers;
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
const out = super.transformRequest(model, body);
|
||||
// Free-tier gate: upstream 403s stream:false even when everything else is valid.
|
||||
if (out && typeof out === "object") out.stream = true;
|
||||
if (!isResponsesModel(model || body?.model)) {
|
||||
ensureChatFingerprintTools(out);
|
||||
return out;
|
||||
}
|
||||
const normalized = normalizeResponsesInput(out.input);
|
||||
if (normalized) out.input = normalized;
|
||||
if (!Array.isArray(out.input) || out.input.length === 0) {
|
||||
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses names the output cap max_output_tokens, not max_tokens.
|
||||
if (out.max_output_tokens === undefined) {
|
||||
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
|
||||
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
|
||||
}
|
||||
delete out.max_tokens;
|
||||
delete out.max_completion_tokens;
|
||||
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
|
||||
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
|
||||
}
|
||||
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
|
||||
if (!out.reasoning.summary) out.reasoning.summary = "auto";
|
||||
}
|
||||
delete out.reasoning_effort;
|
||||
out.stream = true;
|
||||
out.store = false;
|
||||
ensureResponsesFingerprintTools(out);
|
||||
normalizeResponsesTools(out);
|
||||
sanitizeResponsesItems(out);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
@@ -1,32 +1,487 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import { applyFingerprintTools } from "../utils/opencodeFingerprint.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
// Models that use /zen/v1/messages (claude format)
|
||||
const MESSAGES_MODELS = new Set();
|
||||
const OPENCODE_UA = "opencode/1.18.31";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeSession";
|
||||
const REQ_FIELD = "_opencodeRequest";
|
||||
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
|
||||
|
||||
function hasValidOpencodeVersion(ua) {
|
||||
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
|
||||
if (!m) return false;
|
||||
const major = parseInt(m[1], 10);
|
||||
const minor = parseInt(m[2], 10);
|
||||
return major > 1 || (major === 1 && minor >= 17);
|
||||
}
|
||||
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
|
||||
const RESPONSES_MODELS = new Set([
|
||||
"muse-spark-1.2-contributor-free",
|
||||
"muse-spark-1.3-contributor-free",
|
||||
]);
|
||||
const MESSAGES_MODELS = new Set(["union-alpha"]);
|
||||
|
||||
let lastTimestamp = 0;
|
||||
let counter = 0;
|
||||
|
||||
function unstableRandom() {
|
||||
const bytes = crypto.randomBytes(14);
|
||||
let randomPart = "";
|
||||
for (let i = 0; i < 14; i++) {
|
||||
randomPart += BASE62_CHARS[bytes[i] % 62];
|
||||
}
|
||||
return randomPart;
|
||||
}
|
||||
|
||||
export function generateSessionId(timestamp = Date.now()) {
|
||||
if (timestamp !== lastTimestamp) {
|
||||
lastTimestamp = timestamp;
|
||||
counter = 0;
|
||||
}
|
||||
counter++;
|
||||
|
||||
const current = BigInt(timestamp) * 0x1000n + BigInt(counter);
|
||||
const value = ~current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `ses_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function generateRequestId(timestamp = Date.now()) {
|
||||
const current = BigInt(timestamp) * 0x1000n + 1n;
|
||||
const value = current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `msg_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function translateSessionId(sessionId, clientTool = "") {
|
||||
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
|
||||
return sessionId.trim();
|
||||
}
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
return `ses_${timeHex}${randomPart}`;
|
||||
}
|
||||
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
const normalized = normalizeSession(value);
|
||||
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Upstream free-tier quota is accounted per session. Minting a fresh
|
||||
// x-opencode-session on every request burns through it and surfaces as
|
||||
// 429 FreeUsageLimitError with growing reset-after delays, while the real
|
||||
// CLI reuses one long-lived canonical session per conversation. Mirror
|
||||
// that: one stable canonical session per downstream identity, evicted
|
||||
// after MEMORY_CONFIG.sessionTtlMs like the other session stores.
|
||||
const stableOpencodeSessions = new Map();
|
||||
const MAX_STABLE_SESSIONS = 1000;
|
||||
const stableSessionCleanup = setInterval(() => {
|
||||
const now = Date.now();
|
||||
for (const [key, entry] of stableOpencodeSessions) {
|
||||
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) {
|
||||
stableOpencodeSessions.delete(key);
|
||||
}
|
||||
}
|
||||
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
|
||||
if (stableSessionCleanup.unref) stableSessionCleanup.unref();
|
||||
|
||||
function identityKey(credentials) {
|
||||
const connectionId = credentials?.connectionId || credentials?.id;
|
||||
if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`;
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || "";
|
||||
if (auth) {
|
||||
const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32);
|
||||
return `opencode:auth:${digest}`;
|
||||
}
|
||||
return "opencode:default";
|
||||
}
|
||||
|
||||
export function stableSessionId(credentials) {
|
||||
const key = identityKey(credentials);
|
||||
const existing = stableOpencodeSessions.get(key);
|
||||
if (existing) {
|
||||
existing.lastUsed = Date.now();
|
||||
stableOpencodeSessions.delete(key);
|
||||
stableOpencodeSessions.set(key, existing);
|
||||
return existing.sessionId;
|
||||
}
|
||||
const sessionId = generateSessionId();
|
||||
if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) {
|
||||
stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value);
|
||||
}
|
||||
stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() });
|
||||
return sessionId;
|
||||
}
|
||||
|
||||
function lastUserText(body) {
|
||||
try {
|
||||
if (!body || typeof body !== "object") return "";
|
||||
const arr = Array.isArray(body.messages)
|
||||
? body.messages
|
||||
: Array.isArray(body.input)
|
||||
? body.input
|
||||
: null;
|
||||
if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : "";
|
||||
for (let i = arr.length - 1; i >= 0; i--) {
|
||||
const msg = arr[i];
|
||||
if (!msg) continue;
|
||||
if (msg.role && msg.role !== "user") continue;
|
||||
const content = msg.content;
|
||||
if (typeof content === "string" && content.trim()) return content.trim().slice(-600);
|
||||
if (Array.isArray(content)) {
|
||||
const text = content
|
||||
.map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || ""))
|
||||
.join(" ")
|
||||
.trim();
|
||||
if (text) return text.slice(-600);
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
// The real CLI sends the current user message id (stable per turn, same on
|
||||
// retries) as x-opencode-request. Derive it deterministically from the
|
||||
// session plus the last user message so retries share the id.
|
||||
export function deriveRequestId(sessionId, body) {
|
||||
const text = lastUserText(body);
|
||||
if (!text) return generateRequestId();
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-req\0${sessionId || ""}\0${text}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
const id = `msg_${timeHex}${randomPart}`;
|
||||
return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId();
|
||||
}
|
||||
|
||||
function normalizeRequestId(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null;
|
||||
}
|
||||
|
||||
function bodyHasSessionHints(body) {
|
||||
try {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (typeof body.session_id === "string" && body.session_id.trim()) return true;
|
||||
if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true;
|
||||
if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true;
|
||||
if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true;
|
||||
if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true;
|
||||
const arr = Array.isArray(body.messages)
|
||||
? body.messages
|
||||
: Array.isArray(body.input)
|
||||
? body.input
|
||||
: null;
|
||||
if (arr) {
|
||||
let assistantText = "";
|
||||
for (const msg of arr) {
|
||||
if (msg?.role === "assistant") {
|
||||
const content = msg.content;
|
||||
if (typeof content === "string") assistantText += content;
|
||||
else if (Array.isArray(content)) {
|
||||
for (const part of content) assistantText += part?.text || part?.output || "";
|
||||
}
|
||||
if (assistantText.length >= 50) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
function isResponsesModel(model) {
|
||||
const base = baseModelId(model);
|
||||
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
|
||||
}
|
||||
|
||||
function isMessagesModel(model) {
|
||||
return MESSAGES_MODELS.has(baseModelId(model));
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) {
|
||||
const headers = credentials?.rawHeaders || {};
|
||||
const native = nativeSession(headers);
|
||||
if (native) return native;
|
||||
|
||||
let incoming = null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
incoming = normalizeSession(value);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
const hinted = incoming || normalizeSession(providerSessionId);
|
||||
if (hinted) return translateSessionId(hinted, clientTool);
|
||||
|
||||
if (credentials?.connectionId || bodyHasSessionHints(body)) {
|
||||
let viaManager = null;
|
||||
try {
|
||||
viaManager = resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
});
|
||||
} catch {
|
||||
viaManager = null;
|
||||
}
|
||||
if (viaManager) return translateSessionId(viaManager, clientTool);
|
||||
}
|
||||
|
||||
return stableSessionId(credentials);
|
||||
}
|
||||
|
||||
function resolveOpencodeRequestId(body, credentials, sessionId) {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
for (const [key, value] of Object.entries(raw)) {
|
||||
if (key.toLowerCase() === "x-opencode-request") {
|
||||
const normalized = normalizeRequestId(value);
|
||||
if (normalized) return normalized;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return deriveRequestId(sessionId, body);
|
||||
}
|
||||
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials
|
||||
// (`Bearer public`) routing to an upstream OpenAI/Console account pool.
|
||||
// OpenAI Responses API strictly enforces that reasoning `encrypted_content`
|
||||
// can only be decrypted by the exact caller/account that issued it; sending it
|
||||
// across different accounts or rotating proxy relays triggers:
|
||||
// [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400).
|
||||
// Furthermore, under stateless mode (store=false), omitting encrypted_content
|
||||
// causes OpenAI to reject the referenced reasoning item as "not found or was deleted".
|
||||
// Dropping prior reasoning items allows multi-turn conversations and tool-calling
|
||||
// loops to succeed cleanly.
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeOpencodeReasoning(model, body) {
|
||||
const current = body.reasoning;
|
||||
const currentReasoning = current && typeof current === "object" && !Array.isArray(current)
|
||||
? current
|
||||
: null;
|
||||
const requestedEffort = typeof body.reasoning_effort === "string"
|
||||
? body.reasoning_effort
|
||||
: currentReasoning?.effort;
|
||||
if (typeof requestedEffort !== "string") return;
|
||||
|
||||
const cleanModel = baseModelId(model || body.model);
|
||||
const supportedLevels = getThinkingLevels("opencode", cleanModel);
|
||||
let effort = requestedEffort.toLowerCase().trim();
|
||||
if ((effort === "max" || effort === "ultra") && supportedLevels?.length && !supportedLevels.includes(effort)) {
|
||||
if (effort === "ultra" && supportedLevels.includes("max")) effort = "max";
|
||||
else if (supportedLevels.includes("xhigh")) effort = "xhigh";
|
||||
}
|
||||
|
||||
body.reasoning = { ...currentReasoning, effort };
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
delete body.reasoning_effort;
|
||||
}
|
||||
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode", PROVIDERS.opencode);
|
||||
}
|
||||
|
||||
transformRequest(model, body) {
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool);
|
||||
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: session,
|
||||
[REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session),
|
||||
};
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
if (body && typeof body === "object" && model && !body.model) body.model = model;
|
||||
// Zen rejects non-streaming requests on free models with 403 FreeTierError;
|
||||
// always stream upstream and let the handler layer aggregate for non-stream clients.
|
||||
if (body && typeof body === "object") body.stream = true;
|
||||
if (isResponsesModel(model || body?.model) && body && typeof body === "object") {
|
||||
// ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng.
|
||||
if ("tool_choice" in body && body.tool_choice !== "auto"
|
||||
&& this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) {
|
||||
body.tool_choice = "auto";
|
||||
}
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
if (!Array.isArray(body.input) || body.input.length === 0) {
|
||||
body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses API names the output cap max_output_tokens and takes thinking
|
||||
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
|
||||
if (body.max_output_tokens === undefined) {
|
||||
if (body.max_completion_tokens !== undefined) body.max_output_tokens = body.max_completion_tokens;
|
||||
else if (body.max_tokens !== undefined) body.max_output_tokens = body.max_tokens;
|
||||
}
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
normalizeOpencodeReasoning(model, body);
|
||||
body.stream = true;
|
||||
body.store = false;
|
||||
normalizeResponsesTools(body);
|
||||
sanitizeResponsesItems(body);
|
||||
// Free-tier fingerprint tools are required even when an agent client
|
||||
// already supplied tools. ZCode/Claude Code requests normally have
|
||||
// non-empty tool arrays; skipping cloaking here triggers 403 FreeTierError.
|
||||
applyFingerprintTools(body, true);
|
||||
} else if (body && typeof body === "object") {
|
||||
applyFingerprintTools(body, false);
|
||||
}
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) });
|
||||
}
|
||||
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
return MESSAGES_MODELS.has(model)
|
||||
? `${base}/zen/v1/messages`
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
if (isResponsesModel(model)) return `${base}/zen/v1/responses`;
|
||||
if (isMessagesModel(model)) return `${base}/zen/v1/messages`;
|
||||
return `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders() {
|
||||
return {
|
||||
buildHeaders(credentials, stream = true, url = "") {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const lower = {};
|
||||
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
|
||||
|
||||
const downstreamUa = lower["user-agent"] || "";
|
||||
const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa);
|
||||
|
||||
const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD];
|
||||
const downstreamReq = normalizeRequestId(lower["x-opencode-request"]);
|
||||
const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId();
|
||||
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer public",
|
||||
"x-opencode-client": "desktop",
|
||||
"Accept": "text/event-stream"
|
||||
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
|
||||
"x-opencode-client": lower["x-opencode-client"] || "desktop",
|
||||
"x-opencode-session": session,
|
||||
"x-opencode-request": requestId,
|
||||
"x-opencode-project": lower["x-opencode-project"] || "global",
|
||||
"Accept": stream ? "text/event-stream" : "*/*",
|
||||
};
|
||||
if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION;
|
||||
return headers;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,21 +29,24 @@ import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { FETCH_CONNECT_TIMEOUT_MS, HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import {
|
||||
QODER_CHAT_URL_ENCODED,
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
QODER_USERINFO_URL,
|
||||
QODER_MODEL_MAP,
|
||||
QODER_IDE_VERSION,
|
||||
QODER_CLIENT_TYPE,
|
||||
QODER_CHAT_SIG_PATH,
|
||||
QODER_CONTEXT_TIER_ENV,
|
||||
qoderInferenceBase,
|
||||
} from "../shared/qoder/constants.js";
|
||||
import { getQoderModelConfig, resolveQoderModels } from "../services/qoderModels.js";
|
||||
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
|
||||
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
|
||||
import { encodeDataUri } from "../translator/concerns/image.js";
|
||||
import { createQoderSseCoalescer } from "../shared/qoder/sse.js";
|
||||
import { rewriteQoderMessageAttachments } from "../shared/qoder/attachments.js";
|
||||
import { resolveQoderContextTier, applyQoderContextTier } from "../shared/qoder/contextTier.js";
|
||||
|
||||
/**
|
||||
* Hoist role:"system" messages out of the messages array (Qoder rejects
|
||||
* system in messages) and flatten any multipart content arrays.
|
||||
* system in messages) and flatten multipart content arrays — EXCEPT image
|
||||
* blocks, which are preserved (see normalizeContent).
|
||||
*/
|
||||
function normalizeMessages(messages) {
|
||||
if (!Array.isArray(messages) || messages.length === 0) {
|
||||
@@ -53,18 +56,88 @@ function normalizeMessages(messages) {
|
||||
const out = [];
|
||||
for (const msg of messages) {
|
||||
if (!msg || typeof msg !== "object") continue;
|
||||
const text = extractText(msg.content);
|
||||
if (msg.role === "system") {
|
||||
const text = extractText(msg.content);
|
||||
if (text) systemParts.push(text);
|
||||
continue;
|
||||
}
|
||||
const cloned = { ...msg };
|
||||
cloned.content = text;
|
||||
cloned.content = normalizeContent(msg.content);
|
||||
out.push(cloned);
|
||||
}
|
||||
return { messages: out, systemText: systemParts.join("\n\n") };
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize one message's content for Qoder.
|
||||
*
|
||||
* Text-only content is flattened to a plain string (Qoder's historical
|
||||
* shape). When images are present the content stays an array and image
|
||||
* blocks are kept as OpenAI-style `image_url` parts. Native qodercli
|
||||
* uploads inlined bytes to `/api/v2/image/upload` first and then sends
|
||||
* the OSS URL — `buildQoderRequestBody` does that rewrite before this
|
||||
* runs. Tiny leftover data URIs are still accepted. The legacy
|
||||
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
|
||||
* qodercli leaves them null too.
|
||||
*
|
||||
* Claude-style `{type:"image", source:{...}}` blocks are converted to
|
||||
* `image_url`. File/document blocks that survived rewrite become short
|
||||
* stubs so 30MB PDFs never land in agent_chat_generation.
|
||||
*/
|
||||
function normalizeContent(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (content == null) return "";
|
||||
if (!Array.isArray(content)) return String(content);
|
||||
|
||||
const blocks = [];
|
||||
const textParts = [];
|
||||
let hasImage = false;
|
||||
|
||||
const pushText = (text) => {
|
||||
if (!text) return;
|
||||
if (hasImage || blocks.length) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
|
||||
else textParts.push(text);
|
||||
};
|
||||
|
||||
const imageUrlOf = (item) => {
|
||||
if (typeof item.image_url === "string" && item.image_url) return item.image_url;
|
||||
if (typeof item.image_url?.url === "string" && item.image_url.url) return item.image_url.url;
|
||||
return null;
|
||||
};
|
||||
|
||||
for (const item of content) {
|
||||
if (!item || typeof item !== "object") continue;
|
||||
const imageUrl = item.type === OPENAI_BLOCK.IMAGE_URL ? imageUrlOf(item) : null;
|
||||
if (imageUrl) {
|
||||
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: imageUrl } });
|
||||
hasImage = true;
|
||||
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
|
||||
// Claude base64/url image → OpenAI image_url equivalent.
|
||||
const src = item.source;
|
||||
const url = src.type === "base64" && src.data
|
||||
? encodeDataUri(src.media_type || "image/png", src.data)
|
||||
: typeof src.url === "string" && src.url ? src.url : null;
|
||||
if (url) {
|
||||
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
|
||||
hasImage = true;
|
||||
}
|
||||
} else if (item.type === OPENAI_BLOCK.FILE) {
|
||||
const name = item.file?.filename || item.file?.name || "file";
|
||||
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
|
||||
} else if (item.type === CLAUDE_BLOCK.DOCUMENT) {
|
||||
const name = item.title || "document";
|
||||
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
|
||||
} else if (typeof item.text === "string" && item.text) {
|
||||
pushText(item.text);
|
||||
}
|
||||
}
|
||||
|
||||
if (!hasImage) return textParts.join("\n");
|
||||
// Prepend any text collected before the first image block.
|
||||
if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") });
|
||||
return blocks;
|
||||
}
|
||||
|
||||
function extractText(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (content == null) return "";
|
||||
@@ -87,9 +160,9 @@ function extractText(content) {
|
||||
function lastUserText(messages) {
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const m = messages[i];
|
||||
if (m?.role === "user" && typeof m.content === "string") {
|
||||
return m.content;
|
||||
}
|
||||
if (m?.role !== "user") continue;
|
||||
if (typeof m.content === "string") return m.content;
|
||||
if (Array.isArray(m.content)) return extractText(m.content);
|
||||
}
|
||||
return "";
|
||||
}
|
||||
@@ -113,6 +186,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) {
|
||||
if (m.role) { h.update("\0"); h.update(m.role); }
|
||||
if (typeof m.content === "string" && m.content) {
|
||||
h.update("\0"); h.update(m.content);
|
||||
} else if (Array.isArray(m.content)) {
|
||||
// Include image refs so the same prompt with a different image gets
|
||||
// a distinct chat_record_id.
|
||||
h.update("\0");
|
||||
try { h.update(JSON.stringify(m.content)); } catch {}
|
||||
}
|
||||
}
|
||||
if (tools) {
|
||||
@@ -130,16 +208,16 @@ function truncate(s, n) {
|
||||
/**
|
||||
* Map the OpenAI-style request body into the exact shape Qoder expects.
|
||||
*/
|
||||
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }) {
|
||||
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null, region = "intl" }) {
|
||||
const qoderKey = String(model || "").replace(/^qoder\//, "");
|
||||
|
||||
|
||||
// Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP.
|
||||
// This allows support for new Qoder models (e.g., qmodel_latest) without code changes.
|
||||
let modelConfig = await getQoderModelConfig(credentials, qoderKey, { log, proxyOptions, signal });
|
||||
let modelConfig = await getQoderModelConfig(credentials, qoderKey, { log, proxyOptions, signal, region });
|
||||
if (!modelConfig) {
|
||||
// Try a forced refresh once before giving up — the cache may simply
|
||||
// not be populated yet on first ever call for this credential.
|
||||
const refreshed = await resolveQoderModels(credentials, { forceRefresh: true, log, proxyOptions, signal });
|
||||
const refreshed = await resolveQoderModels(credentials, { forceRefresh: true, log, proxyOptions, signal, region });
|
||||
const retried = refreshed?.rawConfigs.get(qoderKey);
|
||||
if (!retried) {
|
||||
throw new Error(
|
||||
@@ -149,7 +227,30 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
modelConfig = { ...retried, key: qoderKey };
|
||||
}
|
||||
|
||||
const { messages, systemText } = normalizeMessages(body.messages || []);
|
||||
const incoming = Array.isArray(body.messages)
|
||||
? body.messages.map((m) => {
|
||||
if (!m || typeof m !== "object") return m;
|
||||
return {
|
||||
...m,
|
||||
content: Array.isArray(m.content)
|
||||
? m.content.map((b) => (b && typeof b === "object" ? { ...b } : b))
|
||||
: m.content,
|
||||
};
|
||||
})
|
||||
: [];
|
||||
try {
|
||||
await rewriteQoderMessageAttachments(incoming, {
|
||||
credentials,
|
||||
log,
|
||||
proxyOptions,
|
||||
signal,
|
||||
uploadFn,
|
||||
});
|
||||
} catch (err) {
|
||||
log?.warn?.("QODER", `attachment rewrite failed: ${err.message}`);
|
||||
}
|
||||
|
||||
const { messages, systemText } = normalizeMessages(incoming);
|
||||
const tools = body.tools;
|
||||
const isReasoning = !!modelConfig.is_reasoning;
|
||||
const maxOutputTokens = Number(modelConfig.max_output_tokens) || 0;
|
||||
@@ -168,7 +269,21 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
const sessionId = stableHash("qoder-session", psd.userId, qoderKey);
|
||||
const recordId = stableChatRecordId(qoderKey, messages, tools, maxTokens);
|
||||
|
||||
return {
|
||||
// Context-window tier (200K/400K/1M): the IDE picks one from model_config.context_config;
|
||||
// qodercli-style requests default to the smallest. Escalate when the prompt no longer fits.
|
||||
const tierChoice = resolveQoderContextTier(
|
||||
modelConfig,
|
||||
{ system: systemText, messages, tools },
|
||||
{ preference: process.env[QODER_CONTEXT_TIER_ENV] },
|
||||
);
|
||||
if (tierChoice) {
|
||||
log?.info?.(
|
||||
"QODER",
|
||||
`context tier ${tierChoice.tier.name} (${tierChoice.tier.tokenCount} tokens, ${tierChoice.reason}) for ~${tierChoice.estimatedTokens} prompt tokens`,
|
||||
);
|
||||
}
|
||||
|
||||
const built = {
|
||||
qoderKey,
|
||||
payload: {
|
||||
request_id: uuidv4(),
|
||||
@@ -216,6 +331,72 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
},
|
||||
modelConfig,
|
||||
};
|
||||
if (tierChoice) applyQoderContextTier(built.payload, tierChoice.tier);
|
||||
return built;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a qoder error message indicates a billing/quota block.
|
||||
* Signatures: code 110 (billing daily count exceeded), code 112 (quota
|
||||
* exhausted), code 10605 (queue throttle), pricingUrl field.
|
||||
*/
|
||||
function isBillingBlock(inner) {
|
||||
if (!inner || typeof inner !== "string") return false;
|
||||
const lowerMsg = inner.toLowerCase();
|
||||
if (lowerMsg.includes("pricingurl")) return true;
|
||||
// Parsed code preferred over regex: matches numeric or string "110"/"112"/"10605".
|
||||
try {
|
||||
const parsed = JSON.parse(inner);
|
||||
const code = String(parsed?.code ?? "");
|
||||
if (code === "110" || code === "112" || code === "10605") return true;
|
||||
} catch { /* not JSON — fall through to legacy shape match */ }
|
||||
// Match legacy exact shapes: {"code":"112",...}, {"code":"10605",...}.
|
||||
return /"code"\s*:\s*"(112|10605)"/.test(inner);
|
||||
}
|
||||
|
||||
/**
|
||||
* Peek the first SSE data line to detect upstream errors before piping.
|
||||
* Returns { isError, isBilling, statusVal, message, consumed } — `consumed` is every
|
||||
* byte read so far (including the peeked line) so the caller can re-process
|
||||
* it and nothing is dropped from the stream.
|
||||
*/
|
||||
async function peekFirstQoderFrame(reader, decoder) {
|
||||
let consumed = "";
|
||||
let offset = 0;
|
||||
let upstreamDone = false;
|
||||
while (true) {
|
||||
let nl = consumed.indexOf("\n", offset);
|
||||
if (nl === -1 && !upstreamDone) {
|
||||
const { done, value } = await reader.read();
|
||||
upstreamDone = done;
|
||||
consumed += done ? decoder.decode() : decoder.decode(value, { stream: true });
|
||||
continue;
|
||||
}
|
||||
if (offset >= consumed.length) return { isError: false, consumed, upstreamDone };
|
||||
if (nl === -1) nl = consumed.length;
|
||||
|
||||
const line = consumed.slice(offset, nl).replace(/\r$/, "").trim();
|
||||
offset = nl + 1;
|
||||
if (!line.startsWith("data:")) continue;
|
||||
|
||||
const data = line.slice(5).trimStart();
|
||||
if (data === "[DONE]") return { isError: false, consumed, upstreamDone };
|
||||
|
||||
let envelope;
|
||||
try { envelope = JSON.parse(data); } catch { return { isError: false, consumed, upstreamDone }; }
|
||||
|
||||
// statusCodeValue is documented numeric, but accept numeric strings defensively.
|
||||
const raw = Number(envelope?.statusCodeValue);
|
||||
const statusVal = Number.isNaN(raw) ? 200 : raw;
|
||||
const inner = typeof envelope?.body === "string"
|
||||
? envelope.body
|
||||
: envelope?.body != null ? JSON.stringify(envelope.body) : "";
|
||||
|
||||
if (statusVal !== 200) {
|
||||
return { isError: true, isBilling: isBillingBlock(inner), statusVal, message: inner || `upstream status ${statusVal}` };
|
||||
}
|
||||
return { isError: false, consumed, upstreamDone };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -225,23 +406,55 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
* Each upstream line looks like:
|
||||
* data: {"statusCodeValue":200,"body":"{\"choices\":[{\"delta\":{...}}]}"}
|
||||
* The inner body is an OpenAI streaming chunk (or "[DONE]"). We unwrap it
|
||||
* and re-emit as `data: <inner>\n\n`. Errors become a synthetic OpenAI error
|
||||
* chunk + [DONE].
|
||||
* and re-emit as `data: <inner>\n\n`. First-frame errors become HTTP errors;
|
||||
* errors after streaming starts retain the synthetic chunk + [DONE] path.
|
||||
*
|
||||
* Critical: Qoder's SSE often keeps the socket open after the terminal
|
||||
* [DONE]/error frame (agent keepalive). Non-streaming clients drain via
|
||||
* response.text() which hangs until the socket closes — so on terminal
|
||||
* events we cancel the upstream reader and close our stream immediately.
|
||||
*
|
||||
* Usage: Qoder puts finish_reason on `delta` and sends token counts on a
|
||||
* later `choices: []` frame. Downstream OpenAI/Claude clients only read
|
||||
* usage from the finish chunk, so we coalesce those two frames (see
|
||||
* createQoderSseCoalescer) before forwarding.
|
||||
*
|
||||
* Peek the first frame for errors before committing to HTTP 200. Preserve
|
||||
* upstream error statuses so chatCore can handle failures instead of recording
|
||||
* error text as a successful completion. Billing blocks retain the existing
|
||||
* 403 mapping for quota/account fallback.
|
||||
*/
|
||||
function wrapQoderSSE(response, model) {
|
||||
async function wrapQoderSSE(response, model, log = null) {
|
||||
if (!response.ok || !response.body) return response;
|
||||
|
||||
const decoder = new TextDecoder();
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
let doneEmitted = false;
|
||||
const reader = response.body.getReader();
|
||||
|
||||
// Detect errors before returning a successful streaming response.
|
||||
const peek = await peekFirstQoderFrame(reader, decoder);
|
||||
if (peek.isError) {
|
||||
await reader.cancel().catch(() => {});
|
||||
const status = peek.isBilling
|
||||
? HTTP_STATUS.FORBIDDEN
|
||||
: Number.isInteger(peek.statusVal) && peek.statusVal >= HTTP_STATUS.BAD_REQUEST && peek.statusVal <= 599
|
||||
? peek.statusVal : HTTP_STATUS.BAD_GATEWAY;
|
||||
return new Response(
|
||||
JSON.stringify({ error: { message: peek.message, code: peek.statusVal } }),
|
||||
{ status, headers: { "Content-Type": "application/json" } }
|
||||
);
|
||||
}
|
||||
|
||||
// Normal flow: re-process every byte the peek consumed, then continue.
|
||||
let buffer = peek.consumed || "";
|
||||
const upstreamDrained = peek.upstreamDone === true;
|
||||
const encoder = new TextEncoder();
|
||||
let doneEmitted = false;
|
||||
const coalescer = createQoderSseCoalescer({ model, encoder, sseDone: SSE_DONE });
|
||||
|
||||
const syncDone = () => {
|
||||
if (coalescer.doneEmitted) doneEmitted = true;
|
||||
};
|
||||
|
||||
// Process one already-extracted SSE line (no trailing newline).
|
||||
const processLine = (line, controller) => {
|
||||
const trimmed = line.replace(/\r$/, "").trim();
|
||||
@@ -251,16 +464,42 @@ function wrapQoderSSE(response, model) {
|
||||
|
||||
const data = trimmed.slice(5).trimStart();
|
||||
if (data === "[DONE]") {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
doneEmitted = true;
|
||||
coalescer.flush(controller);
|
||||
syncDone();
|
||||
return;
|
||||
}
|
||||
|
||||
let envelope;
|
||||
try { envelope = JSON.parse(data); } catch { return; }
|
||||
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
|
||||
const inner = typeof envelope.body === "string" ? envelope.body : "";
|
||||
const statusVal = Number(envelope.statusCodeValue) || 200;
|
||||
const inner = typeof envelope.body === "string"
|
||||
? envelope.body
|
||||
: envelope.body != null ? JSON.stringify(envelope.body) : "";
|
||||
if (statusVal !== 200) {
|
||||
// Always visible: error envelopes are rare and worth one stderr line at
|
||||
// any log level (response bodies carry no credentials).
|
||||
try {
|
||||
console.error(`[QODER] error envelope status=${statusVal} statusType=${typeof envelope.statusCodeValue} bodyType=${typeof envelope.body} body=${truncate(inner, 300)}`);
|
||||
} catch { /* logging must not break the stream */ }
|
||||
if (isBillingBlock(inner)) {
|
||||
// Billing/quota envelope at any stream position (peek only covers the
|
||||
// first frame): emit a structured error chunk, not fake assistant text.
|
||||
// parseSSEToOpenAIResponse understands chunk.error and turns it into a
|
||||
// non-200 result so chat.js locks the model and falls back. Streaming
|
||||
// clients receive a real SSE error instead of "[qoder error ...]" text.
|
||||
const errObj = JSON.stringify({
|
||||
error: {
|
||||
message: inner || `qoder billing block (${statusVal})`,
|
||||
code: "qoder_billing_block",
|
||||
status: 403,
|
||||
type: "quota_error",
|
||||
},
|
||||
});
|
||||
controller.enqueue(encoder.encode(`data: ${errObj}\n\n`));
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
doneEmitted = true;
|
||||
return;
|
||||
}
|
||||
const msg = inner || `upstream status ${statusVal}`;
|
||||
const errChunk = JSON.stringify({
|
||||
id: `qoder-error-${Date.now()}`,
|
||||
@@ -275,14 +514,8 @@ function wrapQoderSSE(response, model) {
|
||||
return;
|
||||
}
|
||||
if (!inner) return;
|
||||
if (inner === "[DONE]") {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
doneEmitted = true;
|
||||
return;
|
||||
}
|
||||
// Strip embedded newlines so the SSE frame stays a single event.
|
||||
const sanitized = inner.replace(/\r?\n/g, "");
|
||||
controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`));
|
||||
coalescer.handleInner(inner, controller);
|
||||
syncDone();
|
||||
};
|
||||
|
||||
const stream = new ReadableStream({
|
||||
@@ -290,7 +523,28 @@ function wrapQoderSSE(response, model) {
|
||||
// enqueueing would never be re-invoked, hanging consumers like .text().
|
||||
async start(controller) {
|
||||
try {
|
||||
while (!doneEmitted) {
|
||||
// Drain whatever the peek already pulled off the socket first.
|
||||
let nlSeed;
|
||||
while ((nlSeed = buffer.indexOf("\n")) !== -1) {
|
||||
const line = buffer.slice(0, nlSeed);
|
||||
buffer = buffer.slice(nlSeed + 1);
|
||||
processLine(line, controller);
|
||||
if (doneEmitted) {
|
||||
await reader.cancel().catch(() => {});
|
||||
controller.close();
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (upstreamDrained) {
|
||||
// Peek hit end-of-stream: flush any trailing partial line.
|
||||
buffer += decoder.decode();
|
||||
if (buffer.length > 0) {
|
||||
processLine(buffer, controller);
|
||||
buffer = "";
|
||||
}
|
||||
}
|
||||
|
||||
while (!doneEmitted && !upstreamDrained) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) {
|
||||
buffer += decoder.decode();
|
||||
@@ -320,7 +574,7 @@ function wrapQoderSSE(response, model) {
|
||||
} finally {
|
||||
if (!doneEmitted) {
|
||||
try {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
coalescer.flush(controller);
|
||||
doneEmitted = true;
|
||||
} catch { /* already closed */ }
|
||||
}
|
||||
@@ -343,99 +597,14 @@ function wrapQoderSSE(response, model) {
|
||||
});
|
||||
}
|
||||
|
||||
// ── PAT (Personal Access Token) → job-token exchange ───────────────────────
|
||||
// PATs (pt-...) cannot sign COSY requests directly. Exchange them for a
|
||||
// short-lived job token (jt-...) via /api/v1/jobToken/exchange (plain JSON,
|
||||
// not COSY-signed), then resolve the userId from userinfo. Mirrors the
|
||||
// official qodercli flow. Cached per-PAT until near-expiry.
|
||||
const PAT_PREFIX = "pt-";
|
||||
const PAT_REFRESH_BUFFER_MS = 5 * 60 * 1000;
|
||||
const patJobCache = new Map();
|
||||
|
||||
export function isQoderPat(token) {
|
||||
return typeof token === "string" && token.startsWith(PAT_PREFIX);
|
||||
}
|
||||
|
||||
async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
{
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
"Cosy-Version": QODER_IDE_VERSION,
|
||||
"Cosy-ClientType": QODER_CLIENT_TYPE,
|
||||
},
|
||||
body: JSON.stringify({ personal_token: pat }),
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
throw new Error(`qoder PAT exchange failed: ${res.status} ${text.slice(0, 200)}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
if (!data.token) throw new Error("qoder PAT exchange returned no job token");
|
||||
|
||||
let expiresAt = Date.now() + 24 * 60 * 60 * 1000;
|
||||
if (data.expires_at) {
|
||||
const parsed = Date.parse(data.expires_at);
|
||||
if (!Number.isNaN(parsed)) expiresAt = parsed;
|
||||
} else if (typeof data.expires_in === "number" && data.expires_in > 0) {
|
||||
expiresAt = Date.now() + data.expires_in;
|
||||
}
|
||||
return { jobToken: data.token, jobRefreshToken: data.refresh_token || "", expiresAt };
|
||||
}
|
||||
|
||||
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null) {
|
||||
try {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_USERINFO_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${jobToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
},
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
const info = await res.json().catch(() => ({}));
|
||||
return info.id || info.userId || info.user_id || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Exchange a PAT for a job token + userId, caching until near-expiry so repeat
|
||||
* chat requests don't re-exchange. Returns { accessToken, userId }.
|
||||
*/
|
||||
async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
|
||||
const cached = patJobCache.get(pat);
|
||||
if (cached && cached.expiresAt - Date.now() > PAT_REFRESH_BUFFER_MS) {
|
||||
return cached;
|
||||
}
|
||||
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal);
|
||||
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal);
|
||||
const entry = { accessToken: jobToken, userId, expiresAt };
|
||||
patJobCache.set(pat, entry);
|
||||
return entry;
|
||||
}
|
||||
|
||||
export class QoderExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("qoder", PROVIDERS.qoder);
|
||||
constructor(provider = "qoder") {
|
||||
super(provider, PROVIDERS[provider]);
|
||||
this.region = provider === "qoder-cn" ? "cn" : "intl";
|
||||
}
|
||||
|
||||
buildUrl() {
|
||||
return QODER_CHAT_URL_ENCODED;
|
||||
buildUrl(credentials) {
|
||||
return `${qoderInferenceBase(credentials, this.region)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
|
||||
}
|
||||
|
||||
// Override execute entirely — Qoder needs:
|
||||
@@ -444,36 +613,24 @@ export class QoderExecutor extends BaseExecutor {
|
||||
// - COSY headers built from the *encoded* body bytes
|
||||
// - response stream re-wrapped from {statusCodeValue, body} to OpenAI SSE
|
||||
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
|
||||
const url = this.buildUrl();
|
||||
|
||||
// PAT (pt-...) → exchange for short-lived job token + resolve userId so
|
||||
// downstream COSY signing + catalog fetch work. Device tokens (dt-...) and
|
||||
// job tokens (jt-...) skip this and are used directly.
|
||||
const rawToken = credentials?.apiKey || credentials?.accessToken;
|
||||
if (isQoderPat(rawToken)) {
|
||||
try {
|
||||
const resolved = await resolvePatCredential(rawToken, proxyOptions, signal);
|
||||
credentials = {
|
||||
...credentials,
|
||||
accessToken: resolved.accessToken,
|
||||
apiKey: undefined,
|
||||
providerSpecificData: {
|
||||
authMethod: "pat",
|
||||
...(credentials?.providerSpecificData || {}),
|
||||
userId: resolved.userId || credentials?.providerSpecificData?.userId || "",
|
||||
machineId: credentials?.providerSpecificData?.machineId || "",
|
||||
},
|
||||
};
|
||||
credentials = await resolveQoderCredentials(credentials, proxyOptions, signal, this.region);
|
||||
} catch (err) {
|
||||
log?.error?.("QODER", `PAT exchange failed: ${err.message}`);
|
||||
const fakeResp = new Response(
|
||||
JSON.stringify({ error: { message: `qoder PAT exchange failed: ${err.message}` } }),
|
||||
{ status: 401, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
return { response: fakeResp, url, headers: {}, transformedBody: body };
|
||||
return { response: fakeResp, url: this.buildUrl(credentials), headers: {}, transformedBody: body };
|
||||
}
|
||||
}
|
||||
|
||||
const url = this.buildUrl(credentials);
|
||||
const psd = credentials?.providerSpecificData || {};
|
||||
if (!psd.userId) {
|
||||
// No user id → no way to sign. Surface a 401 so the dashboard nudges
|
||||
@@ -497,7 +654,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
let qoderKey;
|
||||
let payload;
|
||||
try {
|
||||
({ qoderKey, payload } = await buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }));
|
||||
({ qoderKey, payload } = await buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, region: this.region }));
|
||||
} catch (err) {
|
||||
const fakeResp = new Response(
|
||||
JSON.stringify({ error: { message: err.message } }),
|
||||
@@ -556,8 +713,15 @@ export class QoderExecutor extends BaseExecutor {
|
||||
response = await proxyAwareFetch(
|
||||
url,
|
||||
{ method: "POST", headers, body: encodedBodyBuf, signal: mergedSignal },
|
||||
proxyOptions,
|
||||
// A failed proxy request may already have reached Qoder. Replaying
|
||||
// the same COSY signature directly reuses its requestId and returns
|
||||
// 403/code 103. Let the caller retry through execute() with fresh signing.
|
||||
{ ...proxyOptions, strictProxy: true },
|
||||
);
|
||||
} catch (err) {
|
||||
// strictProxy wraps transport errors; retain caller cancellation semantics.
|
||||
if (mergedSignal.aborted) throw mergedSignal.reason;
|
||||
throw err;
|
||||
} finally {
|
||||
clearTimeout(connectTimer);
|
||||
}
|
||||
@@ -567,7 +731,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
return { response, url, headers, transformedBody: payload };
|
||||
}
|
||||
|
||||
const wrapped = wrapQoderSSE(response, `qoder/${qoderKey}`);
|
||||
const wrapped = await wrapQoderSSE(response, `${this.provider}/${qoderKey}`, log);
|
||||
return { response: wrapped, url, headers, transformedBody: payload };
|
||||
}
|
||||
|
||||
@@ -591,6 +755,5 @@ export const __test__ = {
|
||||
normalizeMessages,
|
||||
wrapQoderSSE,
|
||||
buildQoderRequestBody,
|
||||
isQoderPat,
|
||||
resolvePatCredential,
|
||||
isBillingBlock,
|
||||
};
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS } from "../config/appConstants.js";
|
||||
|
||||
/** portal.qwen.ai — static fingerprint matching stable Qwen Code release */
|
||||
const QWEN_USER_AGENT = "QwenCode/0.12.3 (linux; x64)";
|
||||
const QWEN_STAINLESS = {
|
||||
os: "Linux",
|
||||
arch: "x64",
|
||||
lang: "js",
|
||||
runtime: "node",
|
||||
runtimeVersion: "v18.19.1",
|
||||
packageVersion: "5.11.0",
|
||||
retryCount: "1"
|
||||
};
|
||||
const QWEN_DEFAULT_SYSTEM_MESSAGE = {
|
||||
role: "system",
|
||||
content: [{ type: "text", text: "", cache_control: { type: "ephemeral" } }]
|
||||
};
|
||||
|
||||
function ensureQwenSystemMessage(body) {
|
||||
if (!body || typeof body !== "object") return body;
|
||||
const next = { ...body };
|
||||
if (Array.isArray(next.messages)) {
|
||||
next.messages = [QWEN_DEFAULT_SYSTEM_MESSAGE, ...next.messages];
|
||||
} else {
|
||||
next.messages = [QWEN_DEFAULT_SYSTEM_MESSAGE];
|
||||
}
|
||||
return next;
|
||||
}
|
||||
|
||||
function isQwenThinkingActive(body) {
|
||||
const thinking = body?.thinking;
|
||||
if (thinking === true || body?.enable_thinking === true) return true;
|
||||
return typeof thinking === "object" && thinking !== null && !Array.isArray(thinking) && thinking.type === "enabled";
|
||||
}
|
||||
|
||||
// Qwen rejects tool_choice="required" or object forms when thinking is active; neutralize to "auto".
|
||||
function sanitizeQwenThinkingToolChoice(body) {
|
||||
if (!isQwenThinkingActive(body)) return body;
|
||||
const tc = body.tool_choice;
|
||||
const incompatible = tc === "required" || (typeof tc === "object" && tc !== null);
|
||||
if (!incompatible) return body;
|
||||
return { ...body, tool_choice: "auto" };
|
||||
}
|
||||
|
||||
function buildQwenUpstreamHeaders(credentials, stream = true) {
|
||||
const token = credentials?.apiKey || credentials?.accessToken || "";
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${token}`,
|
||||
"User-Agent": QWEN_USER_AGENT,
|
||||
"X-DashScope-AuthType": "qwen-oauth",
|
||||
"X-DashScope-CacheControl": "enable",
|
||||
"X-DashScope-UserAgent": QWEN_USER_AGENT,
|
||||
"X-Stainless-Arch": QWEN_STAINLESS.arch,
|
||||
"X-Stainless-Lang": QWEN_STAINLESS.lang,
|
||||
"X-Stainless-Os": QWEN_STAINLESS.os,
|
||||
"X-Stainless-Package-Version": QWEN_STAINLESS.packageVersion,
|
||||
"X-Stainless-Retry-Count": QWEN_STAINLESS.retryCount,
|
||||
"X-Stainless-Runtime": QWEN_STAINLESS.runtime,
|
||||
"X-Stainless-Runtime-Version": QWEN_STAINLESS.runtimeVersion,
|
||||
Connection: "keep-alive",
|
||||
"Accept-Language": "*",
|
||||
"Sec-Fetch-Mode": "cors"
|
||||
};
|
||||
headers.Accept = stream ? "text/event-stream" : "application/json";
|
||||
return headers;
|
||||
}
|
||||
|
||||
export class QwenExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("qwen");
|
||||
}
|
||||
|
||||
// Qwen tokens are bound to a resource_url returned at OAuth time.
|
||||
// Using portal.qwen.ai when the token is issued for another shard returns 401/403.
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
const resourceUrl = credentials?.providerSpecificData?.resourceUrl;
|
||||
const host = resourceUrl ? resourceUrl.replace(/^https?:\/\//, "").replace(/\/$/, "") : "portal.qwen.ai";
|
||||
return `https://${host}/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
return buildQwenUpstreamHeaders(credentials, stream);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
let next = body && typeof body === "object" ? { ...body } : body;
|
||||
if (stream && next?.messages && !next.stream_options && !next.thinking && !next.enable_thinking && next.stream !== false) {
|
||||
next.stream_options = { include_usage: true };
|
||||
}
|
||||
next = sanitizeQwenThinkingToolChoice(next);
|
||||
return ensureQwenSystemMessage(next);
|
||||
}
|
||||
|
||||
// Override to capture resource_url from refresh response (required for buildUrl).
|
||||
async refreshCredentials(credentials, log) {
|
||||
if (!credentials?.refreshToken) return null;
|
||||
try {
|
||||
const response = await fetch(OAUTH_ENDPOINTS.qwen.token, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json" },
|
||||
body: new URLSearchParams({
|
||||
grant_type: "refresh_token",
|
||||
refresh_token: credentials.refreshToken,
|
||||
client_id: PROVIDERS.qwen.clientId
|
||||
})
|
||||
});
|
||||
if (!response.ok) return null;
|
||||
const tokens = await response.json();
|
||||
log?.info?.("TOKEN", "qwen refreshed");
|
||||
return {
|
||||
accessToken: tokens.access_token,
|
||||
refreshToken: tokens.refresh_token || credentials.refreshToken,
|
||||
expiresIn: tokens.expires_in,
|
||||
providerSpecificData: {
|
||||
...(credentials.providerSpecificData || {}),
|
||||
...(tokens.resource_url ? { resourceUrl: tokens.resource_url } : {})
|
||||
}
|
||||
};
|
||||
} catch (error) {
|
||||
log?.error?.("TOKEN", `qwen refresh error: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export default QwenExecutor;
|
||||
116
open-sse/executors/xiaomi-mimo.js
Normal file
116
open-sse/executors/xiaomi-mimo.js
Normal file
@@ -0,0 +1,116 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { getMimoAccountCookie, invalidateMimoAccountCookieCache, resolveMimoServerBase, MIMO_API_UA } from "../shared/mimoAccount.js";
|
||||
|
||||
// Dual-route v2.6 models.
|
||||
// v2.6 models dynamically route to the account service when desktop session credentials
|
||||
// (mimoPassToken or account cookie) are present to consume weekly quota, falling back to
|
||||
// the cloud API (sk- key) otherwise.
|
||||
const ACCOUNT_MODELS = new Set([
|
||||
"mimo-v2.6-pro",
|
||||
"mimo-v2.6-flash",
|
||||
"mimo-v2.6-pro-ultraspeed",
|
||||
]);
|
||||
|
||||
// Session cookie resolved in execute() (async) and read back by buildHeaders()
|
||||
// (sync — BaseExecutor.execute does not await it). Carried on the per-request
|
||||
// credentials object, same as runtimeTransport.
|
||||
const COOKIE_KEY = "__mimoAccountCookie";
|
||||
|
||||
// Upstream calls may hand us either the bare id or a `provider/model` ref.
|
||||
function bareModel(model) {
|
||||
const s = String(model || "");
|
||||
const i = s.indexOf("/");
|
||||
return i >= 0 ? s.slice(i + 1) : s;
|
||||
}
|
||||
|
||||
export class XiaomiMimoExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("xiaomi-mimo");
|
||||
}
|
||||
|
||||
static isAccountRoute(model, credentials) {
|
||||
const bare = bareModel(model);
|
||||
if (!ACCOUNT_MODELS.has(bare)) return false;
|
||||
return Boolean(
|
||||
credentials?.[COOKIE_KEY] ||
|
||||
credentials?.providerSpecificData?.mimoPassToken
|
||||
);
|
||||
}
|
||||
|
||||
isAccountRoute(model, credentials) {
|
||||
return XiaomiMimoExecutor.isAccountRoute(model, credentials);
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
// Account route models live on the account-service route, which is not one of the
|
||||
// declared transports — resolve it before the default runtimeTransport path.
|
||||
if (this.isAccountRoute(model, credentials)) {
|
||||
return `${resolveMimoServerBase(credentials?.providerSpecificData)}/api/route/chat/completions`;
|
||||
}
|
||||
// Cloud API models keep default handling, so a Claude-format client reaches
|
||||
// the /anthropic/v1/messages transport.
|
||||
return super.buildUrl(model, stream, urlIndex, credentials);
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
if (this.isAccountRoute(model, credentials) && credentials?.[COOKIE_KEY]) {
|
||||
// Account route models authenticate with the account-session cookie, not the key.
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
Accept: stream ? "text/event-stream" : "application/json",
|
||||
"User-Agent": MIMO_API_UA,
|
||||
Cookie: credentials[COOKIE_KEY],
|
||||
};
|
||||
}
|
||||
return super.buildHeaders(credentials, stream, url, model);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
// super runs stripUnsupportedParams, which flattens content-part
|
||||
// arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js).
|
||||
const out = super.transformRequest(model, body, stream, credentials);
|
||||
|
||||
// Account route models: bridge reasoning_effort to official output_config.effort
|
||||
// (matches MiMo Desktop app.asar behavior).
|
||||
if (this.isAccountRoute(model, credentials)) {
|
||||
const rawEffort = out.reasoning_effort || body?.reasoning_effort || body?.output_config?.effort;
|
||||
if (rawEffort) {
|
||||
delete out.reasoning_effort;
|
||||
const norm = String(rawEffort).toLowerCase() === "xhigh" ? "high" : String(rawEffort).toLowerCase();
|
||||
out.output_config = { ...(out.output_config || {}), effort: norm };
|
||||
}
|
||||
|
||||
if (out.temperature == null) out.temperature = 1.0;
|
||||
if (out.top_p == null) out.top_p = 0.95;
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
const { model, credentials, proxyOptions = null } = args;
|
||||
if (!this.isAccountRoute(model, credentials)) return super.execute(args);
|
||||
|
||||
const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions);
|
||||
if (!cookie) {
|
||||
return super.execute(args);
|
||||
}
|
||||
credentials[COOKIE_KEY] = cookie;
|
||||
const result = await super.execute(args);
|
||||
|
||||
// A cached session can expire early — drop it and retry once with a fresh one.
|
||||
if (result.response.status === 401) {
|
||||
invalidateMimoAccountCookieCache();
|
||||
const fresh = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions).catch(() => null);
|
||||
if (fresh) {
|
||||
credentials[COOKIE_KEY] = fresh;
|
||||
return super.execute(args);
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
export const __test__ = { ACCOUNT_MODELS, bareModel, COOKIE_KEY };
|
||||
|
||||
export default XiaomiMimoExecutor;
|
||||
@@ -29,11 +29,18 @@ import {
|
||||
zedLlmFetch,
|
||||
} from "../shared/zedAuth.js";
|
||||
|
||||
// Wire values for the `provider` field of POST /completions. These are NOT
|
||||
// display names: cloud.zed.dev matches them exactly, and an unrecognized value
|
||||
// fails the whole request with `500 {"message":"An internal server error
|
||||
// occurred."}` before the model is ever looked at. Spellings come from Zed's
|
||||
// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore),
|
||||
// `x_ai` follows the same convention — so feeding a catalog value back through
|
||||
// normalizeZedProvider is identity.
|
||||
const ZED_PROVIDER = {
|
||||
anthropic: "Anthropic",
|
||||
openai: "OpenAi",
|
||||
google: "Google",
|
||||
xai: "XAi",
|
||||
anthropic: "anthropic",
|
||||
openai: "open_ai",
|
||||
google: "google",
|
||||
xai: "x_ai",
|
||||
};
|
||||
|
||||
function normalizeZedProvider(value, model) {
|
||||
@@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) {
|
||||
return openaiToClaudeRequest(model, body, true);
|
||||
}
|
||||
if (provider === ZED_PROVIDER.google) {
|
||||
return openaiToGeminiRequest(model, body, true);
|
||||
const geminiRequest = openaiToGeminiRequest(model, body, true);
|
||||
// Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the
|
||||
// public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`,
|
||||
// `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google
|
||||
// path so Zed applies its own defaults — scoped here so native Gemini/
|
||||
// Antigravity is untouched.
|
||||
delete geminiRequest.safetySettings;
|
||||
return geminiRequest;
|
||||
}
|
||||
if (provider === ZED_PROVIDER.openai) {
|
||||
return openaiToOpenAIResponsesRequest(model, body, true, credentials);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -4,11 +4,16 @@ import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
|
||||
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
|
||||
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine, tokensForDetail, shouldPersistRequestDetail } from "./requestDetail.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
import { restoreToolNames } from "../../utils/opencodeFingerprint.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
function parseToolArguments(value) {
|
||||
if (!value) return {};
|
||||
@@ -60,11 +65,93 @@ function openAICompletionToClaudeMessage(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions non-streaming response body into the
|
||||
* OpenAI Responses API shape. Used when a Responses-format client (e.g. Codex)
|
||||
* is routed to a Chat Completions upstream and `stream:false` — the streaming
|
||||
* path already emits Responses events, but the JSON path returned a raw
|
||||
* `chat.completion` body, so tool_calls were invisible to Responses clients.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function openAICompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
// Reasoning → a reasoning item (summary text), mirroring the streaming path.
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
// Assistant text → a message item with output_text content.
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
// tool_calls → function_call/custom_tool_call items (Responses-native tool shape).
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
const status = choice.finish_reason === "tool_calls" ? "completed" : (choice.finish_reason === "stop" ? "completed" : (choice.finish_reason || "completed"));
|
||||
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status,
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate non-streaming response body from provider format → OpenAI format.
|
||||
*/
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames = null) {
|
||||
if (targetFormat === sourceFormat) return responseBody;
|
||||
// Provider responded in OpenAI Chat Completions shape but the client speaks
|
||||
// Responses API — convert so tool_calls/text surface as Responses `output`.
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return openAICompletionToResponses(responseBody, customToolNames);
|
||||
}
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
||||
return openAICompletionToClaudeMessage(responseBody);
|
||||
}
|
||||
@@ -198,7 +285,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
||||
/**
|
||||
* Handle non-streaming response from provider.
|
||||
*/
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log, streamErrorPatterns, persistUsage = "all" }) {
|
||||
trackDone();
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
let responseBody;
|
||||
@@ -221,6 +308,11 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
}
|
||||
|
||||
// Unwrap before any consumer reads choices/usage so non-stream clients get a
|
||||
// bare OpenAI body and usage tracking sees data.usage. No-op unless the
|
||||
// provider opts in via transport.quirks.clineEnvelope.
|
||||
responseBody = unwrapClineEnvelope(responseBody, provider);
|
||||
|
||||
reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody);
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
@@ -233,15 +325,33 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Decloak tool_use names once on raw Claude body, before any translation (INPUT side)
|
||||
responseBody = decloakToolNames(responseBody, toolNameMap);
|
||||
|
||||
// Config-driven in-stream error detection: the HTTP call succeeded but the
|
||||
// assembled content signals an upstream failure — treat it as an error so
|
||||
// account/combo fallback and FAILED logging kick in (AGENTS.md hook #3).
|
||||
const matchedPattern = matchStreamErrorPatterns(
|
||||
streamErrorPatterns?.[provider],
|
||||
responseBody?.choices?.[0]?.message?.content || responseBody?.content || "",
|
||||
);
|
||||
if (matchedPattern) {
|
||||
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
|
||||
if (log?.errorLine) {
|
||||
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms (in-stream)\n Stream error pattern matched: ${matchedPattern}`);
|
||||
}
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Stream error pattern matched: ${matchedPattern}`);
|
||||
}
|
||||
|
||||
const usage = extractUsageFromResponse(responseBody);
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames)
|
||||
: responseBody;
|
||||
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
||||
// Responses-format translation produces a `object:"response"` body with no
|
||||
// `choices`; skip the Chat-Completions-specific post-processing below for it.
|
||||
const isResponsesResponse = sourceFormat === FORMATS.OPENAI_RESPONSES && translatedResponse?.object === "response";
|
||||
|
||||
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
||||
if (translatedResponse?.choices?.[0]) {
|
||||
@@ -254,13 +364,13 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
|
||||
// Ensure OpenAI-required fields
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
}
|
||||
|
||||
// Strip Azure-specific fields
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
@@ -274,7 +384,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
if (!isClaudeMessageResponse && translatedResponse?.choices) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse && translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
@@ -285,28 +395,30 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
reqLogger.logConvertedResponse(translatedResponse);
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: responseBody || null,
|
||||
response: {
|
||||
content: translatedResponse?.choices?.[0]?.message?.content || translatedResponse?.content || null,
|
||||
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
||||
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
||||
},
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save:", err.message);
|
||||
});
|
||||
if (shouldPersistRequestDetail(persistUsage, "success")) {
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId, apiKey,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: tokensForDetail(usage),
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: responseBody || null,
|
||||
response: {
|
||||
content: translatedResponse?.choices?.[0]?.message?.content || translatedResponse?.content || null,
|
||||
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
||||
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
||||
},
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save:", err.message);
|
||||
});
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(translatedResponse), {
|
||||
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" }
|
||||
response: new Response(JSON.stringify(restoreToolNames(translatedResponse, toolNameMap)), {
|
||||
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*", ...upstreamResponseHeaders(providerResponse.headers) }
|
||||
})
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { saveRequestUsage, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { saveRequestUsage, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { COLORS } from "../../utils/stream.js";
|
||||
import { canonicalizeUsage } from "../../utils/usageTracking.js";
|
||||
|
||||
@@ -25,10 +25,16 @@ export function extractUsageFromResponse(responseBody) {
|
||||
if (!responseBody || typeof responseBody !== "object") return null;
|
||||
|
||||
// Claude format
|
||||
// Note: OpenAI Responses usage ({input_tokens, input_tokens_details:{cached_tokens}})
|
||||
// also matches this branch. Its prompt is cache-INCLUSIVE and its cache rides in
|
||||
// input_tokens_details, so emit it as cached_tokens — the convention
|
||||
// canonicalizeUsage() passes through without folding. Reading it here keeps
|
||||
// cache accounting correct for /v1/responses and codex traffic.
|
||||
if (responseBody.usage?.input_tokens !== undefined) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usage.input_tokens || 0,
|
||||
completion_tokens: responseBody.usage.output_tokens || 0,
|
||||
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.input_tokens_details?.cached_tokens,
|
||||
cache_read_input_tokens: responseBody.usage.cache_read_input_tokens,
|
||||
cache_creation_input_tokens: responseBody.usage.cache_creation_input_tokens
|
||||
};
|
||||
@@ -39,29 +45,40 @@ export function extractUsageFromResponse(responseBody) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usage.prompt_tokens || 0,
|
||||
completion_tokens: responseBody.usage.completion_tokens || 0,
|
||||
cached_tokens: responseBody.usage.prompt_tokens_details?.cached_tokens,
|
||||
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.prompt_tokens_details?.cached_tokens,
|
||||
reasoning_tokens: responseBody.usage.completion_tokens_details?.reasoning_tokens
|
||||
};
|
||||
}
|
||||
|
||||
// Gemini format
|
||||
if (responseBody.usageMetadata) {
|
||||
// Gemini format. Antigravity / gemini-cli wrap the payload in { response: {...} }.
|
||||
const usageMetadata = responseBody.usageMetadata || responseBody.response?.usageMetadata;
|
||||
if (usageMetadata) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
|
||||
prompt_tokens: usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: usageMetadata.thoughtsTokenCount || 0
|
||||
};
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// Mask API keys before they reach the requestDetails data blob / DB column.
|
||||
// Only the prefix is kept — enough to distinguish keys without leaking them.
|
||||
export function maskApiKey(key) {
|
||||
if (!key || typeof key !== "string") return undefined;
|
||||
const trimmed = key.trim();
|
||||
if (trimmed.length <= 8) return trimmed.charAt(0) + "***";
|
||||
return trimmed.slice(0, 8) + "***";
|
||||
}
|
||||
|
||||
export function buildRequestDetail(base, overrides = {}) {
|
||||
return {
|
||||
provider: base.provider || "unknown",
|
||||
model: base.model || "unknown",
|
||||
connectionId: base.connectionId || undefined,
|
||||
apiKey: maskApiKey(base.apiKey),
|
||||
timestamp: new Date().toISOString(),
|
||||
latency: base.latency || { ttft: 0, total: 0 },
|
||||
tokens: base.tokens || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
@@ -93,6 +110,29 @@ export function formatDoneLine({ usage, latency }) {
|
||||
return `DONE ${latency?.total ?? 0}ms${ttftStr} · ${inStr} · OUT ${outTok}`;
|
||||
}
|
||||
|
||||
// Request-details storage convention: always prompt_tokens / completion_tokens.
|
||||
// Translators often hand Claude `{input_tokens, output_tokens}` (or Gemini
|
||||
// counts) to onStreamComplete; the Details tab only reads the OpenAI names,
|
||||
// so an uncanonicalized object shows up as input=0 / output=0.
|
||||
export function tokensForDetail(usage) {
|
||||
if (!usage || typeof usage !== "object") {
|
||||
return { prompt_tokens: 0, completion_tokens: 0 };
|
||||
}
|
||||
return canonicalizeUsage(usage) || {
|
||||
prompt_tokens: usage.prompt_tokens ?? usage.input_tokens ?? 0,
|
||||
completion_tokens: usage.completion_tokens ?? usage.output_tokens ?? 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Combo fallback/account hops must not inflate Details with 0-token rows.
|
||||
// `streaming-start` is never persisted: the placeholder was status=success at
|
||||
// tokens=0, and nested/fusion paths often abandon the stream before complete.
|
||||
export function shouldPersistRequestDetail(persistUsage, kind) {
|
||||
if (kind === "streaming-start") return false;
|
||||
if (persistUsage === "success-only") return kind === "success";
|
||||
return true;
|
||||
}
|
||||
|
||||
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE", silent = false }) {
|
||||
if (!tokens || typeof tokens !== "object") return;
|
||||
|
||||
|
||||
@@ -1,20 +1,17 @@
|
||||
import { convertResponsesStreamToJson } from "../../transformer/streamToJsonConverter.js";
|
||||
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
||||
import { restoreToolNames } from "../../utils/opencodeFingerprint.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
saveUsageStats,
|
||||
formatDoneLine,
|
||||
} from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
|
||||
const isResponsesProvider = (p) =>
|
||||
PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
import { saveRequestDetail, appendRequestLog } from "@/lib/usageDb.js";
|
||||
|
||||
function textFromResponsesMessageItem(item) {
|
||||
if (!item?.content || !Array.isArray(item.content)) return "";
|
||||
@@ -41,6 +38,76 @@ function pickAssistantMessageForChatCompletion(output) {
|
||||
return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions JSON body into the Responses API shape.
|
||||
* Inlined here (not imported from nonStreamingHandler.js) to avoid a circular
|
||||
* import. Mirrors openAICompletionToResponses in nonStreamingHandler.js.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function chatCompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse OpenAI-style SSE text into a single chat completion JSON.
|
||||
* Used when provider forces streaming but client wants non-streaming.
|
||||
@@ -136,6 +203,7 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
export async function handleForcedSSEToJson({
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
provider,
|
||||
model,
|
||||
body,
|
||||
@@ -147,17 +215,14 @@ export async function handleForcedSSEToJson({
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
customToolNames,
|
||||
toolNameMap,
|
||||
trackDone,
|
||||
appendLog,
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
}) {
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
const isSSE =
|
||||
contentType.includes("text/event-stream") ||
|
||||
(contentType === "" && isResponsesProvider(provider));
|
||||
if (!isSSE) return null; // not handled here
|
||||
|
||||
trackDone();
|
||||
|
||||
@@ -170,8 +235,11 @@ export async function handleForcedSSEToJson({
|
||||
};
|
||||
|
||||
// Codex/Responses API SSE path
|
||||
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
|
||||
// not the client format: a Responses-API client behind a chat-native forced-streaming
|
||||
// provider still receives chat SSE chunks, which must go through the standard path.
|
||||
const isCodexResponsesApi =
|
||||
isResponsesProvider(provider) || sourceFormat === FORMATS.OPENAI_RESPONSES;
|
||||
isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
if (isCodexResponsesApi) {
|
||||
try {
|
||||
const jsonResponse = await convertResponsesStreamToJson(
|
||||
@@ -200,6 +268,11 @@ export async function handleForcedSSEToJson({
|
||||
}),
|
||||
);
|
||||
|
||||
// Same cache-inclusive total for the recorded detail, so the DB and the
|
||||
// client-facing usage can never disagree.
|
||||
const inTokensForLog = (usage.input_tokens || 0)
|
||||
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0)
|
||||
+ (usage.cache_creation_input_tokens || 0);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(
|
||||
jsonResponse.output,
|
||||
);
|
||||
@@ -209,9 +282,10 @@ export async function handleForcedSSEToJson({
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
apiKey,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: {
|
||||
prompt_tokens: usage.input_tokens || 0,
|
||||
prompt_tokens: inTokensForLog,
|
||||
completion_tokens: usage.output_tokens || 0,
|
||||
},
|
||||
response: {
|
||||
@@ -229,7 +303,7 @@ export async function handleForcedSSEToJson({
|
||||
if (sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(jsonResponse), {
|
||||
response: new Response(JSON.stringify(restoreToolNames(jsonResponse, toolNameMap)), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
@@ -238,9 +312,22 @@ export async function handleForcedSSEToJson({
|
||||
};
|
||||
}
|
||||
|
||||
// Build client-format response
|
||||
const inTokens = usage.input_tokens || 0;
|
||||
// Build client-format response.
|
||||
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
|
||||
// only input+output under-reports prompt_tokens — measured: 2012 reported
|
||||
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache
|
||||
// counters in, and keep them visible in prompt_tokens_details so a client can
|
||||
// tell a cache hit from a small prompt.
|
||||
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens || 0;
|
||||
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
|
||||
const outTokens = usage.output_tokens || 0;
|
||||
const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
|
||||
? {
|
||||
prompt_tokens_details: {
|
||||
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
|
||||
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
|
||||
: {};
|
||||
let finalResp;
|
||||
|
||||
// Extract tool calls from Responses API output (function_call items)
|
||||
@@ -309,13 +396,14 @@ export async function handleForcedSSEToJson({
|
||||
prompt_tokens: inTokens,
|
||||
completion_tokens: outTokens,
|
||||
total_tokens: inTokens + outTokens,
|
||||
...cacheDetails,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(finalResp), {
|
||||
response: new Response(JSON.stringify(restoreToolNames(finalResp, toolNameMap)), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
@@ -341,8 +429,16 @@ export async function handleForcedSSEToJson({
|
||||
"Invalid SSE response for non-streaming request",
|
||||
);
|
||||
if (parsed.error) {
|
||||
// Structured error chunks may carry the real upstream status (e.g. the
|
||||
// Qoder executor emits status 403 for billing envelopes). Preserve it so
|
||||
// the account loop locks/falls back on the right status instead of a
|
||||
// generic 502. Anything outside 400-599 still maps to 502.
|
||||
const upstreamStatus = Number(parsed.error.status);
|
||||
const status = Number.isInteger(upstreamStatus) && upstreamStatus >= 400 && upstreamStatus <= 599
|
||||
? upstreamStatus
|
||||
: HTTP_STATUS.BAD_GATEWAY;
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
status,
|
||||
parsed.error.message || "Upstream SSE stream failed",
|
||||
);
|
||||
}
|
||||
@@ -384,23 +480,14 @@ export async function handleForcedSSEToJson({
|
||||
}),
|
||||
);
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: usage,
|
||||
response: {
|
||||
content: parsed.choices?.[0]?.message?.content || null,
|
||||
thinking: parsed.choices?.[0]?.message?.reasoning_content || null,
|
||||
finish_reason: parsed.choices?.[0]?.finish_reason || "unknown",
|
||||
},
|
||||
status: "success",
|
||||
},
|
||||
{ endpoint: clientRawRequest?.endpoint || null },
|
||||
),
|
||||
).catch(() => {});
|
||||
// Re-attach usage explicitly. This handler already HAS the correct usage — it is
|
||||
// the same object written to the usage DB, and for a cached Claude request that DB
|
||||
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
|
||||
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
|
||||
// on both sides). Whatever drops it between assembly and serialisation, the client
|
||||
// must not be left unable to account for its own token spend: a caller cannot tell
|
||||
// a 90%-cached request from a cheap one without this.
|
||||
if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
|
||||
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
@@ -414,9 +501,19 @@ export async function handleForcedSSEToJson({
|
||||
}
|
||||
}
|
||||
|
||||
// A Responses-format client (e.g. Codex) forced this provider to stream,
|
||||
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
|
||||
// body; convert it to the Responses `output` shape so tool_calls are not
|
||||
// lost on the non-streaming return path. Inlined (not imported from
|
||||
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
|
||||
// already imports parseSSEToOpenAIResponse from this module.
|
||||
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
|
||||
? chatCompletionToResponses(parsed, customToolNames)
|
||||
: parsed;
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(parsed), {
|
||||
response: new Response(JSON.stringify(restoreToolNames(finalBody, toolNameMap)), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
|
||||
@@ -6,17 +6,21 @@ import {
|
||||
} from "../../utils/stream.js";
|
||||
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildStreamErrorBytes } from "../../utils/streamHelpers.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
saveUsageStats,
|
||||
formatDoneLine,
|
||||
tokensForDetail,
|
||||
shouldPersistRequestDetail,
|
||||
} from "./requestDetail.js";
|
||||
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js";
|
||||
|
||||
// Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat.
|
||||
// Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI.
|
||||
@@ -31,60 +35,20 @@ const CODEX_SOURCE_TO_TARGET = {
|
||||
/**
|
||||
* Determine which SSE transform stream to use based on provider/format.
|
||||
*/
|
||||
function buildTransformStream({
|
||||
provider,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
}) {
|
||||
const isDroidCLI =
|
||||
userAgent?.toLowerCase().includes("droid") ||
|
||||
userAgent?.toLowerCase().includes("codex-cli");
|
||||
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
|
||||
const isResponsesProvider =
|
||||
PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
const needsCodexTranslation =
|
||||
isResponsesProvider &&
|
||||
targetFormat === FORMATS.OPENAI_RESPONSES &&
|
||||
!isDroidCLI;
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) {
|
||||
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
|
||||
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
|
||||
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
const needsCodexTranslation = isResponsesProvider && targetFormat === FORMATS.OPENAI_RESPONSES && !isDroidCLI;
|
||||
|
||||
if (needsCodexTranslation) {
|
||||
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
|
||||
return createSSETransformStreamWithLogger(
|
||||
FORMATS.OPENAI_RESPONSES,
|
||||
codexTarget,
|
||||
provider,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
);
|
||||
}
|
||||
if (needsCodexTranslation) {
|
||||
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
|
||||
}
|
||||
|
||||
if (needsTranslation(targetFormat, sourceFormat)) {
|
||||
return createSSETransformStreamWithLogger(
|
||||
targetFormat,
|
||||
sourceFormat,
|
||||
provider,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
);
|
||||
}
|
||||
if (needsTranslation(targetFormat, sourceFormat)) {
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
|
||||
}
|
||||
|
||||
return createPassthroughStreamWithLogger(
|
||||
provider,
|
||||
@@ -100,41 +64,14 @@ function buildTransformStream({
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export async function handleStreamingResponse({
|
||||
providerResponse,
|
||||
provider,
|
||||
model,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
userAgent,
|
||||
body,
|
||||
stream,
|
||||
translatedBody,
|
||||
finalBody,
|
||||
requestStartTime,
|
||||
connectionId,
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
streamController,
|
||||
onStreamComplete,
|
||||
streamDetailId,
|
||||
pxpipe,
|
||||
reqTag,
|
||||
log,
|
||||
}) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
.catch((err) => {
|
||||
console.error(
|
||||
"[ChatCore] onRequestSuccess failed:",
|
||||
err?.message || err,
|
||||
);
|
||||
});
|
||||
}
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, pxpipe, reqTag, log, credentials }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
.catch(err => {
|
||||
console.error("[ChatCore] onRequestSuccess failed:", err?.message || err);
|
||||
});
|
||||
}
|
||||
|
||||
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
|
||||
// page), piping it through the SSE transform stream causes Next.js
|
||||
@@ -193,27 +130,23 @@ export async function handleStreamingResponse({
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({
|
||||
provider,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
});
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
|
||||
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
// Terminal bytes when the stream aborts after HTTP 200 was already sent, so the
|
||||
// client sees a real error instead of a silently truncated stream.
|
||||
// Responses passthrough keeps its own response.failed shape; every other client
|
||||
// format gets the OpenAI error frame + [DONE], or `event: error` for Claude.
|
||||
const isResponsesPassthrough =
|
||||
sourceFormat === FORMATS.OPENAI_RESPONSES &&
|
||||
targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough
|
||||
? buildAbortedResponsesTerminalBytes
|
||||
: null;
|
||||
: (message) =>
|
||||
buildStreamErrorBytes(
|
||||
HTTP_STATUS.GATEWAY_TIMEOUT,
|
||||
message,
|
||||
sourceFormat,
|
||||
);
|
||||
const stallTimeoutMs =
|
||||
PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(
|
||||
@@ -224,37 +157,14 @@ export async function handleStreamingResponse({
|
||||
stallTimeoutMs,
|
||||
);
|
||||
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: "[Streaming - raw response not captured]",
|
||||
response: {
|
||||
content: "[Streaming in progress...]",
|
||||
thinking: null,
|
||||
type: "streaming",
|
||||
},
|
||||
pxpipe,
|
||||
status: "success",
|
||||
},
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to save streaming request:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(transformedBody, { headers: SSE_HEADERS }),
|
||||
response: new Response(transformedBody, {
|
||||
headers: {
|
||||
...SSE_HEADERS,
|
||||
...upstreamResponseHeaders(providerResponse.headers),
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -276,6 +186,7 @@ export function buildOnStreamComplete({
|
||||
reqTag,
|
||||
log,
|
||||
streamErrorPatterns,
|
||||
persistUsage = "all",
|
||||
}) {
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
|
||||
@@ -286,37 +197,41 @@ export function buildOnStreamComplete({
|
||||
};
|
||||
const safeContent = contentObj?.content || "[Empty streaming response]";
|
||||
const safeThinking = contentObj?.thinking || null;
|
||||
const rawProviderText = typeof contentObj?.rawProviderText === "string" ? contentObj.rawProviderText : "";
|
||||
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
latency,
|
||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: safeContent,
|
||||
response: {
|
||||
content: safeContent,
|
||||
thinking: safeThinking,
|
||||
type: "streaming",
|
||||
if (shouldPersistRequestDetail(persistUsage, "success")) {
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
provider,
|
||||
model,
|
||||
connectionId,
|
||||
apiKey,
|
||||
latency,
|
||||
tokens: tokensForDetail(usage),
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: rawProviderText || safeContent,
|
||||
response: {
|
||||
content: safeContent,
|
||||
thinking: safeThinking,
|
||||
type: "streaming",
|
||||
},
|
||||
pxpipe,
|
||||
status: streamStatusForContent(
|
||||
streamErrorPatterns?.[provider],
|
||||
safeContent,
|
||||
),
|
||||
},
|
||||
pxpipe,
|
||||
status: streamStatusForContent(
|
||||
streamErrorPatterns?.[provider],
|
||||
safeContent,
|
||||
),
|
||||
},
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to update streaming content:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
{ id: streamDetailId },
|
||||
),
|
||||
).catch((err) => {
|
||||
console.error(
|
||||
"[RequestDetail] Failed to update streaming content:",
|
||||
err.message,
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||
saveUsageStats({
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
import createOpenAIEmbeddingAdapter from "./openai.js";
|
||||
import gemini from "./gemini.js";
|
||||
import openaiCompatNode from "./openaiCompatNode.js";
|
||||
import selfhostedEmbedding from "./selfhostedEmbedding.js";
|
||||
|
||||
const OPENAI_COMPAT_PROVIDERS = [
|
||||
"openai", "openrouter", "mistral", "voyage-ai", "fireworks",
|
||||
@@ -13,6 +14,12 @@ const ADAPTERS = {
|
||||
...Object.fromEntries(OPENAI_COMPAT_PROVIDERS.map((id) => [id, createOpenAIEmbeddingAdapter(id)])),
|
||||
gemini,
|
||||
google_ai_studio: gemini,
|
||||
// Self-hosted reads creds.providerSpecificData.baseUrl (one provider, many
|
||||
// servers) — but via its OWN adapter, not openaiCompatNode: that one falls back
|
||||
// to api.openai.com when no baseUrl is set, which under a provider called
|
||||
// "Self-hosted Embedding" means silently shipping the input and API key to
|
||||
// OpenAI. selfhostedEmbedding refuses instead.
|
||||
"selfhosted-embedding": selfhostedEmbedding,
|
||||
};
|
||||
|
||||
export function getEmbeddingAdapter(provider) {
|
||||
|
||||
46
open-sse/handlers/embeddingProviders/selfhostedEmbedding.js
Normal file
46
open-sse/handlers/embeddingProviders/selfhostedEmbedding.js
Normal file
@@ -0,0 +1,46 @@
|
||||
// Self-hosted embeddings — like openaiCompatNode, but the baseUrl is REQUIRED.
|
||||
//
|
||||
// openaiCompatNode falls back to https://api.openai.com/v1 when a connection
|
||||
// carries no providerSpecificData.baseUrl. For a custom NODE that default is
|
||||
// defensible: the node was created by pointing at some OpenAI-compatible URL, and
|
||||
// OpenAI is the archetype. For a provider whose entire purpose is "my own
|
||||
// server", it is actively harmful — a connection saved without a baseUrl sends
|
||||
// the INPUT TEXT and the API KEY to OpenAI, silently, under a provider named
|
||||
// "Self-hosted Embedding".
|
||||
//
|
||||
// Observed exactly that with a placeholder connection (2026-08-04):
|
||||
//
|
||||
// [selfhosted-embedding/embedding] [401]: Incorrect API key provided: abc.
|
||||
// You can find your API key at https://platform.openai.com/account/api-keys.
|
||||
//
|
||||
// The key "abc" was typed as a throwaway for a LOCAL server and left the network.
|
||||
// A self-hosted provider must never have a cloud fallback, so this one refuses
|
||||
// instead: no baseUrl means a configuration error, reported as such.
|
||||
import createOpenAIEmbeddingAdapter from "./openai.js";
|
||||
|
||||
const baseAdapter = createOpenAIEmbeddingAdapter("openai");
|
||||
|
||||
export class MissingBaseUrlError extends Error {
|
||||
constructor() {
|
||||
super(
|
||||
"Self-hosted Embedding needs an endpoint: set this connection's baseUrl to " +
|
||||
"the OpenAI base URL of your server, e.g. http://host:8080/v1 (note the /v1 — " +
|
||||
"\"/embeddings\" is appended to it). Refusing to fall back to api.openai.com, " +
|
||||
"which would send your input and API key to OpenAI."
|
||||
);
|
||||
this.name = "MissingBaseUrlError";
|
||||
this.isConfigError = true;
|
||||
}
|
||||
}
|
||||
|
||||
export default {
|
||||
...baseAdapter,
|
||||
buildUrl: (_model, creds) => {
|
||||
const rawBaseUrl = creds?.providerSpecificData?.baseUrl;
|
||||
if (!rawBaseUrl || !String(rawBaseUrl).trim()) throw new MissingBaseUrlError();
|
||||
// Accept either the OpenAI base or a full embeddings URL, so a value pasted
|
||||
// from a curl example works as well as one typed from the help text.
|
||||
const baseUrl = String(rawBaseUrl).trim().replace(/\/$/, "").replace(/\/embeddings$/, "");
|
||||
return `${baseUrl}/embeddings`;
|
||||
},
|
||||
};
|
||||
@@ -1,5 +1,5 @@
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS, FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { getExecutor } from "../executors/index.js";
|
||||
import { refreshWithRetry } from "../services/tokenRefresh.js";
|
||||
import { getEmbeddingAdapter } from "./embeddingProviders/index.js";
|
||||
@@ -38,13 +38,24 @@ export async function handleEmbeddingsCore({
|
||||
}
|
||||
|
||||
const ctx = { input };
|
||||
const url = adapter.buildUrl(model, credentials, ctx);
|
||||
const headers = adapter.buildHeaders(credentials, ctx);
|
||||
const requestBody = adapter.buildBody(model, {
|
||||
input,
|
||||
encoding_format: body.encoding_format || "float",
|
||||
dimensions: body.dimensions,
|
||||
});
|
||||
// buildUrl/buildHeaders/buildBody were called bare. An adapter that rejects a
|
||||
// misconfigured connection — selfhosted-embedding throws when no baseUrl is set
|
||||
// rather than silently falling back to api.openai.com — would have escaped this
|
||||
// function uncaught, surfacing as a 500 or a request that never settles. A
|
||||
// configuration mistake is a 400 with the reason in it.
|
||||
let url, headers, requestBody;
|
||||
try {
|
||||
url = adapter.buildUrl(model, credentials, ctx);
|
||||
headers = adapter.buildHeaders(credentials, ctx);
|
||||
requestBody = adapter.buildBody(model, {
|
||||
input,
|
||||
encoding_format: body.encoding_format || "float",
|
||||
dimensions: body.dimensions,
|
||||
});
|
||||
} catch (error) {
|
||||
log?.debug?.("EMBEDDINGS", `Request build failed: ${error.message}`);
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}/${model}] ${error.message}`);
|
||||
}
|
||||
|
||||
log?.debug?.("EMBEDDINGS", `${provider.toUpperCase()} | ${model} | input_type=${Array.isArray(input) ? `array[${input.length}]` : "string"}`);
|
||||
|
||||
@@ -54,6 +65,9 @@ export async function handleEmbeddingsCore({
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify(requestBody),
|
||||
...(typeof AbortSignal?.timeout === "function"
|
||||
? { signal: AbortSignal.timeout(FETCH_CONNECT_TIMEOUT_MS) }
|
||||
: {}),
|
||||
});
|
||||
} catch (error) {
|
||||
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa
|
||||
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama
|
||||
// Returns normalized shape across all providers
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 15000;
|
||||
@@ -56,8 +56,8 @@ function parseJinaTitle(text) {
|
||||
return m ? m[1].trim() : null;
|
||||
}
|
||||
|
||||
function buildData({ provider, url, title, format, text, costUsd, responseMs, upstreamMs }) {
|
||||
return {
|
||||
function buildData({ provider, url, title, format, text, links, costUsd, responseMs, upstreamMs }) {
|
||||
const data = {
|
||||
provider,
|
||||
url,
|
||||
title: title || null,
|
||||
@@ -66,6 +66,8 @@ function buildData({ provider, url, title, format, text, costUsd, responseMs, up
|
||||
usage: { fetch_cost_usd: costUsd ?? null },
|
||||
metrics: { response_time_ms: responseMs, upstream_latency_ms: upstreamMs }
|
||||
};
|
||||
if (Array.isArray(links)) data.links = links;
|
||||
return data;
|
||||
}
|
||||
|
||||
async function readJsonOrText(res) {
|
||||
@@ -115,6 +117,18 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
|
||||
if (provider === "exa") {
|
||||
return await runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt });
|
||||
}
|
||||
if (provider === "ollama") {
|
||||
return await runOllama({
|
||||
url,
|
||||
fmt,
|
||||
timeoutMs,
|
||||
apiKey,
|
||||
maxCharacters,
|
||||
costPerQuery,
|
||||
startedAt,
|
||||
baseUrl: providerConfig?.baseUrl,
|
||||
});
|
||||
}
|
||||
return { success: false, status: 400, error: `Unsupported provider: ${provider}` };
|
||||
} catch (err) {
|
||||
log?.("fetch handler error:", err?.message || err);
|
||||
@@ -241,3 +255,56 @@ async function runExa({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery
|
||||
})
|
||||
};
|
||||
}
|
||||
|
||||
async function runOllama({
|
||||
url,
|
||||
fmt,
|
||||
timeoutMs,
|
||||
apiKey,
|
||||
maxCharacters,
|
||||
costPerQuery,
|
||||
startedAt,
|
||||
baseUrl,
|
||||
}) {
|
||||
const upstreamStart = Date.now();
|
||||
const r = await tryFetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"content-type": "application/json",
|
||||
...(apiKey ? { authorization: `Bearer ${apiKey}` } : {})
|
||||
},
|
||||
body: JSON.stringify({ url })
|
||||
}, timeoutMs);
|
||||
|
||||
if (!r.ok) {
|
||||
return { success: false, status: r.timeout ? 504 : 502, error: r.error };
|
||||
}
|
||||
const upstreamMs = Date.now() - upstreamStart;
|
||||
const { json, text: responseText } = await readJsonOrText(r.res);
|
||||
if (!r.res.ok) {
|
||||
const error = json?.error
|
||||
|| json?.message
|
||||
|| responseText?.slice(0, 500)
|
||||
|| `Ollama error: ${r.res.status}`;
|
||||
return { success: false, status: r.res.status, error };
|
||||
}
|
||||
if (!json || typeof json.content !== "string") {
|
||||
return { success: false, status: 502, error: "Ollama returned an empty or invalid web fetch response" };
|
||||
}
|
||||
|
||||
const text = truncate(json.content, maxCharacters);
|
||||
return {
|
||||
success: true,
|
||||
data: buildData({
|
||||
provider: "ollama",
|
||||
url,
|
||||
title: json.title || null,
|
||||
format: fmt,
|
||||
text,
|
||||
links: json.links,
|
||||
costUsd: costPerQuery,
|
||||
responseMs: Date.now() - startedAt,
|
||||
upstreamMs
|
||||
})
|
||||
};
|
||||
}
|
||||
|
||||
266
open-sse/handlers/geminiLiveStt.js
Normal file
266
open-sse/handlers/geminiLiveStt.js
Normal file
@@ -0,0 +1,266 @@
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
// Gemini Live API realtime STT transport.
|
||||
//
|
||||
// The REST generateContent path (sttCore.transcribeGemini) only transcribes
|
||||
// whole files inline. The Live API's `:bidiGenerateContent` WebSocket is the
|
||||
// streaming counterpart: audio goes up as realtimeInput mediaChunks and the
|
||||
// server pushes incremental `serverContent.inputTranscription` events back.
|
||||
// This module owns the socket lifecycle only — envelope/response shaping
|
||||
// stays in sttCore so the engine's single STT exit shape is preserved.
|
||||
//
|
||||
// Marker contract: dispatched from sttCore's format-switch when the model
|
||||
// entry carries `transport: "gemini-live"` (registry) or the caller passes a
|
||||
// transport string (custom models). Never keyed on a hardcoded model id here.
|
||||
//
|
||||
// Transport behavior:
|
||||
// - Node >= 22 global WebSocket (undici). No new dependency.
|
||||
// - Live API expects low-latency PCM; other containers are forwarded with
|
||||
// their declared MIME unchanged (provider-side rejection is surfaced).
|
||||
// - Text accumulation is append-only over inputTranscription segments and
|
||||
// ends on serverContent.turnComplete (or graceful close with partial text).
|
||||
// - Transcription deltas are kept per-frame (chunks[]) so sttCore can shape
|
||||
// verbose_json segments without fabricating timestamps. goAway advisements
|
||||
// rotate the socket once per call: setup replay + byte-offset resume.
|
||||
|
||||
const SETUP_TIMEOUT_MS = 10_000; // open → setupComplete
|
||||
const TURN_TIMEOUT_MS = 60_000; // audio streamed → turnComplete
|
||||
const MAX_TIMEOUT_MS = 300_000; // clamp ceiling for client-supplied lifecycle knobs
|
||||
const CHUNK_BYTES = 16_384; // ~0.5s of 16-bit 16kHz mono PCM
|
||||
const GOAWAY_RECONNECTS = 1; // socket rotations honoured per call
|
||||
|
||||
class GeminiLiveError extends Error {
|
||||
constructor(message, status) {
|
||||
super(message);
|
||||
this.name = "GeminiLiveError";
|
||||
this.status = status || 502;
|
||||
}
|
||||
}
|
||||
|
||||
// REST base (https://host/v1beta/models) → Live WS base
|
||||
// (wss://host/ws/api/v1beta/models), then the bidiGenerateContent endpoint.
|
||||
function toLiveWsUrl(baseUrl, model, token) {
|
||||
const url = new URL(baseUrl);
|
||||
url.protocol = "wss:";
|
||||
if (!url.pathname.startsWith("/ws/")) url.pathname = `/ws/api${url.pathname}`;
|
||||
const base = url.toString().replace(/\/+$/, "");
|
||||
return `${base}/${encodeURIComponent(model)}:bidiGenerateContent?key=${encodeURIComponent(token || "")}`;
|
||||
}
|
||||
|
||||
// Bind socket events supporting BOTH handler styles: addEventListener
|
||||
// (browser WebSocket, undici) and onopen/onmessage property assignment
|
||||
// (minimal polyfills). Whichever the implementation exposes, it works.
|
||||
function bindSocket(ws, { onOpen, onMessage, onError, onClose }) {
|
||||
if (typeof ws.addEventListener === "function") {
|
||||
ws.addEventListener("open", onOpen);
|
||||
ws.addEventListener("message", onMessage);
|
||||
ws.addEventListener("error", onError);
|
||||
ws.addEventListener("close", onClose);
|
||||
return;
|
||||
}
|
||||
ws.onopen = onOpen;
|
||||
ws.onmessage = onMessage;
|
||||
ws.onerror = onError;
|
||||
ws.onclose = onClose;
|
||||
}
|
||||
|
||||
function parseFrame(data) {
|
||||
try {
|
||||
return JSON.parse(typeof data === "string" ? data : String(data));
|
||||
} catch {
|
||||
return null; // non-JSON frames carry no Live API semantics
|
||||
}
|
||||
}
|
||||
|
||||
function firstStringField(formData, key) {
|
||||
const v = typeof formData?.get === "function" ? formData.get(key) : null;
|
||||
return typeof v === "string" && v.trim() ? v.trim() : "";
|
||||
}
|
||||
|
||||
// Lifecycle knobs the live registry entry advertises in params[]
|
||||
// (setup/turn timeouts). They ride the same formData pass-through sttCore
|
||||
// gives every transport — no sttCore change needed to reach this leaf.
|
||||
function firstNumberField(formData, key, fallback) {
|
||||
const n = Number(firstStringField(formData, key));
|
||||
return Number.isFinite(n) && n > 0 ? Math.min(n, MAX_TIMEOUT_MS) : fallback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Transcribe an audio File via the Gemini Live bidirectional stream.
|
||||
* @returns {Promise<{text: string, chunks: string[]}>} transcript plus the raw
|
||||
* incremental inputTranscription deltas (sttCore shapes verbose_json from them).
|
||||
* @throws {GeminiLiveError} with .status for the sttCore error envelope.
|
||||
*/
|
||||
export async function transcribeGeminiLive({ cfg, file, model, token, formData, mimeType }) {
|
||||
const WS = globalThis.WebSocket;
|
||||
if (!WS) throw new GeminiLiveError("Gemini Live transport needs global WebSocket (Node >= 22)", 502);
|
||||
|
||||
const buf = Buffer.from(await file.arrayBuffer());
|
||||
if (!buf.length) throw new GeminiLiveError("Empty audio file", 400);
|
||||
|
||||
const instruction = firstStringField(formData, "prompt") || "Transcribe the spoken audio verbatim.";
|
||||
const language = firstStringField(formData, "language");
|
||||
const setupTimeoutMs = firstNumberField(formData, "setup_timeout_ms", SETUP_TIMEOUT_MS);
|
||||
const turnTimeoutMs = firstNumberField(formData, "turn_timeout_ms", TURN_TIMEOUT_MS);
|
||||
// system_instruction (registry param) overrides the built-in transcription
|
||||
// directive wholesale; prompt/language only shape the default.
|
||||
const instructionOverride = firstStringField(formData, "system_instruction");
|
||||
const systemText = instructionOverride
|
||||
|| (language ? `${instruction} Language: ${language}.` : instruction);
|
||||
const wsUrl = toLiveWsUrl(cfg.baseUrl, model, token);
|
||||
|
||||
return await new Promise((resolve, reject) => {
|
||||
let text = "";
|
||||
const chunks = []; // raw inputTranscription deltas, shaped by sttCore
|
||||
let settled = false;
|
||||
let timer = null;
|
||||
let goAwayTimer = null;
|
||||
let ws = null;
|
||||
let generation = 0; // socket identity: superseded closes never settle
|
||||
let sentBytes = 0; // audio prefix already handed to the live socket
|
||||
let goAwayReconnects = GOAWAY_RECONNECTS;
|
||||
|
||||
const arm = (ms, message) => {
|
||||
if (timer) clearTimeout(timer);
|
||||
timer = setTimeout(() => fail(new GeminiLiveError(message, 504)), ms);
|
||||
};
|
||||
const shutdown = () => {
|
||||
if (timer) { clearTimeout(timer); timer = null; }
|
||||
if (goAwayTimer) { clearTimeout(goAwayTimer); goAwayTimer = null; }
|
||||
// ws is null until the first open() dials (and stays null when the
|
||||
// constructor throws) — fail() runs shutdown() on that path.
|
||||
if (!ws) return;
|
||||
try {
|
||||
if (ws.readyState === WS.OPEN || ws.readyState === WS.CONNECTING) ws.close(1000);
|
||||
} catch { /* socket already dead — outcome is already settled */ }
|
||||
};
|
||||
const succeed = () => {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
shutdown();
|
||||
resolve({ text, chunks });
|
||||
};
|
||||
const fail = (err) => {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
shutdown();
|
||||
reject(err);
|
||||
};
|
||||
const send = (frame) => {
|
||||
if (ws.readyState !== WS.OPEN) return false;
|
||||
try {
|
||||
ws.send(JSON.stringify(frame));
|
||||
} catch {
|
||||
return false; // socket died mid-send — streamAudioAndPrompt maps this to a 502
|
||||
}
|
||||
return true;
|
||||
};
|
||||
|
||||
// Streams every byte not yet sent, then the flushing text turn. After a
|
||||
// goAway rotation this resumes from sentBytes — no audio re-upload.
|
||||
const streamAudioAndPrompt = () => {
|
||||
for (let off = sentBytes; off < buf.length; off += CHUNK_BYTES) {
|
||||
const mediaChunk = buf.subarray(off, off + CHUNK_BYTES).toString("base64");
|
||||
if (!send({ realtimeInput: { mediaChunks: [{ mimeType, data: mediaChunk }] } })) {
|
||||
fail(new GeminiLiveError("Gemini Live socket closed while streaming audio", 502));
|
||||
return;
|
||||
}
|
||||
sentBytes = Math.min(off + CHUNK_BYTES, buf.length);
|
||||
}
|
||||
// Final user turn: flushes the recognizer and yields turnComplete.
|
||||
send({ clientContent: { turns: [{ parts: [{ text: systemText }] }], turnComplete: true } });
|
||||
};
|
||||
|
||||
// goAway: the server names the instant it will force-close this socket.
|
||||
// Graceful play = rotate BEFORE the deadline: retire the live socket,
|
||||
// dial a fresh one, replay setup, resume audio from sentBytes — text and
|
||||
// chunks survive the hop. Once the advisory budget is spent a later
|
||||
// goAway is left to the close path, which settles on partial transcript.
|
||||
const scheduleGoAwayReconnect = (goAway) => {
|
||||
if (settled || goAwayTimer || goAwayReconnects <= 0) return;
|
||||
const deadline = Date.parse(typeof goAway?.time === "string" ? goAway.time : "");
|
||||
const delay = Number.isFinite(deadline)
|
||||
? Math.max(0, Math.min(deadline - Date.now(), setupTimeoutMs))
|
||||
: 0;
|
||||
goAwayTimer = setTimeout(() => {
|
||||
goAwayTimer = null;
|
||||
if (settled) return;
|
||||
goAwayReconnects--;
|
||||
generation++;
|
||||
try { ws?.close(1000); } catch { /* deadline crossed mid-flight — re-dial anyway */ }
|
||||
open();
|
||||
}, delay);
|
||||
};
|
||||
|
||||
const open = () => {
|
||||
const gen = ++generation;
|
||||
try {
|
||||
ws = new WS(wsUrl);
|
||||
} catch {
|
||||
fail(new GeminiLiveError("Gemini Live websocket connection failed", 502));
|
||||
return;
|
||||
}
|
||||
bindSocket(ws, {
|
||||
onOpen: () => {
|
||||
if (settled || gen !== generation) return;
|
||||
arm(setupTimeoutMs, "Gemini Live timed out waiting for setupComplete");
|
||||
send({
|
||||
setup: {
|
||||
model: `models/${model}`,
|
||||
generationConfig: {
|
||||
responseModalities: ["TEXT"],
|
||||
inputAudioTranscription: {},
|
||||
},
|
||||
systemInstruction: { parts: [{ text: systemText }] },
|
||||
},
|
||||
});
|
||||
},
|
||||
onMessage: (ev) => {
|
||||
if (settled || gen !== generation) return;
|
||||
const frame = parseFrame(ev?.data);
|
||||
if (!frame) return;
|
||||
|
||||
if (frame.error) {
|
||||
const e = frame.error;
|
||||
fail(new GeminiLiveError(`Gemini Live error${e.status ? ` (${e.status})` : ""}: ${e.message || "unknown"}`, 502));
|
||||
return;
|
||||
}
|
||||
if (frame.goAway) {
|
||||
scheduleGoAwayReconnect(frame.goAway);
|
||||
return;
|
||||
}
|
||||
|
||||
const sc = frame.serverContent;
|
||||
if (!sc) return;
|
||||
|
||||
const delta = typeof sc.inputTranscription?.text === "string" ? sc.inputTranscription.text : "";
|
||||
// Trim before testing: a padding-only frame carries no transcript and
|
||||
// must not make an empty run look like a partial success on close.
|
||||
if (delta.trim()) {
|
||||
text += delta;
|
||||
chunks.push(delta);
|
||||
}
|
||||
|
||||
if (sc.setupComplete) {
|
||||
arm(turnTimeoutMs, "Gemini Live transcription timed out");
|
||||
streamAudioAndPrompt();
|
||||
return;
|
||||
}
|
||||
if (sc.turnComplete) succeed();
|
||||
},
|
||||
onError: () => {
|
||||
if (settled || gen !== generation) return;
|
||||
fail(new GeminiLiveError("Gemini Live websocket connection failed", 502));
|
||||
},
|
||||
onClose: (ev) => {
|
||||
if (settled || gen !== generation) return;
|
||||
// Partial transcript beats a hard error on graceful close; silence is one.
|
||||
if (text.trim()) succeed();
|
||||
else fail(new GeminiLiveError(`Gemini Live socket closed before completion${ev?.code ? ` (code ${ev.code})` : ""}`, 502));
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
open();
|
||||
});
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
// Antigravity image adapter - delegates to the executor for correct request
|
||||
// envelope (project, model, requestType, sessionId) and auth headers.
|
||||
import { nowSec } from "./_base.js";
|
||||
import { nowSec, sizeToAspectRatio } from "./_base.js";
|
||||
import { getExecutor } from "../../executors/index.js";
|
||||
|
||||
// Convert image input (data URI or raw base64) to Gemini inlineData part
|
||||
@@ -31,6 +31,19 @@ export default {
|
||||
const executor = getExecutor("antigravity");
|
||||
if (!executor) throw new Error("Antigravity executor not found");
|
||||
|
||||
// Ensure we use an image model for image generation
|
||||
const isImageModel = (m) => /image|imagen|image-generation/i.test(m || "");
|
||||
let targetModel = isImageModel(model) ? model : "gemini-3.1-flash-image";
|
||||
|
||||
// If body.size is provided, resolve aspect ratio and append to model
|
||||
if (body.size && typeof body.size === "string") {
|
||||
const ratio = sizeToAspectRatio(body.size);
|
||||
const suffix = ratio.replace(":", "x");
|
||||
if (!targetModel.includes(suffix)) {
|
||||
targetModel = `${targetModel}-${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
// Build parts: text prompt + optional input image for editing
|
||||
const parts = [{ text: body.prompt }];
|
||||
const imageInput = body.image || (Array.isArray(body.images) && body.images[0]);
|
||||
@@ -44,7 +57,7 @@ export default {
|
||||
};
|
||||
|
||||
const result = await executor.execute({
|
||||
model,
|
||||
model: targetModel,
|
||||
body: chatBody,
|
||||
stream: false,
|
||||
credentials,
|
||||
|
||||
@@ -2,13 +2,21 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { nowSec } from "./_base.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { CODEX_CLI_VERSION } from "../../config/appConstants.js";
|
||||
|
||||
const CODEX_RESPONSES_URL = PROVIDERS["codex"].baseUrl;
|
||||
const CODEX_USER_AGENT = "codex_cli_rs/0.136.0";
|
||||
const CODEX_VERSION = "0.136.0";
|
||||
const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`;
|
||||
const CODEX_ORIGINATOR = "codex_cli_rs";
|
||||
const CODEX_MODEL_SUFFIX = "-image";
|
||||
const CODEX_REF_DETAIL = "high";
|
||||
const CODEX_IMAGES_MAIN_MODEL = "gpt-5.5";
|
||||
const CODEX_TOOL_IMAGE_MODELS = new Set([
|
||||
"gpt-image-1.5",
|
||||
"gpt-image-2",
|
||||
"gpt-image-2.5",
|
||||
"gpt-image-2.5-flare",
|
||||
"gpt-image-2.5-sunburst",
|
||||
]);
|
||||
|
||||
function decodeAccountId(idToken) {
|
||||
try {
|
||||
@@ -27,6 +35,13 @@ function stripImageSuffix(model) {
|
||||
return model.endsWith(CODEX_MODEL_SUFFIX) ? model.slice(0, -CODEX_MODEL_SUFFIX.length) : model;
|
||||
}
|
||||
|
||||
function resolveCodexImageModels(model) {
|
||||
if (CODEX_TOOL_IMAGE_MODELS.has(model)) {
|
||||
return { responsesModel: CODEX_IMAGES_MAIN_MODEL, toolModel: model };
|
||||
}
|
||||
return { responsesModel: stripImageSuffix(model), toolModel: null };
|
||||
}
|
||||
|
||||
function toDataUrl(input) {
|
||||
if (!input || typeof input !== "string") return null;
|
||||
if (/^data:image\//i.test(input) || /^https?:\/\//i.test(input)) return input;
|
||||
@@ -157,7 +172,7 @@ export default {
|
||||
"originator": CODEX_ORIGINATOR,
|
||||
"session_id": randomUUID(),
|
||||
"user-agent": CODEX_USER_AGENT,
|
||||
"version": CODEX_VERSION,
|
||||
"version": CODEX_CLI_VERSION,
|
||||
"x-client-request-id": randomUUID(),
|
||||
};
|
||||
},
|
||||
@@ -167,21 +182,26 @@ export default {
|
||||
const single = toDataUrl(body.image);
|
||||
if (single) refs.push(single);
|
||||
const detail = body.image_detail || CODEX_REF_DETAIL;
|
||||
const { responsesModel, toolModel } = resolveCodexImageModels(model);
|
||||
const imgTool = { type: "image_generation", output_format: (body.output_format || "png").toLowerCase() };
|
||||
if (toolModel) {
|
||||
imgTool.action = refs.length > 0 ? "edit" : "generate";
|
||||
imgTool.model = toolModel;
|
||||
}
|
||||
if (body.size && body.size !== "") imgTool.size = body.size;
|
||||
if (body.quality && body.quality !== "") imgTool.quality = body.quality;
|
||||
if (body.background && body.background !== "") imgTool.background = body.background;
|
||||
return {
|
||||
model: stripImageSuffix(model),
|
||||
model: responsesModel,
|
||||
instructions: "",
|
||||
input: [{ type: "message", role: "user", content: buildContent(body.prompt, refs, detail) }],
|
||||
tools: [imgTool],
|
||||
tool_choice: "auto",
|
||||
tool_choice: toolModel ? { type: "image_generation" } : "auto",
|
||||
parallel_tool_calls: false,
|
||||
prompt_cache_key: randomUUID(),
|
||||
stream: true,
|
||||
store: false,
|
||||
reasoning: null,
|
||||
reasoning: toolModel ? { effort: "medium", summary: "auto" } : null,
|
||||
};
|
||||
},
|
||||
// Custom: codex parses SSE → either pipe to client or collect b64
|
||||
|
||||
@@ -1,18 +1,93 @@
|
||||
// HuggingFace Inference API — returns binary image
|
||||
import { nowSec } from "./_base.js";
|
||||
// HuggingFace Inference Providers router — returns binary image
|
||||
//
|
||||
// The router is a switchboard in front of many inference providers and is
|
||||
// addressed as `<baseUrl>/<provider>/<providerModelId>`. `providerModelId` is
|
||||
// the id the *provider* uses, which is not the Hub model id, so it is resolved
|
||||
// through `imageConfig.modelMap` (built from the Hub API's
|
||||
// inferenceProviderMapping and limited to providers the router forwards to).
|
||||
//
|
||||
// The legacy `api-inference.huggingface.co` host is gone (DNS ENOTFOUND) and is
|
||||
// deliberately not referenced anywhere here.
|
||||
import { nowSec, urlToBase64 } from "./_base.js";
|
||||
import { PROVIDER_MEDIA } from "../../providers/index.js";
|
||||
|
||||
const BASE_URL = PROVIDER_MEDIA["huggingface"]?.imageConfig?.baseUrl;
|
||||
const imageConfig = () => PROVIDER_MEDIA["huggingface"]?.imageConfig || {};
|
||||
const BASE_URL = imageConfig().baseUrl;
|
||||
const MODEL_MAP = imageConfig().modelMap || {};
|
||||
|
||||
// A plain-object lookup returns inherited truthy values for keys like "toString" or
|
||||
// "constructor", which would build nonsense URLs. Resolve own keys only.
|
||||
const lookup = (model) => (Object.hasOwn(MODEL_MAP, model) ? MODEL_MAP[model] : undefined);
|
||||
|
||||
// modelMap values are either a bare path (text-to-image) or { path, task }.
|
||||
const mappingPath = (entry) => (typeof entry === "string" ? entry : entry.path);
|
||||
const mappingTask = (entry) => (typeof entry === "string" ? "text-to-image" : entry.task || "text-to-image");
|
||||
|
||||
// A connection may point at its own endpoint (self-hosted Text Generation
|
||||
// Inference / TGI container). That endpoint already knows its own model ids, so
|
||||
// the router mapping does not apply and the Hub id is passed through verbatim.
|
||||
function customBaseUrl(creds) {
|
||||
const url = creds?.providerSpecificData?.baseUrl;
|
||||
return typeof url === "string" && url.trim() ? url.trim().replace(/\/+$/, "") : null;
|
||||
}
|
||||
|
||||
// The router's image-to-image payload wants raw base64 — not a data URL, not a URL.
|
||||
// Accept every shape our own callers use (data URL, bare base64, remote URL, array).
|
||||
async function sourceImage(body) {
|
||||
const raw = body?.image || (Array.isArray(body?.images) ? body.images[0] : null);
|
||||
if (typeof raw !== "string" || !raw.trim()) return null;
|
||||
const value = raw.trim();
|
||||
if (/^https?:\/\//i.test(value)) return await urlToBase64(value);
|
||||
const match = /^data:image\/[^;]+;base64,(.+)$/i.exec(value);
|
||||
return match ? match[1] : value;
|
||||
}
|
||||
|
||||
export default {
|
||||
buildUrl: (model) => `${BASE_URL}/${model}`,
|
||||
buildUrl: (model, creds) => {
|
||||
const override = customBaseUrl(creds);
|
||||
if (override) {
|
||||
// The model id is client-controlled; on a custom endpoint it lands in a URL
|
||||
// path verbatim, so reject traversal/query injection (mirrors sttCore's guard).
|
||||
if (model.includes("..") || model.includes("//") || /[?#]/.test(model)) {
|
||||
throw new Error(`HuggingFace: invalid model ID "${model}"`);
|
||||
}
|
||||
return `${override}/${model}`;
|
||||
}
|
||||
|
||||
const entry = lookup(model);
|
||||
if (!entry) {
|
||||
throw new Error(
|
||||
`HuggingFace: no HuggingFace router mapping for model "${model}". ` +
|
||||
`Add it to imageConfig.modelMap in open-sse/providers/registry/huggingface.js, ` +
|
||||
`or set a custom base URL on the connection.`
|
||||
);
|
||||
}
|
||||
return `${BASE_URL}/${mappingPath(entry)}`;
|
||||
},
|
||||
buildHeaders: (creds) => {
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
const key = creds?.apiKey || creds?.accessToken;
|
||||
if (key) headers["Authorization"] = `Bearer ${key}`;
|
||||
return headers;
|
||||
},
|
||||
buildBody: (_model, body) => ({ inputs: body.prompt }),
|
||||
buildBody: async (model, body) => {
|
||||
const entry = lookup(model);
|
||||
const task = mappingTask(entry || "");
|
||||
|
||||
if (task === "image-to-image") {
|
||||
const image = await sourceImage(body);
|
||||
if (!image) {
|
||||
throw new Error(
|
||||
`HuggingFace: model "${model}" requires a source image. ` +
|
||||
`Send it as "image" (or "images") in the request body.`
|
||||
);
|
||||
}
|
||||
// inputs carries the source image; the prompt moves under parameters.
|
||||
return { inputs: image, parameters: { prompt: body.prompt } };
|
||||
}
|
||||
|
||||
return { inputs: body.prompt };
|
||||
},
|
||||
// HF returns raw image bytes — convert to b64_json
|
||||
async parseResponse(response) {
|
||||
const buf = await response.arrayBuffer();
|
||||
|
||||
@@ -29,6 +29,8 @@
|
||||
* @property {Record<string,unknown>} [providerSpecificData]
|
||||
*/
|
||||
|
||||
import { assertPublicUrl } from "../../../src/shared/utils/ssrfGuard.js";
|
||||
|
||||
// ── Helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
@@ -63,12 +65,31 @@ export function getProviderSetting(params, key) {
|
||||
|
||||
/**
|
||||
* Resolve base URL with optional override from providerOptions.baseUrl.
|
||||
*
|
||||
* The override is client-controlled and therefore SSRF-hardened: only public
|
||||
* http(s) URLs are accepted (internal/private/loopback/metadata addresses are
|
||||
* rejected via assertPublicUrl). The provider's own configured baseUrl is
|
||||
* trusted as-is (admin-controlled).
|
||||
*
|
||||
* @param {SearchProviderConfig} config
|
||||
* @param {SearchRequestParams} params
|
||||
* @returns {string}
|
||||
*/
|
||||
export function resolveBaseUrl(config, params) {
|
||||
const override = getProviderSetting(params, "baseUrl");
|
||||
if (override) {
|
||||
// SSRF guard: client-supplied base URLs must be public http(s) only.
|
||||
let parsed;
|
||||
try {
|
||||
parsed = new URL(override);
|
||||
} catch {
|
||||
throw new Error(`Invalid baseUrl: ${override}`);
|
||||
}
|
||||
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") {
|
||||
throw new Error(`Invalid baseUrl protocol: ${parsed.protocol}`);
|
||||
}
|
||||
assertPublicUrl(override);
|
||||
}
|
||||
return (override || config.baseUrl).replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
@@ -326,6 +347,81 @@ function buildSearxngRequest(config, params) {
|
||||
};
|
||||
}
|
||||
|
||||
function buildXquikRequest(config, params) {
|
||||
const apiKey = params.token;
|
||||
if (!apiKey) throw new Error("Xquik requires an API key");
|
||||
|
||||
const queryType = getProviderSetting(params, "queryType");
|
||||
if (queryType && !["Latest", "Top"].includes(queryType)) {
|
||||
throw new Error("Xquik queryType must be Latest or Top");
|
||||
}
|
||||
|
||||
const qp = new URLSearchParams({
|
||||
q: params.query,
|
||||
limit: String(params.maxResults),
|
||||
});
|
||||
const cursor = getProviderSetting(params, "cursor");
|
||||
if (cursor) qp.set("cursor", cursor);
|
||||
if (queryType) qp.set("queryType", queryType);
|
||||
if (params.language) qp.set("language", params.language);
|
||||
|
||||
return {
|
||||
url: `${resolveBaseUrl(config, params)}?${qp}`,
|
||||
init: {
|
||||
method: "GET",
|
||||
headers: { Accept: "application/json", "x-api-key": apiKey },
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── Ollama Cloud web_search ──────────────────────────────────────────────
|
||||
// POST https://ollama.com/api/web_search { query, max_results }
|
||||
// Response: { results: [{ title, url, content, published_at? }] }
|
||||
function buildOllamaSearchRequest(config, params) {
|
||||
const body = { query: params.query, max_results: params.maxResults };
|
||||
if (params.country) body.country = params.country;
|
||||
if (params.language) body.language = params.language;
|
||||
return {
|
||||
url: resolveBaseUrl(config, params),
|
||||
init: {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── GLM Coding plan MCP web_search_prime ──────────────────────────────────
|
||||
// POST https://api.z.ai/api/mcp/web_search_prime/mcp
|
||||
// JSON-RPC envelope: { jsonrpc, id, method: "tools/call",
|
||||
// params: { name: "web_search_prime", arguments: { search_query, count } } }
|
||||
// Response: { result: { content: [{ type: "text", text: "<json>" }] } }
|
||||
function buildGlmSearchRequest(config, params) {
|
||||
const body = {
|
||||
jsonrpc: "2.0",
|
||||
id: `9r-${Date.now()}`,
|
||||
method: "tools/call",
|
||||
params: {
|
||||
name: "web_search_prime",
|
||||
arguments: { search_query: params.query, count: params.maxResults },
|
||||
},
|
||||
};
|
||||
return {
|
||||
url: resolveBaseUrl(config, params),
|
||||
init: {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(params.token ? { Authorization: `Bearer ${params.token}` } : {}),
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── Dispatcher ──────────────────────────────────────────────────────────
|
||||
|
||||
const BUILDERS = {
|
||||
@@ -339,6 +435,9 @@ const BUILDERS = {
|
||||
"searchapi": buildSearchApiRequest,
|
||||
"youcom": buildYouComRequest,
|
||||
"searxng": buildSearxngRequest,
|
||||
"xquik": buildXquikRequest,
|
||||
"ollama-search": buildOllamaSearchRequest,
|
||||
"glm": buildGlmSearchRequest,
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
/**
|
||||
* Wrap chat-completions endpoints (with built-in web search) into the unified
|
||||
* /v1/search response format. Supports gemini, openai, xai, kimi, minimax, perplexity.
|
||||
* /v1/search response format. Supports gemini, antigravity, openai, xai, kimi,
|
||||
* minimax, perplexity.
|
||||
*/
|
||||
import { PROVIDER_MEDIA } from "../../providers/index.js";
|
||||
import { ANTIGRAVITY_IDE_USER_AGENT } from "../../providers/shared.js";
|
||||
|
||||
// Default search model + endpoint derive from registry searchViaChat (single source)
|
||||
const searchModel = (id) => PROVIDER_MEDIA[id]?.searchViaChat?.defaultModel;
|
||||
@@ -28,13 +30,37 @@ function toResult(c, index, provider, retrievedAt) {
|
||||
score: null,
|
||||
published_at: null,
|
||||
favicon_url: null,
|
||||
content: null,
|
||||
content: c.content || null,
|
||||
metadata: {},
|
||||
citation: { provider, retrieved_at: retrievedAt, rank: index + 1 },
|
||||
provider_raw: null
|
||||
};
|
||||
}
|
||||
|
||||
// Antigravity search request envelope (mirrors the IDE client)
|
||||
const AG_CLIENT_NAME = "antigravity";
|
||||
const AG_SEARCH_GENERATION_CONFIG = { temperature: 1.0, maxOutputTokens: 8192 };
|
||||
const AG_CONTEXT_BEFORE = 150;
|
||||
const AG_CONTEXT_AFTER = 250;
|
||||
|
||||
/** Widen a grounded segment to its surrounding sentence(s) in the answer text. */
|
||||
function expandSegment(text, segment) {
|
||||
const { startIndex, endIndex } = segment || {};
|
||||
if (!text || !Number.isInteger(startIndex) || !Number.isInteger(endIndex)) return "";
|
||||
const start = Math.max(0, startIndex - AG_CONTEXT_BEFORE);
|
||||
const end = Math.min(text.length, endIndex + AG_CONTEXT_AFTER);
|
||||
let out = text.slice(start, end).trim();
|
||||
// Drop the partial words the window cut off at either edge
|
||||
if (start > 0) out = `...${out.replace(/^\S+/, "")}`;
|
||||
if (end < text.length) out = `${out.replace(/\S+$/, "")}...`;
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
/** Join deduped grounding pieces, skipping empties. */
|
||||
function joinPieces(set, sep) {
|
||||
return [...(set || [])].filter(Boolean).join(sep).trim();
|
||||
}
|
||||
|
||||
/** Coerce a citation that might be a raw URL string or an object. */
|
||||
function normalizeCitation(c) {
|
||||
if (!c) return null;
|
||||
@@ -46,6 +72,8 @@ function normalizeCitation(c) {
|
||||
/**
|
||||
* Provider-specific configuration map. All providers must implement:
|
||||
* { endpoint, defaultModel, buildBody, buildHeaders, extractAnswer }
|
||||
* Optional: requireCredentials(credentials) → error string when a provider needs
|
||||
* more than a token (returns null when satisfied).
|
||||
*/
|
||||
const CHAT_SEARCH_CONFIG = {
|
||||
gemini: {
|
||||
@@ -73,6 +101,71 @@ const CHAT_SEARCH_CONFIG = {
|
||||
}
|
||||
},
|
||||
|
||||
antigravity: {
|
||||
endpoint: () => searchEndpoint("antigravity"),
|
||||
// Upstream 403s on a missing or fabricated project — surface the real cause
|
||||
requireCredentials: (credentials) =>
|
||||
credentials?.projectId ? null : "Antigravity account has no projectId — reconnect the account",
|
||||
buildBody: (query, model, credentials) => ({
|
||||
project: credentials.projectId,
|
||||
model,
|
||||
userAgent: AG_CLIENT_NAME,
|
||||
requestType: "search",
|
||||
request: {
|
||||
contents: [{ role: "user", parts: [{ text: query }] }],
|
||||
tools: [{ googleSearch: {} }],
|
||||
generationConfig: AG_SEARCH_GENERATION_CONFIG
|
||||
}
|
||||
}),
|
||||
buildHeaders: (token) => ({
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${token}`,
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT
|
||||
}),
|
||||
extractAnswer: (data) => {
|
||||
// Antigravity wraps the Gemini payload in { response: {...} }
|
||||
const response = data?.response || data;
|
||||
const candidate = response?.candidates?.[0];
|
||||
const parts = candidate?.content?.parts || [];
|
||||
const text = parts.map((p) => p?.text || "").filter(Boolean).join("");
|
||||
const grounding = candidate?.groundingMetadata || {};
|
||||
const chunks = grounding.groundingChunks || [];
|
||||
const supports = grounding.groundingSupports || [];
|
||||
|
||||
// Upstream repeats the same source across chunks — key by URL so it stays one citation.
|
||||
// Map, not a plain object: both the index and the URL come from upstream.
|
||||
const sources = new Map();
|
||||
const byIndex = chunks.map((ch) => {
|
||||
const web = ch?.web;
|
||||
const url = web?.uri || web?.url || "";
|
||||
if (!url) return null;
|
||||
if (!sources.has(url)) sources.set(url, { title: web.title || "", snippets: new Set(), contexts: new Set() });
|
||||
return sources.get(url);
|
||||
});
|
||||
|
||||
// Each support ties a sentence of the answer back to the chunks that grounded it
|
||||
for (const s of supports) {
|
||||
const segment = s?.segment;
|
||||
const grounded = segment?.text || "";
|
||||
const expanded = expandSegment(text, segment) || grounded;
|
||||
for (const idx of s?.groundingChunkIndices || []) {
|
||||
const source = Number.isInteger(idx) ? byIndex[idx] : null;
|
||||
if (!source) continue;
|
||||
if (grounded) source.snippets.add(grounded);
|
||||
if (expanded) source.contexts.add(expanded);
|
||||
}
|
||||
}
|
||||
|
||||
const citations = [...sources].map(([url, src]) => {
|
||||
const snippet = joinPieces(src.snippets, " | ") || src.title;
|
||||
return { url, title: src.title, snippet, content: joinPieces(src.contexts, "\n\n") || snippet };
|
||||
});
|
||||
|
||||
const tokens = response?.usageMetadata?.totalTokenCount || 0;
|
||||
return { text, citations, tokens };
|
||||
}
|
||||
},
|
||||
|
||||
openai: {
|
||||
endpoint: () => searchEndpoint("openai"),
|
||||
buildBody: (query, model) => {
|
||||
@@ -366,13 +459,18 @@ export async function handleChatSearch({
|
||||
};
|
||||
}
|
||||
|
||||
const credentialError = cfg.requireCredentials?.(credentials);
|
||||
if (credentialError) {
|
||||
return { success: false, status: 401, error: credentialError };
|
||||
}
|
||||
|
||||
const limit =
|
||||
Number.isFinite(maxResults) && maxResults > 0
|
||||
? Math.floor(maxResults)
|
||||
: DEFAULT_MAX_RESULTS;
|
||||
const useModel = model || searchModel(provider);
|
||||
const url = cfg.endpoint(useModel);
|
||||
const body = cfg.buildBody(query, useModel);
|
||||
const body = cfg.buildBody(query, useModel, credentials);
|
||||
const headers = cfg.buildHeaders(token);
|
||||
|
||||
const controller = new AbortController();
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
import { buildSearchRequest } from "./callers.js";
|
||||
import { normalizeSearchResponse } from "./normalizers.js";
|
||||
import { handleChatSearch } from "./chatSearch.js";
|
||||
import { fetchPublic } from "../../../src/shared/utils/ssrfGuard.js";
|
||||
|
||||
const GLOBAL_TIMEOUT_MS = 15000;
|
||||
const NON_RETRIABLE = new Set([400, 401, 403, 404]);
|
||||
@@ -100,7 +101,7 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
|
||||
log?.info?.("SEARCH", `${provider.id} | "${params.query.slice(0, 80)}" | type=${params.searchType}`);
|
||||
|
||||
try {
|
||||
const resp = await fetch(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
|
||||
const resp = await fetchPublic(url, { ...init, headers: sanitizeHeaders(init.headers), signal: controller.signal });
|
||||
clearTimeout(timer);
|
||||
if (!resp.ok) {
|
||||
const errText = await resp.text().catch(() => "");
|
||||
@@ -111,6 +112,13 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
|
||||
const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType);
|
||||
const results = normalized.results.slice(0, params.maxResults);
|
||||
const duration = Date.now() - startTime;
|
||||
const usage = {
|
||||
queries_used: 1,
|
||||
search_cost_usd: providerConfig.costPerQuery ?? null,
|
||||
};
|
||||
if (Number.isFinite(providerConfig.creditsPerResult)) {
|
||||
usage.provider_credits_used = results.length * providerConfig.creditsPerResult;
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
@@ -119,7 +127,8 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
|
||||
query: params.query,
|
||||
results,
|
||||
answer: null,
|
||||
usage: { queries_used: 1, search_cost_usd: providerConfig.costPerQuery || 0 },
|
||||
usage,
|
||||
...(normalized.pagination ? { pagination: normalized.pagination } : {}),
|
||||
metrics: { response_time_ms: duration, upstream_latency_ms: duration, total_results_available: normalized.totalResults },
|
||||
errors: []
|
||||
}
|
||||
|
||||
@@ -199,6 +199,89 @@ function normalizeSearxng(data, _query, _searchType) {
|
||||
return { results, totalResults: results.length };
|
||||
}
|
||||
|
||||
function normalizeXquik(data, _query, _searchType) {
|
||||
const now = new Date().toISOString();
|
||||
const items = Array.isArray(data.tweets) ? data.tweets : [];
|
||||
const results = items.map((item, idx) => {
|
||||
const username = typeof item?.author?.username === "string" ? item.author.username : "";
|
||||
const authorName = typeof item?.author?.name === "string" ? item.author.name : "";
|
||||
const tweetId = typeof item?.id === "string" ? item.id : String(item?.id || "");
|
||||
const url = username && tweetId
|
||||
? `https://x.com/${encodeURIComponent(username)}/status/${encodeURIComponent(tweetId)}`
|
||||
: tweetId
|
||||
? `https://x.com/i/web/status/${encodeURIComponent(tweetId)}`
|
||||
: "";
|
||||
const author = username ? `@${username}` : authorName || null;
|
||||
const title = author ? `${author} on X` : "X post";
|
||||
const imageUrl = Array.isArray(item?.media)
|
||||
? item.media.find((media) => typeof media?.mediaUrl === "string")?.mediaUrl
|
||||
: null;
|
||||
|
||||
return makeResult("xquik", {
|
||||
title,
|
||||
url,
|
||||
snippet: typeof item?.text === "string" ? item.text : "",
|
||||
published_at: typeof item?.createdAt === "string" ? item.createdAt : null,
|
||||
author,
|
||||
image_url: imageUrl || null,
|
||||
source_type: "x_post",
|
||||
full_text: typeof item?.text === "string" ? item.text : undefined,
|
||||
text_format: "text",
|
||||
}, idx, now);
|
||||
});
|
||||
const nextCursor = typeof data.next_cursor === "string" && data.next_cursor ? data.next_cursor : null;
|
||||
return {
|
||||
results,
|
||||
totalResults: null,
|
||||
pagination: {
|
||||
has_more: data.has_next_page === true,
|
||||
next_cursor: nextCursor,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function normalizeOllamaSearch(data, _query, _searchType) {
|
||||
const now = new Date().toISOString();
|
||||
const items = Array.isArray(data?.results) ? data.results : (Array.isArray(data) ? data : []);
|
||||
const results = items.map((item, idx) =>
|
||||
makeResult("ollama-search", {
|
||||
title: item.title,
|
||||
url: item.url,
|
||||
snippet: item.content || item.snippet || "",
|
||||
full_text: item.content,
|
||||
text_format: "text",
|
||||
published_at: item.published_at || null,
|
||||
source_type: item.source || null,
|
||||
}, idx, now)
|
||||
);
|
||||
return { results, totalResults: results.length };
|
||||
}
|
||||
|
||||
function normalizeGlmSearch(data, _query, _searchType) {
|
||||
const now = new Date().toISOString();
|
||||
// MCP envelope: { result: { content: [{ type: "text", text: "<json>" }] } }
|
||||
let payload = data;
|
||||
const textContent = data?.result?.content?.[0]?.text;
|
||||
if (typeof textContent === "string") {
|
||||
try { payload = JSON.parse(textContent); } catch { payload = {}; }
|
||||
}
|
||||
const items = Array.isArray(payload?.results) ? payload.results
|
||||
: Array.isArray(payload?.news) ? payload.news
|
||||
: Array.isArray(payload) ? payload
|
||||
: [];
|
||||
const results = items.map((item, idx) =>
|
||||
makeResult("glm", {
|
||||
title: item.title,
|
||||
url: item.link || item.url,
|
||||
snippet: item.content || "",
|
||||
published_at: item.publish_date || item.published_at || null,
|
||||
favicon_url: item.icon || null,
|
||||
source_type: item.media || null,
|
||||
}, idx, now)
|
||||
);
|
||||
return { results, totalResults: results.length };
|
||||
}
|
||||
|
||||
const NORMALIZERS = {
|
||||
"serper": normalizeSerper,
|
||||
"brave-search": normalizeBrave,
|
||||
@@ -210,11 +293,14 @@ const NORMALIZERS = {
|
||||
"searchapi": normalizeSearchApi,
|
||||
"youcom": normalizeYouCom,
|
||||
"searxng": normalizeSearxng,
|
||||
"xquik": normalizeXquik,
|
||||
"ollama-search": normalizeOllamaSearch,
|
||||
"glm": normalizeGlmSearch,
|
||||
};
|
||||
|
||||
/**
|
||||
* Dispatch to the appropriate normalizer based on providerId.
|
||||
* @returns {{results: Array, totalResults: number|null}}
|
||||
* @returns {{results: Array, totalResults: number|null, pagination?: object}}
|
||||
*/
|
||||
export function normalizeSearchResponse(providerId, data, query, searchType) {
|
||||
const fn = NORMALIZERS[providerId];
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import { Buffer } from "node:buffer";
|
||||
import { createErrorResult } from "../utils/error.js";
|
||||
import { transcribeGeminiLive } from "./geminiLiveStt.js";
|
||||
import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
|
||||
// Build auth headers from sttConfig + token
|
||||
@@ -162,24 +164,68 @@ function jsonResponse(obj) {
|
||||
};
|
||||
}
|
||||
|
||||
// Model-level transport marker (registry models[].transport, e.g. the Gemini
|
||||
// live STT entry's "gemini-live", or a custom model's stored transport).
|
||||
// Dispatch reads the marker — never a hardcoded model id — so new realtime
|
||||
// providers extend sttCore through data, not code.
|
||||
function resolveModelTransport(provider, model) {
|
||||
const key = PROVIDER_ID_TO_ALIAS[provider] || provider;
|
||||
const models = PROVIDER_MODELS[key] || PROVIDER_MODELS[provider];
|
||||
if (!Array.isArray(models)) return null;
|
||||
const entry = models.find((m) => m && m.id === model && (m.kind || "llm") === "stt");
|
||||
const marker = typeof entry?.transport === "string" ? entry.transport.trim() : "";
|
||||
return marker || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* STT core handler — dispatch by sttConfig.format.
|
||||
* STT core handler — dispatch by model transport marker, else sttConfig.format.
|
||||
* `transport` is the caller-supplied marker override (custom models resolve
|
||||
* it in the app layer; built-ins fall back to the registry entry marker).
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleSttCore({ provider, model, formData, credentials, sttConfig }) {
|
||||
export async function handleSttCore({ provider, model, formData, credentials, sttConfig, transport }) {
|
||||
const file = formData.get("file");
|
||||
if (!file) return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: file");
|
||||
|
||||
const cfg = sttConfig;
|
||||
let cfg = sttConfig;
|
||||
if (!cfg) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support STT`);
|
||||
|
||||
// Per-connection endpoint override. Registry entries carry a fixed baseUrl,
|
||||
// which is right for a named cloud service but useless for a self-hosted one
|
||||
// whose address only the operator knows. Opt-in: absent unless the connection
|
||||
// sets it, so cloud providers are untouched. Mirrors the custom embedding
|
||||
// providers, which already resolve baseUrl the same way.
|
||||
const overrideUrl = credentials?.providerSpecificData?.baseUrl;
|
||||
if (overrideUrl) cfg = { ...cfg, baseUrl: String(overrideUrl).replace(/\/+$/, "") };
|
||||
|
||||
const token = cfg.authType === "none" ? null : (credentials?.apiKey || credentials?.accessToken);
|
||||
if (cfg.authType !== "none" && !token) {
|
||||
return createErrorResult(HTTP_STATUS.UNAUTHORIZED, `No credentials for STT provider: ${provider}`);
|
||||
}
|
||||
|
||||
// Format-switch extension: an explicit caller marker wins over the registry
|
||||
// marker; with neither, the provider-default sttConfig.format applies.
|
||||
const marker = (typeof transport === "string" && transport.trim()) ? transport.trim() : resolveModelTransport(provider, model);
|
||||
|
||||
try {
|
||||
switch (cfg.format) {
|
||||
switch (marker || cfg.format) {
|
||||
case "gemini-live": {
|
||||
const live = await transcribeGeminiLive({ cfg, file, model, token, formData, mimeType: resolveAudioContentType(file) });
|
||||
// response_format parity with the OpenAI-compatible transport: default
|
||||
// envelope stays {text}; verbose_json adds segments mapped from the
|
||||
// Live API's incremental inputTranscription deltas. Those frames carry
|
||||
// NO timestamps, so segments expose {id,text} only (id = delta order,
|
||||
// Whisper-compatible 0-based) — start/end/duration are deliberately
|
||||
// absent rather than fabricated as zeros, which would misrepresent
|
||||
// provider data to callers diffing transports.
|
||||
const fmt = typeof formData?.get === "function"
|
||||
? String(formData.get("response_format") ?? "").trim().toLowerCase()
|
||||
: "";
|
||||
if (fmt === "verbose_json") {
|
||||
return jsonResponse({ text: live.text, segments: live.chunks.map((segText, id) => ({ id, text: segText })) });
|
||||
}
|
||||
return jsonResponse({ text: live.text });
|
||||
}
|
||||
case "deepgram": return await transcribeDeepgram(cfg, file, model, token, formData);
|
||||
case "assemblyai": return await transcribeAssemblyAI(cfg, file, model, token);
|
||||
case "nvidia-asr": return await transcribeNvidia(cfg, file, model, token);
|
||||
@@ -188,6 +234,6 @@ export async function handleSttCore({ provider, model, formData, credentials, st
|
||||
default: return await transcribeOpenAICompatible(cfg, file, model, token, formData);
|
||||
}
|
||||
} catch (err) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed");
|
||||
return createErrorResult(err.status || HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed");
|
||||
}
|
||||
}
|
||||
|
||||
95
open-sse/handlers/systemoneCore.js
Normal file
95
open-sse/handlers/systemoneCore.js
Normal file
@@ -0,0 +1,95 @@
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { HTTP_STATUS, FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { PROVIDER_MEDIA } from "../providers/index.js";
|
||||
import { generateSessionId } from "../executors/opencode-zen.js";
|
||||
|
||||
/**
|
||||
* Core System One (Jev) handler — native decision payload pass-through.
|
||||
* URL/headers come from the registry's systemoneConfig; body and JSON response
|
||||
* are forwarded untouched (decision models have no chat translation layer).
|
||||
*
|
||||
* @returns {Promise<{ success: boolean, response: Response, usage?: object, status?: number, error?: string }>}
|
||||
*/
|
||||
export async function handleSystemoneCore({
|
||||
body,
|
||||
modelInfo,
|
||||
credentials,
|
||||
log,
|
||||
onRequestSuccess,
|
||||
}) {
|
||||
const { provider, model } = modelInfo;
|
||||
const cfg = PROVIDER_MEDIA[provider]?.systemoneConfig;
|
||||
if (!cfg?.baseUrl) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_REQUEST,
|
||||
`Provider '${provider}' does not support System One.`
|
||||
);
|
||||
}
|
||||
|
||||
// Validate input at the trust boundary; question-level shape is upstream's job.
|
||||
if (body.state === undefined || body.state === null) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: state");
|
||||
}
|
||||
if (!body.questions || typeof body.questions !== "object" || Array.isArray(body.questions)) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: questions");
|
||||
}
|
||||
|
||||
// noAuth free lanes carry accessToken "public" from the credential stub.
|
||||
const token = credentials?.apiKey || credentials?.accessToken;
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
...(token ? { Authorization: `Bearer ${token}` } : {}),
|
||||
...(cfg.headers || {}),
|
||||
// Zen lanes expect the official client session header on every request.
|
||||
"x-opencode-session": generateSessionId(),
|
||||
};
|
||||
const requestBody = { ...body, model };
|
||||
|
||||
log?.debug?.("SYSTEMONE", `${provider.toUpperCase()} | ${model}`);
|
||||
|
||||
let providerResponse;
|
||||
try {
|
||||
providerResponse = await fetch(cfg.baseUrl, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify(requestBody),
|
||||
...(typeof AbortSignal?.timeout === "function"
|
||||
? { signal: AbortSignal.timeout(FETCH_CONNECT_TIMEOUT_MS) }
|
||||
: {}),
|
||||
});
|
||||
} catch (error) {
|
||||
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
|
||||
log?.debug?.("SYSTEMONE", `Fetch error: ${errMsg}`);
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, errMsg);
|
||||
}
|
||||
|
||||
if (!providerResponse.ok) {
|
||||
const { statusCode, message } = await parseUpstreamError(providerResponse);
|
||||
const errMsg = formatProviderError(new Error(message), provider, model, statusCode);
|
||||
log?.debug?.("SYSTEMONE", `Provider error: ${errMsg}`);
|
||||
return createErrorResult(statusCode, errMsg);
|
||||
}
|
||||
|
||||
let responseBody;
|
||||
try {
|
||||
responseBody = await providerResponse.json();
|
||||
} catch {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Invalid JSON response from ${provider}`);
|
||||
}
|
||||
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
|
||||
const usage = responseBody?.usage;
|
||||
return {
|
||||
success: true,
|
||||
usage: usage
|
||||
? { prompt_tokens: usage.input_tokens || 0, completion_tokens: usage.output_tokens || 0 }
|
||||
: null,
|
||||
response: new Response(JSON.stringify(responseBody), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
@@ -48,16 +48,16 @@ function createTtsResponse(base64Audio, format, responseFormat) {
|
||||
*
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language }) {
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language, style }) {
|
||||
if (!input?.trim()) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: input");
|
||||
}
|
||||
|
||||
try {
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini)
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini, xiaomi-mimo)
|
||||
const adapter = getTtsAdapter(provider);
|
||||
if (adapter) {
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language });
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language, style });
|
||||
// Adapter may return a full {success, response} (legacy) or {base64, format}
|
||||
if (result.success !== undefined) return result;
|
||||
return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
|
||||
@@ -51,6 +51,25 @@ async function huggingface({ baseUrl, apiKey, text, modelId }) {
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Fish Audio: model travels in an HTTP header, the voice is a reference_id, returns binary
|
||||
async function fishAudio({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
"model": modelId || "s2.1-pro-free",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
format: "mp3",
|
||||
...(voiceId ? { reference_id: voiceId } : {}),
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Inworld: Basic auth, JSON { audioContent }
|
||||
async function inworld({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
@@ -166,4 +185,5 @@ export const FORMAT_HANDLERS = {
|
||||
tortoise,
|
||||
openai: openaiCompat,
|
||||
"minimax-tts": minimaxTts,
|
||||
"fish-audio": fishAudio,
|
||||
};
|
||||
|
||||
@@ -6,6 +6,8 @@ import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
||||
import openai from "./openai.js";
|
||||
import openrouter from "./openrouter.js";
|
||||
import gemini, { fetchGeminiVoices } from "./gemini.js";
|
||||
import xiaomiMimo from "./xiaomi-mimo.js";
|
||||
import selfhostedTts from "./selfhostedTts.js";
|
||||
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
@@ -18,6 +20,8 @@ const SPECIAL_ADAPTERS = {
|
||||
openai,
|
||||
openrouter,
|
||||
gemini,
|
||||
"xiaomi-mimo": xiaomiMimo,
|
||||
"selfhosted-tts": selfhostedTts,
|
||||
};
|
||||
|
||||
export function getTtsAdapter(provider) {
|
||||
|
||||
69
open-sse/handlers/ttsProviders/selfhostedTts.js
Normal file
69
open-sse/handlers/ttsProviders/selfhostedTts.js
Normal file
@@ -0,0 +1,69 @@
|
||||
// Self-hosted OpenAI-compatible TTS — POST {baseUrl}/v1/audio/speech.
|
||||
//
|
||||
// A SPECIAL_ADAPTER rather than a genericFormats handler on purpose: the generic
|
||||
// dispatcher resolves baseUrl from the static registry entry
|
||||
// (`synthesizeViaConfig` reads `cfg.baseUrl`) and never looks at the connection,
|
||||
// which is exactly the limitation this provider exists to lift.
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
const DEFAULT_BASE_URL = "http://localhost:8880";
|
||||
const DEFAULT_MODEL = "kokoro";
|
||||
const DEFAULT_VOICE = "af_heart";
|
||||
|
||||
export default {
|
||||
async synthesize(text, model, credentials, responseFormat = "mp3") {
|
||||
// Accept either providerSpecificData.baseUrl (how the custom embedding and
|
||||
// STT providers carry it) or a bare credentials.baseUrl (how the OpenAI TTS
|
||||
// adapter does), so a connection configured either way works.
|
||||
const raw = credentials?.providerSpecificData?.baseUrl || credentials?.baseUrl || DEFAULT_BASE_URL;
|
||||
// Tolerate a baseUrl given as the full endpoint or with a trailing /v1 —
|
||||
// both are natural things to paste, and silently double-appending the path
|
||||
// would 404 with nothing pointing at the cause.
|
||||
const base = String(raw)
|
||||
.replace(/\/+$/, "")
|
||||
.replace(/\/v1\/audio\/speech$/, "")
|
||||
.replace(/\/v1$/, "");
|
||||
|
||||
// The provider prefix is already stripped by getModelInfo, so `model` here is
|
||||
// "kokoro" or "kokoro/af_heart" — NOT "selfhosted-tts/...".
|
||||
//
|
||||
// A bare value is the MODEL, not the voice. The OpenAI adapter reads a bare
|
||||
// value as a voice, which is right for a service whose model is fixed
|
||||
// ("tts-1") and whose voice varies — but wrong here, where the model is the
|
||||
// variable part. Treating it as a voice sent voice="kokoro" upstream and
|
||||
// Kokoro answered 400, so `selfhosted-tts/kokoro` — the obvious way to
|
||||
// address this provider — was the one form that did not work (verified
|
||||
// against a live Kokoro through 9router, 2026-08-03).
|
||||
let ttsModel = DEFAULT_MODEL;
|
||||
let voice = DEFAULT_VOICE;
|
||||
if (model) {
|
||||
const parts = String(model).split("/").filter(Boolean);
|
||||
if (parts.length >= 2) {
|
||||
ttsModel = parts[0];
|
||||
voice = parts.slice(1).join("/");
|
||||
} else if (parts.length === 1) {
|
||||
ttsModel = parts[0];
|
||||
}
|
||||
}
|
||||
|
||||
const res = await fetch(`${base}/v1/audio/speech`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(credentials?.apiKey ? { Authorization: `Bearer ${credentials.apiKey}` } : {}),
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: ttsModel,
|
||||
voice,
|
||||
input: text,
|
||||
response_format: responseFormat,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.error?.message || `Self-hosted TTS failed: ${res.status}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: responseFormat };
|
||||
},
|
||||
};
|
||||
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
@@ -0,0 +1,65 @@
|
||||
// Xiaomi MiMo TTS — via OpenAI-compatible chat completions (non-streaming).
|
||||
// Docs: https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5
|
||||
// Message contract: target text in `role: assistant` content, style/voice
|
||||
// instructions in `role: user` content. Voice is selected via the top-level
|
||||
// `audio.voice` field (NOT embedded in the model name).
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
const DEFAULT_MODEL = "mimo-v2.5-tts";
|
||||
const DEFAULT_VOICE = "mimo_default";
|
||||
|
||||
export default {
|
||||
synthesize(text, model, credentials, responseFormat, { style, language } = {}) {
|
||||
if (!credentials?.apiKey) throw new Error("xiaomi-mimo API key required");
|
||||
return synthesizeMiMo(text, model, credentials.apiKey, style, language);
|
||||
},
|
||||
};
|
||||
|
||||
export async function synthesizeMiMo(text, model, apiKey, style, language) {
|
||||
const { modelId, voiceId } = parseModelVoice(model, DEFAULT_MODEL, DEFAULT_VOICE, [DEFAULT_MODEL]);
|
||||
|
||||
// Language and style are soft instructions → prepend as a role:user message.
|
||||
// MiMo auto-detects the spoken language of the text; the hint only nudges it
|
||||
// (e.g. "Speak in English.") and is independent of the chosen voice.
|
||||
const instructions = [];
|
||||
if (language) instructions.push(`Speak in ${language}.`);
|
||||
if (style) instructions.push(style);
|
||||
|
||||
const messages = [{ role: "assistant", content: text }];
|
||||
if (instructions.length) messages.unshift({ role: "user", content: instructions.join(" ") });
|
||||
|
||||
const res = await fetch("https://api.xiaomimimo.com/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
stream: false,
|
||||
messages,
|
||||
audio: {
|
||||
format: "wav",
|
||||
voice: voiceId || DEFAULT_VOICE,
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
const rawText = await res.text();
|
||||
let data = {};
|
||||
if (rawText) {
|
||||
try { data = JSON.parse(rawText); } catch { data = {}; }
|
||||
}
|
||||
|
||||
if (!res.ok) {
|
||||
throw new Error(data?.error?.message || rawText || `MiMo TTS error (${res.status})`);
|
||||
}
|
||||
|
||||
const audio = data?.choices?.[0]?.message?.audio?.data;
|
||||
if (!audio) throw new Error(data?.error?.message || "MiMo TTS returned no audio");
|
||||
|
||||
return {
|
||||
base64: audio,
|
||||
format: data?.choices?.[0]?.message?.audio?.format || "wav",
|
||||
};
|
||||
}
|
||||
@@ -2,6 +2,7 @@ import { createErrorResult } from "../utils/error.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { refreshTokenByProvider } from "../services/tokenRefresh.js";
|
||||
import { PROVIDER_MEDIA } from "../providers/index.js";
|
||||
import { getVideoAdapter } from "./videoProviders/index.js";
|
||||
|
||||
// Upstream fetch deadline for video job submission/polling (the job itself is
|
||||
// async upstream — this only bounds the HTTP round-trip, not video rendering).
|
||||
@@ -94,21 +95,49 @@ export async function handleVideoProxyCore({
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`);
|
||||
}
|
||||
|
||||
const method = requestId ? "GET" : "POST";
|
||||
const url = buildUpstreamUrl(config, action, requestId);
|
||||
const adapter = getVideoAdapter(provider);
|
||||
const fetchSignal = combineSignals(signal, timeoutMs);
|
||||
|
||||
const doFetch = (token) =>
|
||||
fetch(url, {
|
||||
// Default (xAI shape) request plan; adapters override URL/method/headers/body.
|
||||
const defaultPlan = () => {
|
||||
const method = requestId ? "GET" : "POST";
|
||||
return {
|
||||
method,
|
||||
headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }),
|
||||
url: buildUpstreamUrl(config, action, requestId),
|
||||
headers: buildHeaders({
|
||||
token: credentials?.accessToken || credentials?.apiKey,
|
||||
contentType: method === "POST" ? contentType : null,
|
||||
idempotencyKey: method === "POST" ? idempotencyKey : null,
|
||||
}),
|
||||
body: method === "POST" ? rawBody : undefined,
|
||||
signal: fetchSignal,
|
||||
});
|
||||
};
|
||||
};
|
||||
|
||||
// Rebuilt per attempt so the auth retry below picks up the refreshed token.
|
||||
const doFetch = async () => {
|
||||
const plan = adapter
|
||||
? await adapter.buildRequest({
|
||||
config, action, requestId, rawBody, contentType, idempotencyKey, credentials, log,
|
||||
token: credentials?.accessToken || credentials?.apiKey,
|
||||
})
|
||||
: defaultPlan();
|
||||
if (plan.error) return { planError: plan.error };
|
||||
return {
|
||||
response: await fetch(plan.url, {
|
||||
method: plan.method,
|
||||
headers: plan.headers,
|
||||
body: plan.body,
|
||||
signal: fetchSignal,
|
||||
}),
|
||||
};
|
||||
};
|
||||
|
||||
const method = requestId ? "GET" : "POST";
|
||||
let upstream;
|
||||
try {
|
||||
upstream = await doFetch(credentials?.accessToken || credentials?.apiKey);
|
||||
const first = await doFetch();
|
||||
if (first.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${first.planError}`);
|
||||
upstream = first.response;
|
||||
} catch (error) {
|
||||
if (error?.name === "AbortError" || error?.name === "TimeoutError") {
|
||||
return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`);
|
||||
@@ -136,7 +165,9 @@ export async function handleVideoProxyCore({
|
||||
await upstream.body?.cancel?.();
|
||||
} catch { /* noop */ }
|
||||
try {
|
||||
upstream = await doFetch(credentials.accessToken || credentials.apiKey);
|
||||
const retry = await doFetch();
|
||||
if (retry.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${retry.planError}`);
|
||||
upstream = retry.response;
|
||||
} catch (error) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials));
|
||||
}
|
||||
@@ -152,13 +183,25 @@ export async function handleVideoProxyCore({
|
||||
return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`);
|
||||
}
|
||||
|
||||
// Success: pass the upstream JSON through untouched (request_id / status / video.url).
|
||||
// Success: pass the upstream JSON through untouched (request_id / status / video.url),
|
||||
// unless the adapter maps a provider-native shape onto it (Vertex operations).
|
||||
let outBody = bodyText;
|
||||
let outType = upstream.headers.get("content-type") || "application/json";
|
||||
if (adapter?.transformResponse) {
|
||||
try {
|
||||
outBody = JSON.stringify(adapter.transformResponse(JSON.parse(bodyText)));
|
||||
outType = "application/json";
|
||||
} catch {
|
||||
// Non-JSON or unexpected shape — fall back to the raw upstream body.
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(bodyText, {
|
||||
response: new Response(outBody, {
|
||||
status: upstream.status,
|
||||
headers: {
|
||||
"Content-Type": upstream.headers.get("content-type") || "application/json",
|
||||
"Content-Type": outType,
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
|
||||
13
open-sse/handlers/videoProviders/index.js
Normal file
13
open-sse/handlers/videoProviders/index.js
Normal file
@@ -0,0 +1,13 @@
|
||||
// Video provider adapters.
|
||||
//
|
||||
// Default (no adapter) = xAI shape: raw body forwarded to {baseUrl}/{action},
|
||||
// polled at {baseUrl}/{id}, upstream JSON passed through verbatim.
|
||||
// A provider only needs an adapter when its wire format differs from that.
|
||||
import openrouter from "./openrouter.js";
|
||||
import vertex from "./vertex.js";
|
||||
|
||||
const ADAPTERS = { openrouter, vertex };
|
||||
|
||||
export function getVideoAdapter(provider) {
|
||||
return ADAPTERS[provider] || null;
|
||||
}
|
||||
39
open-sse/handlers/videoProviders/openrouter.js
Normal file
39
open-sse/handlers/videoProviders/openrouter.js
Normal file
@@ -0,0 +1,39 @@
|
||||
// OpenRouter video jobs — https://openrouter.ai/docs/api/api-reference/videos
|
||||
//
|
||||
// Same async shape as xAI (POST → { id, status }, GET → status/unsigned_urls),
|
||||
// two differences only: creation POSTs to the collection root (no `/generations`
|
||||
// suffix) and the account headers come from the registry entry.
|
||||
// Response bodies are passed through verbatim.
|
||||
|
||||
// ponytail: generations only — OpenRouter has no edits/extensions endpoint today.
|
||||
const SUPPORTED_ACTIONS = new Set(["generations"]);
|
||||
|
||||
function headers(config, token) {
|
||||
return {
|
||||
Accept: "application/json",
|
||||
...(config.headers || {}),
|
||||
...(token ? { Authorization: `Bearer ${token}` } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
export default {
|
||||
buildRequest({ config, action, requestId, rawBody, contentType, token }) {
|
||||
const base = config.baseUrl.replace(/\/$/, "");
|
||||
|
||||
if (requestId) {
|
||||
return { method: "GET", url: `${base}/${encodeURIComponent(requestId)}`, headers: headers(config, token) };
|
||||
}
|
||||
if (!SUPPORTED_ACTIONS.has(action)) {
|
||||
return { error: `OpenRouter video supports 'generations' only (got '${action}')` };
|
||||
}
|
||||
if (contentType && !contentType.includes("application/json")) {
|
||||
return { error: "OpenRouter video requires an application/json body" };
|
||||
}
|
||||
return {
|
||||
method: "POST",
|
||||
url: base,
|
||||
headers: { ...headers(config, token), "Content-Type": "application/json" },
|
||||
body: rawBody,
|
||||
};
|
||||
},
|
||||
};
|
||||
159
open-sse/handlers/videoProviders/vertex.js
Normal file
159
open-sse/handlers/videoProviders/vertex.js
Normal file
@@ -0,0 +1,159 @@
|
||||
// Vertex AI (Veo) video jobs.
|
||||
//
|
||||
// Vertex does NOT speak the OpenAI-ish /v1/videos shape, so unlike OpenRouter
|
||||
// this adapter translates both directions:
|
||||
// create → POST {model}:predictLongRunning { instances[], parameters{} } → { name }
|
||||
// poll → POST {model}:fetchPredictOperation { operationName } → { done, response }
|
||||
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
|
||||
//
|
||||
// The operation name is a resource path (contains "/"), so it is base64url-encoded
|
||||
// into the job id returned to the client — GET /v1/videos/{id} stays a flat path.
|
||||
import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefresh.js";
|
||||
|
||||
const DEFAULT_LOCATION = "us-central1";
|
||||
|
||||
const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url");
|
||||
|
||||
// Operation name shape: projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}.
|
||||
// Anchored and single-segment-per-field so a decoded path can never carry `..` or a
|
||||
// host-changing prefix into the request URL.
|
||||
const OPERATION_NAME_RE = /^projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/models\/[^/]+\/operations\/[^/]+$/;
|
||||
|
||||
function modelPathOf(operationName) {
|
||||
return operationName.slice(0, operationName.indexOf("/operations/"));
|
||||
}
|
||||
|
||||
function decodeJobId(id) {
|
||||
const raw = String(id ?? "");
|
||||
// Buffer.from(x, "base64url") silently drops invalid characters instead of
|
||||
// throwing, so only ids that re-encode byte-for-byte are accepted.
|
||||
if (!raw || raw.length > 1024 || !/^[A-Za-z0-9_-]+$/.test(raw)) return null;
|
||||
const decoded = Buffer.from(raw, "base64url").toString("utf8");
|
||||
if (Buffer.from(decoded, "utf8").toString("base64url") !== raw) return null;
|
||||
return OPERATION_NAME_RE.test(decoded) ? decoded : null;
|
||||
}
|
||||
|
||||
async function resolveAuth(credentials, log) {
|
||||
const saJson = parseVertexSaJson(credentials?.apiKey);
|
||||
const projectId =
|
||||
saJson?.project_id ||
|
||||
credentials?.projectId ||
|
||||
credentials?.providerSpecificData?.projectId;
|
||||
const location = credentials?.providerSpecificData?.location || DEFAULT_LOCATION;
|
||||
|
||||
if (!projectId) {
|
||||
return { error: "Vertex video requires a project_id — use Service Account JSON or set providerSpecificData.projectId" };
|
||||
}
|
||||
|
||||
let token = credentials?.accessToken;
|
||||
if (saJson) {
|
||||
const minted = await refreshVertexToken(saJson, log);
|
||||
if (!minted?.accessToken) return { error: "Vertex video: failed to mint access token from service account JSON" };
|
||||
token = minted.accessToken;
|
||||
}
|
||||
if (!token) return { error: "Vertex video requires Service Account JSON or an OAuth access token (raw API keys are not supported)" };
|
||||
|
||||
return { token, projectId, location };
|
||||
}
|
||||
|
||||
/** OpenAI-ish video body → Vertex predictLongRunning body. */
|
||||
function toVertexBody(body) {
|
||||
const instance = { prompt: body.prompt };
|
||||
// Image-to-video: accept the Vertex-native shape or a bare data URL / base64 string.
|
||||
const image = body.image ?? body.image_url;
|
||||
if (image && typeof image === "object") {
|
||||
instance.image = image;
|
||||
} else if (typeof image === "string") {
|
||||
const match = image.match(/^data:([^;]+);base64,(.*)$/s);
|
||||
instance.image = match
|
||||
? { bytesBase64Encoded: match[2], mimeType: match[1] }
|
||||
: { gcsUri: image };
|
||||
}
|
||||
if (body.video && typeof body.video === "object") instance.video = body.video;
|
||||
|
||||
const parameters = {};
|
||||
if (body.n != null) parameters.sampleCount = Number(body.n);
|
||||
if (body.duration != null) parameters.durationSeconds = Number(body.duration);
|
||||
if (body.aspect_ratio) parameters.aspectRatio = body.aspect_ratio;
|
||||
if (body.resolution) parameters.resolution = body.resolution;
|
||||
if (body.seed != null) parameters.seed = body.seed;
|
||||
if (body.negative_prompt) parameters.negativePrompt = body.negative_prompt;
|
||||
// Without storageUri Vertex returns inline base64 bytes; a GCS bucket keeps
|
||||
// the poll response small and is what production callers want.
|
||||
if (body.storage_uri) parameters.storageUri = body.storage_uri;
|
||||
if (body.generate_audio != null) parameters.generateAudio = !!body.generate_audio;
|
||||
|
||||
return { instances: [instance], ...(Object.keys(parameters).length ? { parameters } : {}) };
|
||||
}
|
||||
|
||||
/** Vertex operation → the async-job shape 9Router clients already poll for. */
|
||||
function fromVertexOperation(json) {
|
||||
if (!json?.name) return json;
|
||||
const id = encodeJobId(json.name);
|
||||
if (json.error) {
|
||||
return { id, request_id: id, status: "failed", error: json.error };
|
||||
}
|
||||
if (!json.done) {
|
||||
return { id, request_id: id, status: "pending" };
|
||||
}
|
||||
const samples =
|
||||
json.response?.videos ||
|
||||
json.response?.generateVideoResponse?.generatedSamples ||
|
||||
[];
|
||||
const videos = samples.map((s) => ({
|
||||
url: s.gcsUri || s.video?.uri || s.uri || null,
|
||||
b64_json: s.bytesBase64Encoded || s.video?.bytesBase64Encoded || null,
|
||||
mime_type: s.mimeType || s.video?.mimeType || "video/mp4",
|
||||
}));
|
||||
return { id, request_id: id, status: "completed", video: videos[0] || null, videos };
|
||||
}
|
||||
|
||||
export default {
|
||||
async buildRequest({ config, action, requestId, rawBody, contentType, credentials, log }) {
|
||||
if (contentType && !contentType.includes("application/json")) {
|
||||
return { error: "Vertex video requires an application/json body" };
|
||||
}
|
||||
|
||||
const auth = await resolveAuth(credentials, log);
|
||||
if (auth.error) return { error: auth.error };
|
||||
const { token, projectId, location } = auth;
|
||||
const base = (config.baseUrl || "https://aiplatform.googleapis.com").replace(/\/$/, "");
|
||||
const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` };
|
||||
|
||||
if (requestId) {
|
||||
const operationName = decodeJobId(requestId);
|
||||
if (!operationName) return { error: "Invalid Vertex video job id" };
|
||||
return {
|
||||
method: "POST",
|
||||
url: `${base}/v1/${modelPathOf(operationName)}:fetchPredictOperation`,
|
||||
headers,
|
||||
body: JSON.stringify({ operationName }),
|
||||
};
|
||||
}
|
||||
|
||||
if (action !== "generations") {
|
||||
// ponytail: Veo extend/edit go through generations with `video`/`image` in the body.
|
||||
return { error: `Vertex video supports 'generations' only (got '${action}')` };
|
||||
}
|
||||
|
||||
let body;
|
||||
try {
|
||||
body = JSON.parse(typeof rawBody === "string" ? rawBody : rawBody.toString("utf8"));
|
||||
} catch {
|
||||
return { error: "Invalid JSON body" };
|
||||
}
|
||||
if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" };
|
||||
// Plain model id only — a path segment carrying "/" or ".." would rewrite the URL.
|
||||
if (!/^[A-Za-z0-9._-]+$/.test(body.model)) return { error: "Invalid Vertex video model id" };
|
||||
if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" };
|
||||
|
||||
return {
|
||||
method: "POST",
|
||||
url: `${base}/v1/projects/${projectId}/locations/${location}/publishers/google/models/${body.model}:predictLongRunning`,
|
||||
headers,
|
||||
body: JSON.stringify(toVertexBody(body)),
|
||||
};
|
||||
},
|
||||
|
||||
transformResponse: fromVertexOperation,
|
||||
};
|
||||
@@ -47,7 +47,6 @@ export {
|
||||
refreshAccessToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshIflowToken,
|
||||
refreshGitHubToken,
|
||||
|
||||
@@ -6,6 +6,16 @@
|
||||
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
|
||||
// 4. DEFAULT_CAPABILITIES — safe floor (always returned)
|
||||
//
|
||||
// Two extra layers then refine the result, and neither can override the hand
|
||||
// written tables above (steps 1-2 short-circuit before they are consulted):
|
||||
// • the synced catalog — modalities keyed by model, limits keyed by provider
|
||||
// + model, refreshed from models.dev in the background. It reads a file, so
|
||||
// the server installs it via setCatalogSource(); this module stays free of
|
||||
// node:fs because the dashboard bundles it into the browser too.
|
||||
// • visionPatterns.js — name-based vision detection, last resort so a model
|
||||
// nobody has catalogued yet still accepts images.
|
||||
// Both only ever turn a capability ON.
|
||||
//
|
||||
// ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
|
||||
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+
|
||||
// models, MIT). Each model exposes the exact fields we map below:
|
||||
@@ -23,6 +33,7 @@
|
||||
// 2.0+, Grok, Perplexity). Verify with: curl -s https://models.dev/api.json
|
||||
|
||||
import { matchPattern } from "./pricing.js";
|
||||
import { looksLikeVisionModel } from "./visionPatterns.js";
|
||||
|
||||
/**
|
||||
* Safe floor — every resolved result is merged over this so consumers
|
||||
@@ -46,6 +57,7 @@ export const DEFAULT_CAPABILITIES = {
|
||||
thinkingFormat: null,
|
||||
thinkingCanDisable: true, // false → model cannot turn thinking off (clamp to min instead of disable)
|
||||
thinkingRange: null, // { min, max } for budget formats; null = no clamp
|
||||
thinkingEffortSupported: false, // zai format only: model accepts a reasoning_effort level (GLM-5.2+; older GLM ignores it)
|
||||
// limits (tokens)
|
||||
contextWindow: 200000,
|
||||
maxOutput: 64000,
|
||||
@@ -71,7 +83,8 @@ export function capabilitiesFromServiceKind(kind) {
|
||||
* otherwise mis-match. Only declare deltas vs DEFAULT.
|
||||
*/
|
||||
export const MODEL_CAPABILITIES = {
|
||||
// Claude Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
|
||||
// Claude Fable 5.1, Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
|
||||
"claude-fable-5-1": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
"claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
|
||||
@@ -94,8 +107,26 @@ export const MODEL_CAPABILITIES = {
|
||||
// Gemini image-gen / OpenAI image / xai image variants
|
||||
"gpt-image-1": { imageOutput: true, tools: false },
|
||||
|
||||
// GLM vision variant (text GLM has no vision)
|
||||
"glm-4.6v": { vision: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000 },
|
||||
// GLM vision variants (text GLM has no vision) — 5.3-Flash and 5V-Turbo are
|
||||
// natively multimodal per z.ai, and 5.3-Flash carries the full 1M window.
|
||||
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "zai", contextWindow: 1000000, maxOutput: 131072 },
|
||||
"glm-4.6v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 128000, maxOutput: 32768 },
|
||||
"glm-4.5v": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "zai", contextWindow: 64000, maxOutput: 16384 },
|
||||
// GLM-5.2 has 1M context — pattern *glm-5* only gives 200k, so override here
|
||||
"glm-5.2": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 131072 },
|
||||
|
||||
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
|
||||
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
|
||||
// DeepSeek V4.1-Flash is natively multimodal — models.dev lists
|
||||
// opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream
|
||||
// the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the
|
||||
// same image capability as the exp id above. "deepseek-flash" is the GA id on the
|
||||
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
|
||||
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
|
||||
// short-circuits the pattern table, so a vision-only delta would drop them.
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
|
||||
|
||||
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
|
||||
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
@@ -108,6 +139,12 @@ export const MODEL_CAPABILITIES = {
|
||||
"kimi-for-coding-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
"kimi-k2.7-code": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
"kimi-k2.7-code-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
// OpenCode Free Muse Spark — multimodal (text+image per models.dev meta/muse-spark)
|
||||
// via OpenAI Responses input_image; reasoning supports up to xhigh.
|
||||
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
// OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output
|
||||
"union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 },
|
||||
};
|
||||
|
||||
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
@@ -130,7 +167,14 @@ export const PROVIDER_CAPABILITIES = {
|
||||
"deepseek-ai/deepseek-v4-pro": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
},
|
||||
// glm-5.3-flash on OpenCode Go is served by a backend that rejects the z.ai
|
||||
// `thinking` object (400: unknown field "thinking") and wants reasoning_effort.
|
||||
// Overrides the global entry, whose z.ai shape is correct for z.ai itself.
|
||||
"opencode-go": {
|
||||
"glm-5.3-flash": { vision: true, videoInput: true, pdf: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 131072 },
|
||||
},
|
||||
"codex": {
|
||||
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
|
||||
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
|
||||
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
|
||||
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
|
||||
@@ -155,32 +199,66 @@ export const PROVIDER_CAPABILITIES = {
|
||||
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
|
||||
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
|
||||
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
|
||||
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking
|
||||
// off → thinkingCanDisable:false (clamped to minimal instead of disabled).
|
||||
// (see registry thinkingFormat). For thinkingCanDisable use the server's
|
||||
// reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
|
||||
// below; it is NOT the inverse of onlyReasoning.
|
||||
"codebuddy-cn": {
|
||||
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 },
|
||||
// maxOutput 64000 per both the plugin-baked fallback and the live server
|
||||
// table (the old 38000 had no source and truncated output).
|
||||
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
|
||||
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
|
||||
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 },
|
||||
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
|
||||
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
|
||||
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
|
||||
// Per-model values mirror the server's product-config payload (the plugin
|
||||
// fetches it from copilot.tencent.com; the `models[]` entries carry
|
||||
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
|
||||
// maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
|
||||
// plugin-baked fallback disagree, the server table wins.
|
||||
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
|
||||
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
|
||||
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
|
||||
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
|
||||
// their thinking is switchable; the hy* models are forced always-on.
|
||||
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
|
||||
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
|
||||
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
|
||||
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
|
||||
// deepseek-v4.1-flash replaces v4-flash (dropped from the server list;
|
||||
// the old endpoint still answers 200 but the published list is the
|
||||
// contract). maxOutput 128000 per the server's product-config payload.
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
|
||||
},
|
||||
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
|
||||
"poolside": {
|
||||
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
|
||||
"laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 },
|
||||
},
|
||||
// Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge
|
||||
// the library page publishes for this model (text+image in, 1M context).
|
||||
// ponytail: thinkingFormat stays "deepseek" to preserve today's body shape;
|
||||
// Ollama's native toggle is the top-level `think` field (bool or
|
||||
// low/medium/high/max), which no format in thinkingUnified.js emits yet —
|
||||
// openai-to-ollama.js drops it. Wire a "think" format when thinking on
|
||||
// Ollama Cloud is actually needed.
|
||||
"ollama": {
|
||||
"deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
},
|
||||
};
|
||||
|
||||
// Qoder CN serves the identical model catalog from the CN gateway, so it shares
|
||||
// the intl Qoder capability table verbatim (vision/reasoning/contextWindow).
|
||||
PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"];
|
||||
|
||||
/**
|
||||
* Pattern fallback — glob (* = wildcard), matched case-insensitively and
|
||||
* anchored (^...$) so a pattern must match the full model id. ORDER MATTERS:
|
||||
@@ -205,6 +283,8 @@ export const PATTERN_CAPABILITIES = [
|
||||
|
||||
// ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─
|
||||
{ pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } },
|
||||
{ pattern: "*gemini-3.8*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } },
|
||||
{ pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-2.5*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-budget", thinkingRange: { min: 0, max: 24576 }, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
@@ -213,6 +293,9 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
|
||||
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
|
||||
|
||||
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
|
||||
|
||||
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
@@ -233,6 +316,8 @@ export const PATTERN_CAPABILITIES = [
|
||||
// ── Grok (vision + Live Search) ──────────────────────────────────
|
||||
{ pattern: "*grok*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*grok-code*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
// Grok 4.6: 500k context, no text output limit (docs.x.ai/developers/grok-4-6)
|
||||
{ pattern: "*grok-4.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 500000 } },
|
||||
// Grok 4.5 (Grok CLI / Grok Build): 500k context per cli-chat-proxy /v1/models
|
||||
{ pattern: "*grok-4.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 64000 } },
|
||||
{ pattern: "*grok-4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
@@ -243,7 +328,7 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*qwen*vl*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144 } },
|
||||
{ pattern: "*qwen*omni*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 262144, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen*coder*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 } },
|
||||
{ pattern: "*qwen*max*", caps: { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen*max*", caps: { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen3.5*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen3.6*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
{ pattern: "*qwen3.7*", caps: { vision: true, videoInput: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000, maxOutput: 65536 } },
|
||||
@@ -260,13 +345,21 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*kimi*", caps: { reasoning: true, thinkingFormat: "kimi", contextWindow: 262144 } },
|
||||
|
||||
// ── GLM / Z.ai (thinking.enabled; disable via enable_thinking:false) ─
|
||||
// reasoning_effort is only read by z.ai from GLM-5.2 onward (docs.z.ai/guides/capabilities/thinking) —
|
||||
// older GLM (4.x, 5.0, 5.1, 5-turbo, 5v-turbo) ignore it, so gate it per exact version, not the "*glm-5*" catch-all.
|
||||
{ pattern: "*glm-5.3*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-5.2*", caps: { reasoning: true, thinkingFormat: "zai", thinkingEffortSupported: true, contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-5*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-4.7*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000, maxOutput: 128000 } },
|
||||
{ pattern: "*glm-4*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
|
||||
{ pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
|
||||
|
||||
// ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─
|
||||
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } },
|
||||
// v4.1+ has real image input (probed live on Alibaba MaaS: correct color
|
||||
// read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore
|
||||
// them (answered "Unknown"), so vision stays scoped to v4.* dotted releases.
|
||||
{ pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } },
|
||||
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } },
|
||||
{ pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
|
||||
{ pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
|
||||
{ pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } },
|
||||
@@ -274,14 +367,16 @@ export const PATTERN_CAPABILITIES = [
|
||||
|
||||
// ── MiniMax (M3 = adaptive; M2.x cannot disable) ─────────────────
|
||||
{ pattern: "*minimax*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*minimax-m3*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", contextWindow: 1048576, maxOutput: 512000 } },
|
||||
{ pattern: "*minimax-m2.7*", caps: { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 204800, maxOutput: 131072 } },
|
||||
{ pattern: "*minimax-m3*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", contextWindow: 1000000, maxOutput: 131072 } },
|
||||
{ pattern: "*minimax-m2.7*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 204800, maxOutput: 131072 } },
|
||||
{ pattern: "*minimax-m2.5*", caps: { vision: true, reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 204800, maxOutput: 131072 } },
|
||||
{ pattern: "*minimax*", caps: { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 131072 } },
|
||||
|
||||
// ── Xiaomi MiMo (vision, 1M / 262K ctx) ──────────────────────────
|
||||
{ pattern: "*mimo*v2.5*", caps: { vision: true, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*omni*", caps: { vision: true, audioInput: true, contextWindow: 262144, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*", caps: { vision: true, contextWindow: 262144, maxOutput: 131072 } },
|
||||
// ── Xiaomi MiMo (vision + <think>-tag reasoning, always-on, can't disable) ──
|
||||
{ pattern: "*mimo*v2.6*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*v2.5*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*omni*", caps: { vision: true, audioInput: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 131072 } },
|
||||
|
||||
// ── Llama (4 = vision/1M; 3.x = text-only/128K) ──────────────────
|
||||
{ pattern: "*llama-4*", caps: { vision: true, contextWindow: 1000000 } },
|
||||
@@ -308,6 +403,9 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*laguna-s-2.1*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 } },
|
||||
{ pattern: "*laguna*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 } },
|
||||
|
||||
|
||||
// ── OpenCode Free Muse Spark (multimodal text+image; OpenAI Responses reasoning supports up to xhigh) ─
|
||||
{ pattern: "*muse*spark*", caps: { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 } },
|
||||
// ── Others ───────────────────────────────────────────────────────
|
||||
{ pattern: "*hunyuan*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
|
||||
{ pattern: "hy3*", caps: { reasoning: true, thinkingFormat: "hunyuan", contextWindow: 262144, maxOutput: 262144 } },
|
||||
@@ -316,6 +414,65 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*ling-*", caps: { reasoning: true, contextWindow: 128000 } },
|
||||
];
|
||||
|
||||
// OpenRouter-style gateways validate modalities upstream — a text-only model
|
||||
// sent an image gets a clear upstream error instead of silent corruption. So for
|
||||
// unknown models on these providers, trust vision instead of stripping images.
|
||||
const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
|
||||
|
||||
/**
|
||||
* Aggregate capabilities for a combo from its constituent model IDs.
|
||||
* Each entry in comboModels is a fully-qualified "provider/model" string.
|
||||
*
|
||||
* Union: vision, pdf, audioInput, videoInput, imageOutput, audioOutput, search
|
||||
* Intersection: tools
|
||||
* Primary: reasoning fields from the first (primary) model
|
||||
* Conservative: contextWindow = min; maxOutput = max
|
||||
*
|
||||
* @param {string[]} comboModels
|
||||
* @param {Object|null} [comboLookup] optional map of combo name → models array for nested resolution
|
||||
* @param {Function|null} [resolveCaps] optional (fullId) → caps override. The synced model
|
||||
* catalog is server-only (it reads a file), so a browser-side resolution cannot see the
|
||||
* limits it supplies and silently falls back to the generic patterns below. Callers that
|
||||
* have the server's answer (/api/models, via useModelCaps) pass it here; it is merged over
|
||||
* the local tables, so fields it does not carry (tools, pdf, audio/video, thinking*) survive.
|
||||
* @param {number} [_depth] internal recursion depth guard
|
||||
* @returns {object|null} full capabilities object, or null for empty input
|
||||
*/
|
||||
export function aggregateComboCapabilities(comboModels, comboLookup = null, resolveCaps = null, _depth = 0) {
|
||||
if (!comboModels?.length || _depth > 6) return null;
|
||||
const allCaps = comboModels.map((fullId) => {
|
||||
// Nested combo: bare name (no slash) that exists in the lookup — recurse
|
||||
if (!fullId.includes("/") && comboLookup?.[fullId]) {
|
||||
return aggregateComboCapabilities(comboLookup[fullId], comboLookup, resolveCaps, _depth + 1)
|
||||
?? resolveCaps?.(fullId)
|
||||
?? getCapabilitiesForModel(null, fullId);
|
||||
}
|
||||
const slash = fullId.indexOf("/");
|
||||
const provider = slash === -1 ? null : fullId.slice(0, slash);
|
||||
const model = slash === -1 ? fullId : fullId.slice(slash + 1);
|
||||
const local = getCapabilitiesForModel(provider, model);
|
||||
const override = resolveCaps?.(fullId);
|
||||
return override ? { ...local, ...override } : local;
|
||||
});
|
||||
const first = allCaps[0];
|
||||
return {
|
||||
vision: allCaps.some((c) => c.vision),
|
||||
pdf: allCaps.some((c) => c.pdf),
|
||||
audioInput: allCaps.some((c) => c.audioInput),
|
||||
videoInput: allCaps.some((c) => c.videoInput),
|
||||
imageOutput: allCaps.some((c) => c.imageOutput),
|
||||
audioOutput: allCaps.some((c) => c.audioOutput),
|
||||
search: allCaps.some((c) => c.search),
|
||||
tools: allCaps.every((c) => c.tools),
|
||||
reasoning: first.reasoning,
|
||||
thinkingFormat: first.thinkingFormat,
|
||||
thinkingCanDisable: first.thinkingCanDisable,
|
||||
thinkingRange: first.thinkingRange,
|
||||
contextWindow: Math.min(...allCaps.map((c) => c.contextWindow)),
|
||||
maxOutput: Math.max(...allCaps.map((c) => c.maxOutput)),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve capabilities for a model using the 4-step fallback chain,
|
||||
* merged over DEFAULT_CAPABILITIES so the result is always complete.
|
||||
@@ -324,30 +481,187 @@ export const PATTERN_CAPABILITIES = [
|
||||
* @param {string} model
|
||||
* @returns {object} full capabilities object
|
||||
*/
|
||||
const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
|
||||
// ── Server-injected readers ──────────────────────────────────────────
|
||||
// Next.js compiles instrumentation and each API route into SEPARATE server
|
||||
// bundles, so a plain module-local `let` would give every bundle its own copy of
|
||||
// this file and a source installed at boot would be invisible to the request
|
||||
// handlers (silently: the setters still "succeed"). The slots therefore live on
|
||||
// globalThis, which IS shared across server bundles in the same process.
|
||||
// Same reason the browser bundle is safe: it never calls a setter, so the slots
|
||||
// stay empty and every consumer below short-circuits. Every read goes through
|
||||
// globalThis: caching it locally would keep a reader alive in other copies after
|
||||
// setCatalogSource(null).
|
||||
let catalogSource = null;
|
||||
const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
|
||||
catalog: null, // { getModalities, getLimits } — synced models.dev catalog
|
||||
userCaps: null, // (provider, model) => asserted caps — dashboard toggles
|
||||
});
|
||||
|
||||
/**
|
||||
* Install the synced catalog reader (server only).
|
||||
* @param {{ getModalities: (provider: string, model: string) => object|null,
|
||||
* getLimits: (provider: string, model: string) => object|null } | null} source
|
||||
*/
|
||||
export function setCatalogSource(source) {
|
||||
catalogSource = source || null;
|
||||
if (SOURCE_SLOTS) SOURCE_SLOTS.catalog = source || null;
|
||||
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source || null;
|
||||
}
|
||||
|
||||
function getCatalogSource() {
|
||||
if (typeof globalThis === "undefined") return catalogSource;
|
||||
return SOURCE_SLOTS?.catalog || globalThis.__9rCatalogSource || null;
|
||||
}
|
||||
|
||||
// Capabilities the user asserted per provider+model (dashboard "Add/Edit Model"
|
||||
// toggles), installed by the server from the custom-model store. Unlike the
|
||||
// catalog and name heuristics this is authoritative and two-directional: it can
|
||||
// turn a capability OFF as well as on.
|
||||
const USER_CAPS_KEYS = ["vision", "pdf", "audioInput", "videoInput", "imageOutput", "audioOutput", "search", "tools", "reasoning", "thinkingFormat", "contextWindow", "maxOutput"];
|
||||
// (slot lives in SOURCE_SLOTS above — see the cross-bundle note)
|
||||
|
||||
/**
|
||||
* Install the user-asserted caps reader (server only).
|
||||
* @param {(provider: string|null, model: string) => object|null} source sync lookup
|
||||
*/
|
||||
export function setUserCapsSource(source) {
|
||||
SOURCE_SLOTS.userCaps = typeof source === "function" ? source : null;
|
||||
}
|
||||
|
||||
// Last step of every resolution path: the user's own assertion wins over any
|
||||
// heuristic, including the tables above (a hand-typed model id can collide with
|
||||
// a pattern entry that describes a different product).
|
||||
function applyUserCaps(result, provider, model) {
|
||||
const userCapsSource = SOURCE_SLOTS.userCaps;
|
||||
if (!userCapsSource) return result;
|
||||
let asserted = null;
|
||||
try {
|
||||
asserted = userCapsSource(provider, model);
|
||||
} catch {
|
||||
return result;
|
||||
}
|
||||
if (!asserted || typeof asserted !== "object") return result;
|
||||
for (const key of USER_CAPS_KEYS) {
|
||||
if (asserted[key] === undefined) continue;
|
||||
result[key] = asserted[key];
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// Apply the synced catalog + name heuristic on top of a table-resolved result.
|
||||
// Strictly additive: a capability already true stays true, and a false one only
|
||||
// flips when an outside source positively declares support.
|
||||
function refine(base, provider, model) {
|
||||
const result = { ...DEFAULT_CAPABILITIES, ...base };
|
||||
|
||||
const source = getCatalogSource();
|
||||
if (source) {
|
||||
const modalities = source.getModalities(provider, model);
|
||||
if (modalities) {
|
||||
for (const key of MODALITY_KEYS) {
|
||||
if (modalities[key] === true) result[key] = true;
|
||||
}
|
||||
}
|
||||
|
||||
const limits = source.getLimits(provider, model);
|
||||
if (limits) {
|
||||
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
|
||||
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
|
||||
}
|
||||
}
|
||||
|
||||
if (!result.vision && looksLikeVisionModel(model)) result.vision = true;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models
|
||||
// default to vision; only this denylist stays text-only.
|
||||
const COMMANDCODE_TEXT_ONLY = new Set([
|
||||
"deepseek/deepseek-v4-pro",
|
||||
"deepseek/deepseek-v4-flash",
|
||||
"deepseek/deepseek-v4-flash-fast",
|
||||
"zai-org/glm-5.3",
|
||||
"zai-org/glm-5.2",
|
||||
"zai-org/glm-5.2-fast",
|
||||
"zai-org/glm-5.1",
|
||||
"zai-org/glm-5",
|
||||
"minimaxai/minimax-m2.7",
|
||||
"minimax/minimax-m2.7-free",
|
||||
"minimaxai/minimax-m2.5",
|
||||
"xiaomi/mimo-v2.5-pro",
|
||||
"qwen/qwen3.6-max-preview",
|
||||
"qwen/qwen3.7-max",
|
||||
"meituan/longcat-2.0:free",
|
||||
"stepfun/step-3.5-flash",
|
||||
"tencent/hy4-preview",
|
||||
"tencent/hy3",
|
||||
"tencent/hy3-paid",
|
||||
"nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"poolside/laguna-s-2.1-free",
|
||||
"inclusionai/ling-3.0-flash-free",
|
||||
"inclusionai/ling-3.0-flash-sante:free",
|
||||
]);
|
||||
|
||||
function isCommandCodeTextOnly(model) {
|
||||
const key = String(model || "").toLowerCase();
|
||||
if (COMMANDCODE_TEXT_ONLY.has(key)) return true;
|
||||
for (const id of COMMANDCODE_TEXT_ONLY) {
|
||||
const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
|
||||
if (key === base || key.endsWith("/" + base)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
export function getCapabilitiesForModel(provider, model) {
|
||||
if (!model) return { ...DEFAULT_CAPABILITIES };
|
||||
|
||||
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
|
||||
// 1. Provider-specific override
|
||||
if (provider) {
|
||||
const providerCaps = PROVIDER_CAPABILITIES[provider];
|
||||
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
}
|
||||
|
||||
// 2. Canonical exact
|
||||
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
|
||||
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
|
||||
|
||||
// 3. Pattern match (first match wins)
|
||||
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
|
||||
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
|
||||
return { ...DEFAULT_CAPABILITIES, ...caps };
|
||||
const resolve = () => {
|
||||
// CommandCode wire is /alpha/generate for every model. Family patterns
|
||||
// (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here.
|
||||
if (provider === "commandcode" || provider === "cmc") {
|
||||
const providerCaps = PROVIDER_CAPABILITIES.commandcode;
|
||||
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
return {
|
||||
...DEFAULT_CAPABILITIES,
|
||||
reasoning: true,
|
||||
thinkingFormat: "commandcode",
|
||||
thinkingEffortSupported: true,
|
||||
vision: !isCommandCodeTextOnly(model),
|
||||
contextWindow: 1000000,
|
||||
maxOutput: 384000,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Floor
|
||||
return { ...DEFAULT_CAPABILITIES };
|
||||
// 1. Provider-specific override
|
||||
if (provider) {
|
||||
const providerCaps = PROVIDER_CAPABILITIES[provider];
|
||||
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
}
|
||||
|
||||
// 2. Canonical exact
|
||||
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
|
||||
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
|
||||
|
||||
// 3. Pattern match (first match wins), refined by catalog + name heuristic
|
||||
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
|
||||
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
|
||||
return refine(caps, provider, model);
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Floor (upstream-validated gateways keep vision on for unknown models)
|
||||
if (provider && TRUST_UPSTREAM_VISION.has(provider)) {
|
||||
return { ...refine(null, provider, model), vision: true };
|
||||
}
|
||||
return refine(null, provider, model);
|
||||
};
|
||||
|
||||
return applyUserCaps(resolve(), provider, model);
|
||||
}
|
||||
|
||||
82
open-sse/providers/catalogOverride.js
Normal file
82
open-sse/providers/catalogOverride.js
Normal file
@@ -0,0 +1,82 @@
|
||||
// Read side of the model catalog synced from models.dev.
|
||||
//
|
||||
// The file is the source of truth; the only thing held in memory is a parsed
|
||||
// copy dropped as soon as the file's mtime changes. getCapabilitiesForModel is
|
||||
// synchronous and runs per request, so the hot path is one stat (~1us) and the
|
||||
// parse (~0.1ms on a ~18KB file) only reruns after a sync.
|
||||
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import { DATA_DIR } from "@/lib/dataDir.js";
|
||||
|
||||
export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
|
||||
// Trimmed upstream catalog, read by the add-models skill (not by the router).
|
||||
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
|
||||
|
||||
// Schema of the file this module reads. The writer stamps it; a file carrying an
|
||||
// older value predates provider-scoped modality keys, and its flat keys are not
|
||||
// looked up here, so the sync rebuilds it instead of asking upstream for a 304.
|
||||
export const CATALOG_VERSION = 2;
|
||||
|
||||
const EMPTY = { models: {}, providers: {} };
|
||||
let cache = EMPTY;
|
||||
let cachedMtime = -1;
|
||||
|
||||
// "zai-org/GLM-4.6V:free" -> "glm-4.6v"
|
||||
function baseId(model) {
|
||||
if (!model) return "";
|
||||
const withoutVendor = model.includes("/") ? model.split("/").pop() : model;
|
||||
return withoutVendor.toLowerCase().split(":")[0];
|
||||
}
|
||||
|
||||
function load() {
|
||||
let mtime;
|
||||
try {
|
||||
mtime = fs.statSync(CATALOG_FILE).mtimeMs;
|
||||
} catch {
|
||||
cache = EMPTY;
|
||||
cachedMtime = -1;
|
||||
return cache;
|
||||
}
|
||||
if (mtime === cachedMtime) return cache;
|
||||
|
||||
cachedMtime = mtime;
|
||||
try {
|
||||
const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8"));
|
||||
cache = { models: parsed?.models || {}, providers: parsed?.providers || {} };
|
||||
} catch {
|
||||
cache = EMPTY;
|
||||
}
|
||||
return cache;
|
||||
}
|
||||
|
||||
// Modalities are recorded per gateway upstream, and gateways disagree about the
|
||||
// same weights — some do not proxy images at all — so the key is provider +
|
||||
// model, in the local provider id space, exactly like the limits below. Keying
|
||||
// by model id alone made short ids collide across vendors: "auto", "free" and
|
||||
// "efficient" are router modes in one catalog and model names in another, and a
|
||||
// request to the router mode inherited a stranger's vision.
|
||||
export function getCatalogModalities(provider, model) {
|
||||
if (!provider) return null;
|
||||
return load().models[`${provider}:${baseId(model)}`] || null;
|
||||
}
|
||||
|
||||
// Context and output limits are a property of the gateway too: each one
|
||||
// truncates differently, so these stay keyed by provider + model.
|
||||
export function getCatalogLimits(provider, model) {
|
||||
const byProvider = provider && load().providers[provider];
|
||||
if (!byProvider) return null;
|
||||
return byProvider[model] || byProvider[baseId(model)] || null;
|
||||
}
|
||||
|
||||
// Force a re-read on the next lookup (called right after a sync writes the file).
|
||||
export function invalidateCatalog() {
|
||||
cachedMtime = -1;
|
||||
}
|
||||
|
||||
// Hand the reader to capabilities.js. That module is bundled into the browser
|
||||
// too, so it cannot import this file directly — the server pushes it in.
|
||||
export async function installCatalogSource() {
|
||||
const { setCatalogSource } = await import("./capabilities.js");
|
||||
setCatalogSource({ getModalities: getCatalogModalities, getLimits: getCatalogLimits });
|
||||
}
|
||||
@@ -23,7 +23,7 @@ function buildTransport(transport, oauth) {
|
||||
const MEDIA_KEYS = new Set([
|
||||
"serviceKinds", "ttsConfig", "sttConfig", "embeddingConfig",
|
||||
"imageConfig", "imageToTextConfig", "videoConfig", "musicConfig",
|
||||
"searchViaChat", "searchConfig", "fetchConfig",
|
||||
"searchViaChat", "searchConfig", "fetchConfig", "systemoneConfig",
|
||||
"modelsFetcher", "mediaPriority", "hiddenKinds",
|
||||
]);
|
||||
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
|
||||
// Codex auto-generates a "-review" variant for each llm model (review quota family)
|
||||
export const CODEX_REVIEW_SUFFIX = "-review";
|
||||
|
||||
@@ -18,3 +20,27 @@ export function withCodexReviewModels(models) {
|
||||
];
|
||||
});
|
||||
}
|
||||
|
||||
export function isMuseSparkModel(modelId) {
|
||||
if (!modelId || typeof modelId !== "string") return false;
|
||||
const clean = modelId.replace(/\([^()]+\)\s*$/, "").trim();
|
||||
const base = clean.includes("/") ? clean.split("/").pop() : clean;
|
||||
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
|
||||
}
|
||||
|
||||
// Endpoint families for OpenCode models outside the curated registry (modelsFetcher /
|
||||
// passthrough ids) — regex keeps auto-fetched models on the right endpoint:
|
||||
// /responses (gpt/grok/muse-spark), /messages (minimax/qwen), /chat/completions (rest).
|
||||
// Curated registry entries always win; this is the unknown-id fallback only.
|
||||
const OPENCODE_FAMILIES = [
|
||||
{ match: /^(grok|gpt|muse[-_]?spark)/i, supportedFormats: [FORMATS.OPENAI_RESPONSES], targetFormat: FORMATS.OPENAI_RESPONSES },
|
||||
{ match: /^deepseek-v4-(pro|flash)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE, FORMATS.OPENAI_RESPONSES] },
|
||||
{ match: /^(minimax|qwen)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE] },
|
||||
{ match: /^claude-/i, supportedFormats: [FORMATS.CLAUDE] },
|
||||
];
|
||||
|
||||
export function opencodeFamilyFormats(modelId) {
|
||||
if (!modelId || typeof modelId !== "string") return null;
|
||||
const base = modelId.replace(/\([^()]+\)\s*$/, "").trim();
|
||||
return OPENCODE_FAMILIES.find((f) => f.match.test(base)) || null;
|
||||
}
|
||||
|
||||
@@ -38,3 +38,11 @@ export function modelStrip(model) {
|
||||
export function modelTargetFormat(model) {
|
||||
return model?.targetFormat || MODEL_DEFAULTS.targetFormat;
|
||||
}
|
||||
|
||||
// Per-model declared upstream formats (e.g. ["openai", "claude"]). Guards the
|
||||
// sourceFormat-matched transport for multi-endpoint providers whose models differ
|
||||
// in endpoint support (opencode-go: kimi/glm only do /chat/completions, minimax/qwen
|
||||
// also do /messages, deepseek also does /responses).
|
||||
export function modelSupportedFormats(model) {
|
||||
return model?.supportedFormats || null;
|
||||
}
|
||||
|
||||
@@ -2,8 +2,28 @@
|
||||
//
|
||||
// Fallback order (first match wins):
|
||||
// 1. PROVIDER_PRICING[provider][model] — provider-specific override
|
||||
// 2. MODEL_PRICING[model] — canonical model price (provider-agnostic)
|
||||
// 3. PATTERN_PRICING — glob pattern match (e.g. "codex-*")
|
||||
// 2. FREE_MODEL_NAMESPACES — upstream bills these at $0
|
||||
// 3. MODEL_PRICING[model] — canonical model price (provider-agnostic)
|
||||
// 4. PATTERN_PRICING — glob pattern match (e.g. "codex-*")
|
||||
|
||||
/**
|
||||
* Namespaces upstream meters at $0. A free model must never inherit a paid
|
||||
* rate: the vendor-prefix strip in getPricingForModel() would turn
|
||||
* "cline-free/deepseek-v4.1-flash" into "deepseek-v4.1-flash" and match
|
||||
* MODEL_PRICING, so the namespace is checked before both fallbacks.
|
||||
*/
|
||||
export const FREE_MODEL_NAMESPACES = ["cline-free/"];
|
||||
|
||||
export const ZERO_PRICING = {
|
||||
input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0,
|
||||
};
|
||||
|
||||
/** True when the model id sits in a namespace upstream bills at $0. */
|
||||
export function isFreeModel(model) {
|
||||
if (!model) return false;
|
||||
const lower = String(model).toLowerCase();
|
||||
return FREE_MODEL_NAMESPACES.some((ns) => lower.startsWith(ns));
|
||||
}
|
||||
|
||||
/**
|
||||
* Canonical model pricing — provider-agnostic.
|
||||
@@ -53,10 +73,19 @@ export const MODEL_PRICING = {
|
||||
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
|
||||
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
|
||||
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
|
||||
// === Gemini ===
|
||||
"gemini-3.8-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.8-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.8-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.8-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
@@ -102,6 +131,8 @@ export const MODEL_PRICING = {
|
||||
"deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 },
|
||||
|
||||
// === GLM ===
|
||||
@@ -141,6 +172,123 @@ export const PROVIDER_PRICING = {
|
||||
gh: {
|
||||
"gpt-5.3-codex": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
},
|
||||
// TokenRouter — exact rates from https://api.tokenrouter.com/api/pricing ($1/1M tokens).
|
||||
// Ratio→USD: input = model_ratio×2, output = model_ratio×completion_ratio×2.
|
||||
// These override the canonical MODEL_PRICING/PATTERN_PRICING, whose rates often
|
||||
// differ from TokenRouter's reseller pricing.
|
||||
tokenrouter: {
|
||||
"MiniMax-M3": { input: 0.3, output: 1.2, cached: 0.06, reasoning: 1.2 },
|
||||
"anthropic/claude-fable-5": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-haiku-4.5": { input: 1.0, output: 5.0, cached: 0.1, cache_creation: 1.25, reasoning: 5.0 },
|
||||
"anthropic/claude-opus-4.5": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.6": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.7": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.7-fast": { input: 30, output: 150, cached: 3.0, reasoning: 150 },
|
||||
"anthropic/claude-opus-4.8": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.8-fast": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-opus-5": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-5-fast": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-sonnet-4": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-4.5": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-4.6": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-5": { input: 2, output: 10, cached: 0.2, reasoning: 10 },
|
||||
"claude-opus-4-8-m-aws": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"deepseek/deepseek-v3.2": { input: 0.26, output: 0.38, cached: 0.13, reasoning: 0.38 },
|
||||
"deepseek/deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28 },
|
||||
"deepseek/deepseek-v4-flash-0731": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28 },
|
||||
"deepseek/deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87 },
|
||||
"ex/gpt-5.4": { input: 2.5, output: 15.0, cached: 0.25, reasoning: 15.0 },
|
||||
"google/gemini-2.5-flash-image": { input: 0.3, output: 2.5, reasoning: 2.5 },
|
||||
"google/gemini-3-flash-preview": { input: 0.5, output: 3.0, cached: 0.05, cache_creation: 0.08333, reasoning: 3.0 },
|
||||
"google/gemini-3-pro-image-preview": { input: 2, output: 12, reasoning: 12 },
|
||||
"google/gemini-3.1-flash-image-preview": { input: 0.5, output: 3.0, reasoning: 3.0 },
|
||||
"google/gemini-3.1-flash-lite-image": { input: 0.25, output: 1.5, reasoning: 1.5 },
|
||||
"google/gemini-3.1-pro-preview": { input: 2, output: 12, cached: 0.2, cache_creation: 0.375, reasoning: 12 },
|
||||
"google/gemini-3.5-flash": { input: 1.5, output: 9.0, cached: 0.15, cache_creation: 0.08333, reasoning: 9.0 },
|
||||
"google/gemini-3.5-flash-lite": { input: 0.3, output: 2.5, cached: 0.03, cache_creation: 0.08333, reasoning: 2.5 },
|
||||
"google/gemini-3.6-flash": { input: 1.5, output: 7.5, cached: 0.15, cache_creation: 0.08333, reasoning: 7.5 },
|
||||
"google/gemini-embedding-2": { input: 1.0, output: 6.0, cached: 0.1, reasoning: 6.0 },
|
||||
"google/gemma-4-26b-a4b-it": { input: 0.06, output: 0.33, reasoning: 0.33 },
|
||||
"kling-3.0-turbo": { input: 2.1, output: 2.1, reasoning: 2.1 },
|
||||
"microsoft/mai-image-2.5": { input: 5.0, output: 47.0, reasoning: 47.0 },
|
||||
"minimax/minimax-m2-her": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.1": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.1-highspeed": { input: 0.6, output: 2.4, cached: 0.06, reasoning: 2.4 },
|
||||
"minimax/minimax-m2.5": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.7": { input: 0.3, output: 1.2, cached: 0.06, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.7-highspeed": { input: 0.6, output: 2.4, cached: 0.06, reasoning: 2.4 },
|
||||
"miromind/mirothinker-1-7-deepresearch": { input: 4, output: 25.0, reasoning: 25.0 },
|
||||
"miromind/mirothinker-1-7-deepresearch-mini": { input: 1.25, output: 10.0, reasoning: 10.0 },
|
||||
"mistralai/devstral-2512": { input: 0.4, output: 2.0, cached: 0.04, reasoning: 2.0 },
|
||||
"mistralai/mistral-medium-3-5": { input: 1.5, output: 7.5, reasoning: 7.5 },
|
||||
"mistralai/mistral-small-2603": { input: 0.15, output: 0.6, cached: 0.015, reasoning: 0.6 },
|
||||
"mistralai/voxtral-small-24b-2507": { input: 0.1, output: 0.3, cached: 0.01, reasoning: 0.3 },
|
||||
"moonshotai/kimi-k2.5": { input: 0.6, output: 3.0, cached: 0.1, reasoning: 3.0 },
|
||||
"moonshotai/kimi-k2.6": { input: 0.95, output: 4.0, cached: 0.16, reasoning: 4.0 },
|
||||
"moonshotai/kimi-k2.7-code": { input: 0.9286, output: 3.8571, cached: 0.1857, reasoning: 3.8571 },
|
||||
"moonshotai/kimi-k3": { input: 3.0, output: 15.0, cached: 0.3, reasoning: 15.0 },
|
||||
"nvidia/nemotron-3-super-120b-a12b": { input: 0.3, output: 0.9, cached: 0.1, reasoning: 0.9 },
|
||||
"openai/gpt-4o-mini": { input: 0.15, output: 0.6, cached: 0.075, reasoning: 0.6 },
|
||||
"openai/gpt-5": { input: 1.25, output: 10.0, cached: 0.125, reasoning: 10.0 },
|
||||
"openai/gpt-5-image": { input: 10, output: 40, cached: 2.5, reasoning: 40 },
|
||||
"openai/gpt-5-image-mini": { input: 2.5, output: 8.0, cached: 0.25, reasoning: 8.0 },
|
||||
"openai/gpt-5-mini": { input: 0.25, output: 2.0, cached: 0.025, reasoning: 2.0 },
|
||||
"openai/gpt-5.2": { input: 1.75, output: 14.0, cached: 0.175, reasoning: 14.0 },
|
||||
"openai/gpt-5.3-codex": { input: 1.75, output: 14.0, cached: 0.175, reasoning: 14.0 },
|
||||
"openai/gpt-5.4": { input: 2.5, output: 15.0, cached: 0.25, reasoning: 15.0 },
|
||||
"openai/gpt-5.4-image-2": { input: 8, output: 30.0, cached: 2.0, reasoning: 30.0 },
|
||||
"openai/gpt-5.4-mini": { input: 0.75, output: 4.5, cached: 0.075, reasoning: 4.5 },
|
||||
"openai/gpt-5.4-nano": { input: 0.2, output: 1.25, cached: 0.02, reasoning: 1.25 },
|
||||
"openai/gpt-5.4-pro": { input: 30, output: 180, reasoning: 180 },
|
||||
"openai/gpt-5.5": { input: 5.0, output: 30.0, cached: 0.5, reasoning: 30.0 },
|
||||
"openai/gpt-5.5-pro": { input: 30, output: 180, reasoning: 180 },
|
||||
"openai/gpt-5.6-luna": { input: 0.2, output: 1.2, cached: 0.02, cache_creation: 0.25, reasoning: 1.2 },
|
||||
"openai/gpt-5.6-sol": { input: 5.0, output: 30.0, cached: 0.5, cache_creation: 6.25, reasoning: 30.0 },
|
||||
"openai/gpt-5.6-terra": { input: 2, output: 12, cached: 0.2, cache_creation: 2.5, reasoning: 12 },
|
||||
"openai/gpt-audio": { input: 2.5, output: 10.0, reasoning: 10.0 },
|
||||
"openai/gpt-audio-mini": { input: 0.6, output: 2.4, reasoning: 2.4 },
|
||||
"openai/gpt-oss-120b": { input: 0.039, output: 0.18, reasoning: 0.18 },
|
||||
"qwen/qwen3-coder-next": { input: 0.12, output: 0.75, cached: 0.06, reasoning: 0.75 },
|
||||
"qwen/qwen3.5-122b-a10b": { input: 0.26, output: 2.08, reasoning: 2.08 },
|
||||
"qwen/qwen3.5-35b-a3b": { input: 0.1625, output: 1.3, reasoning: 1.3 },
|
||||
"qwen/qwen3.5-397b-a17b": { input: 0.39, output: 2.34, reasoning: 2.34 },
|
||||
"qwen/qwen3.5-9b": { input: 0.1, output: 0.15, reasoning: 0.15 },
|
||||
"qwen/qwen3.5-flash": { input: 0.1048, output: 0.4194, reasoning: 0.4194 },
|
||||
"qwen/qwen3.5-plus-02-15": { input: 0.26, output: 1.56, reasoning: 1.56 },
|
||||
"qwen/qwen3.6-plus": { input: 0.54, output: 3.21, reasoning: 3.21 },
|
||||
"qwen/qwen3.7-max": { input: 1.25, output: 3.75, cached: 0.25, reasoning: 3.75 },
|
||||
"qwen/qwen3.7-plus": { input: 0.4, output: 1.6, cached: 0.08, reasoning: 1.6 },
|
||||
"qwen/qwen3.8-max": { input: 2, output: 6, cached: 0.25, cache_creation: 2.5, reasoning: 6 },
|
||||
"qwen3.5-omni-plus": { input: 1.0, output: 5.7143, reasoning: 5.7143 },
|
||||
"qwen3.6-flash": { input: 0.171, output: 1.029, cached: 0.017, cache_creation: 0.214, reasoning: 1.029 },
|
||||
"sakana/fugu-ultra": { input: 5.0, output: 30.0, cached: 0.5, reasoning: 30.0 },
|
||||
"seed-2-0-code-preview-260328": { input: 1.0, output: 6.0, cached: 0.2, cache_creation: 0.008333, reasoning: 6.0 },
|
||||
"seed-2-0-lite-260428": { input: 0.5, output: 4.0, cached: 0.1, cache_creation: 0.008333, reasoning: 4.0 },
|
||||
"seed-2-0-mini-260428": { input: 0.2, output: 0.8, cached: 0.04, cache_creation: 0.00833, reasoning: 0.8 },
|
||||
"seed-2-0-pro-260328": { input: 1.0, output: 6.0, cached: 0.2, cache_creation: 0.008333, reasoning: 6.0 },
|
||||
"stepfun/step-3.5-flash": { input: 0.1, output: 0.3, cached: 0.02, reasoning: 0.3 },
|
||||
"stepfun/step-3.7-flash": { input: 0.2, output: 1.15, cached: 0.04, reasoning: 1.15 },
|
||||
"tencent/hy3-preview": { input: 0.066, output: 0.26, cached: 0.029, reasoning: 0.26 },
|
||||
"x-ai/grok-4.1-fast": { input: 0.2, output: 0.5, cached: 0.05, reasoning: 0.5 },
|
||||
"x-ai/grok-4.20-beta": { input: 2, output: 6, cached: 0.2, reasoning: 6 },
|
||||
"x-ai/grok-4.3": { input: 1.25, output: 2.5, cached: 0.2, reasoning: 2.5 },
|
||||
"x-ai/grok-4.5": { input: 2, output: 6, cached: 0.5, reasoning: 6 },
|
||||
"x-ai/grok-build-0.1": { input: 1.0, output: 2.0, cached: 0.2, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2-flash": { input: 0.1, output: 0.3, cached: 0.01, reasoning: 0.3 },
|
||||
"xiaomi/mimo-v2-omni": { input: 0.4, output: 2.0, cached: 0.08, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2-pro": { input: 1.0, output: 3.0, cached: 0.2, reasoning: 3.0 },
|
||||
"xiaomi/mimo-v2.5": { input: 0.4, output: 2.0, cached: 0.08, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2.5-pro": { input: 1.0, output: 3.0, cached: 0.2, reasoning: 3.0 },
|
||||
"z-ai/glm-4.5-air": { input: 0.13, output: 0.85, cached: 0.025, reasoning: 0.85 },
|
||||
"z-ai/glm-4.6": { input: 0.6, output: 2.2, cached: 0.11, reasoning: 2.2 },
|
||||
"z-ai/glm-4.6v": { input: 0.3, output: 0.9, reasoning: 0.9 },
|
||||
"z-ai/glm-4.7": { input: 0.6, output: 2.2, cached: 0.11, reasoning: 2.2 },
|
||||
"z-ai/glm-5": { input: 1.0, output: 3.2, cached: 0.2, reasoning: 3.2 },
|
||||
"z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 },
|
||||
"z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 },
|
||||
"z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 },
|
||||
"z-ai/glm-5.3-free": { input: 0, output: 0, cached: 0, reasoning: 0 },
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -233,10 +381,11 @@ export function matchPattern(pattern, model) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve pricing for a model using the 3-step fallback chain:
|
||||
* Resolve pricing for a model using the 4-step fallback chain:
|
||||
* 1. PROVIDER_PRICING[provider][model]
|
||||
* 2. MODEL_PRICING[model]
|
||||
* 3. PATTERN_PRICING (glob match)
|
||||
* 2. free namespace (upstream bills $0)
|
||||
* 3. MODEL_PRICING[model]
|
||||
* 4. PATTERN_PRICING (glob match)
|
||||
*
|
||||
* @param {string} provider
|
||||
* @param {string} model
|
||||
@@ -250,12 +399,15 @@ export function getPricingForModel(provider, model) {
|
||||
return PROVIDER_PRICING[provider][model];
|
||||
}
|
||||
|
||||
// 2. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat")
|
||||
// 2. Free namespaces bill $0 regardless of the model name behind them.
|
||||
if (isFreeModel(model)) return ZERO_PRICING;
|
||||
|
||||
// 3. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat")
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
if (MODEL_PRICING[baseModel]) return MODEL_PRICING[baseModel];
|
||||
if (MODEL_PRICING[model]) return MODEL_PRICING[model];
|
||||
|
||||
// 3. Pattern match
|
||||
// 4. Pattern match
|
||||
for (const { pattern, pricing } of PATTERN_PRICING) {
|
||||
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
|
||||
return pricing;
|
||||
|
||||
29
open-sse/providers/registry/agnes.js
Normal file
29
open-sse/providers/registry/agnes.js
Normal file
@@ -0,0 +1,29 @@
|
||||
export default {
|
||||
id: "agnes",
|
||||
priority: 120,
|
||||
alias: "agnes",
|
||||
aliases: [
|
||||
"agnes-ai",
|
||||
],
|
||||
uiAlias: "agnes",
|
||||
display: {
|
||||
name: "Agnes AI",
|
||||
icon: "auto_awesome",
|
||||
color: "#7C3AED",
|
||||
textIcon: "AG",
|
||||
website: "https://agnes-ai.com",
|
||||
notice: {
|
||||
text: "OpenAI-compatible gateway from Agnes AI, offering free API credits on sign-up. Accepts a bearer token or an x-api-key header.",
|
||||
apiKeyUrl: "https://platform.agnes-ai.com",
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions",
|
||||
validateUrl: "https://apihub.agnes-ai.com/v1/models",
|
||||
},
|
||||
// No model ids could be verified without a key, so discovery is left to the
|
||||
// live endpoint and any id is accepted through passthroughModels.
|
||||
passthroughModels: true,
|
||||
};
|
||||
35
open-sse/providers/registry/alitp-intl.js
Normal file
35
open-sse/providers/registry/alitp-intl.js
Normal file
@@ -0,0 +1,35 @@
|
||||
// Token Plan — credit subscription keys on token-plan.<region>.maas.aliyuncs.com.
|
||||
// Fourth Alibaba key type: Coding Plan (alicode/alicode-intl) and Model Studio
|
||||
// (alims-intl) both reject these keys, and they reject Model Studio keys back.
|
||||
// Singapore is the only region that serves the plan; eu-central-1 answers
|
||||
// IllegalEndpoint. The Anthropic surface (/apps/anthropic/v1/messages) is not
|
||||
// authorized for this plan, so OpenAI-compatible mode is the only transport.
|
||||
export default {
|
||||
id: "alitp-intl",
|
||||
priority: 11,
|
||||
alias: "alitp-intl",
|
||||
display: {
|
||||
name: "Alibaba Token Plan",
|
||||
icon: "cloud",
|
||||
color: "#FF6A00",
|
||||
textIcon: "ATP",
|
||||
website: "https://www.alibabacloud.com/campaign/ai-landing-page-token",
|
||||
notice: {
|
||||
apiKeyUrl: "https://modelstudio.console.alibabacloud.com/?apiKey=1",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1/chat/completions",
|
||||
headers: {},
|
||||
quirks: { preserveCacheControl: true },
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3.8-max-preview", name: "Qwen3.8 Max Preview" },
|
||||
{ id: "qwen3.7-max", name: "Qwen3.7 Max" },
|
||||
{ id: "qwen3.7-plus", name: "Qwen3.7 Plus" },
|
||||
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
],
|
||||
};
|
||||
@@ -17,7 +17,7 @@ export default {
|
||||
deprecationNotice: "RISK_NOTICE",
|
||||
},
|
||||
category: "oauth",
|
||||
serviceKinds: ["llm", "image"],
|
||||
serviceKinds: ["llm", "image", "webSearch"],
|
||||
transport: {
|
||||
baseUrls: [ANTIGRAVITY_IDE_BASE_URL],
|
||||
format: "antigravity",
|
||||
@@ -36,8 +36,8 @@ export default {
|
||||
},
|
||||
},
|
||||
usage: {
|
||||
// Discovery (quota/project) on PROD; daily host rejects these.
|
||||
quotaApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels",
|
||||
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
|
||||
quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`,
|
||||
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
|
||||
tokenUrl: "https://oauth2.googleapis.com/token",
|
||||
},
|
||||
@@ -45,6 +45,13 @@ export default {
|
||||
clientSecret: "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",
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.8-flash-high", name: "Gemini 3.8 Flash (High)", upstreamModelId: "gemini-3.8-flash-high(high)" },
|
||||
{ id: "gemini-3.8-flash-medium", name: "Gemini 3.8 Flash (Medium)", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
|
||||
{ id: "gemini-3.8-flash-low", name: "Gemini 3.8 Flash (Low)", upstreamModelId: "gemini-3.8-flash-low(low)" },
|
||||
{ id: "gemini-3.8-flash", name: "Gemini 3.8 Flash", upstreamModelId: "gemini-3.8-flash-medium(medium)" },
|
||||
{ id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" },
|
||||
{ id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" },
|
||||
{ id: "gemini-3.6-flash-high", name: "Gemini 3.6 Flash (High)", upstreamModelId: "gemini-3.6-flash-tiered(high)" },
|
||||
{ id: "gemini-3.6-flash-medium", name: "Gemini 3.6 Flash (Medium)", upstreamModelId: "gemini-3.6-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.6-flash-low", name: "Gemini 3.6 Flash (Low)", upstreamModelId: "gemini-3.6-flash-tiered(low)" },
|
||||
@@ -76,10 +83,14 @@ export default {
|
||||
apiVersion: "v1internal",
|
||||
loadCodeAssistEndpoint: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
|
||||
onboardUserEndpoint: "https://cloudcode-pa.googleapis.com/v1internal:onboardUser",
|
||||
loadCodeAssistUserAgent: "google-api-nodejs-client/9.15.1",
|
||||
loadCodeAssistApiClient: "google-cloud-sdk vscode_cloudshelleditor/0.1",
|
||||
loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT,
|
||||
refreshLeadMs: 300000,
|
||||
},
|
||||
searchViaChat: {
|
||||
defaultModel: "gemini-2.5-flash",
|
||||
endpoint: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:generateContent`,
|
||||
freeTier: "Free — Google Search grounding through an Antigravity OAuth account.",
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
},
|
||||
|
||||
@@ -20,6 +20,8 @@ export default {
|
||||
authModes: [
|
||||
"apikey",
|
||||
],
|
||||
passthroughModels: true,
|
||||
modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" },
|
||||
transport: {
|
||||
baseUrl: "https://api.airforce/v1/chat/completions",
|
||||
validateUrl: "https://api.airforce/v1/models",
|
||||
@@ -27,10 +29,11 @@ export default {
|
||||
"HTTP-Referer": "https://endpoint-proxy.local",
|
||||
"X-Title": "Endpoint Proxy",
|
||||
},
|
||||
forceStream: true,
|
||||
},
|
||||
models: [
|
||||
{ id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 },
|
||||
{ id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 },
|
||||
{ id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 },
|
||||
{ id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 },
|
||||
{ id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 },
|
||||
],
|
||||
};
|
||||
|
||||
32
open-sse/providers/registry/atria.js
Normal file
32
open-sse/providers/registry/atria.js
Normal file
@@ -0,0 +1,32 @@
|
||||
export default {
|
||||
id: "atria",
|
||||
priority: 120,
|
||||
alias: "atria",
|
||||
aliases: [
|
||||
"atria-asi",
|
||||
],
|
||||
uiAlias: "atria",
|
||||
display: {
|
||||
name: "Atria Dawn",
|
||||
icon: "flare",
|
||||
color: "#C2410C",
|
||||
textIcon: "AD",
|
||||
website: "https://atria-asi.ai",
|
||||
notice: {
|
||||
text: "OpenAI-compatible endpoint from Atria Dawn (AtomInnoLab). Currently a research preview offering a single text model, Atria-Dawn-Preview.",
|
||||
apiKeyUrl: "https://api.atria-asi.ai/dashboard",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://api.atria-asi.ai/v1/chat/completions",
|
||||
validateUrl: "https://api.atria-asi.ai/v1/models",
|
||||
},
|
||||
// Docs pin the model field to one case-sensitive id. Text-only for now: the
|
||||
// service ships a hook that blocks image/PDF input, so no vision is claimed.
|
||||
models: [
|
||||
{ id: "Atria-Dawn-Preview", name: "Atria Dawn Preview" },
|
||||
],
|
||||
passthroughModels: true,
|
||||
};
|
||||
30
open-sse/providers/registry/bai.js
Normal file
30
open-sse/providers/registry/bai.js
Normal file
@@ -0,0 +1,30 @@
|
||||
export default {
|
||||
id: "bai",
|
||||
priority: 120,
|
||||
alias: "bai",
|
||||
aliases: [
|
||||
"b-ai",
|
||||
],
|
||||
uiAlias: "bai",
|
||||
display: {
|
||||
name: "B.AI",
|
||||
icon: "account_balance",
|
||||
color: "#0369A1",
|
||||
textIcon: "BA",
|
||||
website: "https://b.ai",
|
||||
notice: {
|
||||
text: "OpenAI-compatible gateway with one of the larger catalogues here. Accepts a bearer token or an x-api-key header. Model ids are fetched live from the provider.",
|
||||
apiKeyUrl: "https://b.ai",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://api.b.ai/v1/chat/completions",
|
||||
validateUrl: "https://api.b.ai/v1/models",
|
||||
},
|
||||
// No ids hardcoded: the catalogue is large and rotates, so the live endpoint
|
||||
// is the source of truth and any id is accepted via passthroughModels.
|
||||
modelsFetcher: { url: "https://api.b.ai/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user