diff --git a/.github/workflows/tray-binaries.yml b/.github/workflows/tray-binaries.yml new file mode 100644 index 00000000..d1d99c98 --- /dev/null +++ b/.github/workflows/tray-binaries.yml @@ -0,0 +1,176 @@ +name: Build macOS tray binary (arm64) + +# systray2 ships only an x86_64 tray_darwin_release, so Apple Silicon users need +# Rosetta 2 for the menubar icon. This builds the native arm64 overlay that +# cli/hooks/trayRuntime.js downloads from the `tray-binaries` release. +# +# Manual-only: the artifact's sha256 is pinned in cli/hooks/trayRuntime.js and +# verified on every download, so a new build is only publishable together with a +# matching pin. Running this with publish=true against a mismatched pin fails +# rather than silently bricking every Apple Silicon client. + +on: + workflow_dispatch: + inputs: + publish: + description: "Upload to the tray-binaries release (requires sha to match ARM64_TRAY_SHA256)" + required: false + default: false + type: boolean + +concurrency: + group: tray-binaries-${{ github.repository }} + cancel-in-progress: false + +permissions: + contents: read + +env: + # Pinned because -trimpath only makes the build reproducible for a given Go + # version and macOS SDK. Bumping this changes the sha256. + GO_VERSION: "1.27.1" + +jobs: + build: + name: Build darwin/arm64 + runs-on: macos-15 + timeout-minutes: 20 + permissions: + contents: write + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-node@v4 + with: + node-version: 22 + + - uses: actions/setup-go@v5 + with: + go-version: ${{ env.GO_VERSION }} + # The Go module lives in a temp clone of the upstream repo, so there is + # no go.sum at the workspace root for setup-go's cache to key on. + cache: false + + - name: Record SDK provenance + run: | + { + echo "runner macOS: $(sw_vers -productVersion)" + echo "Xcode: $(xcodebuild -version | head -1)" + echo "clang: $(clang --version | head -1)" + echo "Go: $(go version)" + } | tee sdk-provenance.txt + + - name: Build + run: node cli/scripts/buildTrayArm64.js + + - name: Compare against pinned checksum + id: sha + run: | + BUILT=$(shasum -a 256 cli/.tray-build/tray_darwin_arm64 | cut -d' ' -f1) + # Whitespace-tolerant, and a missing constant must fail loudly: a null + # match would otherwise surface as an opaque TypeError from [1]. + PINNED=$(node -e ' + const m = require("fs").readFileSync("cli/hooks/trayRuntime.js", "utf8") + .match(/ARM64_TRAY_SHA256\s*=\s*"([0-9a-f]{64})"/); + if (!m) { console.error("::error::ARM64_TRAY_SHA256 not found in cli/hooks/trayRuntime.js"); process.exit(1); } + process.stdout.write(m[1]); + ') + { + echo "built=$BUILT" + echo "pinned=$PINNED" + if [ "$BUILT" = "$PINNED" ]; then echo "match=true"; else echo "match=false"; fi + } >> "$GITHUB_OUTPUT" + + - name: Write job summary + run: | + { + echo "### tray_darwin_arm64" + echo "" + echo "| | |" + echo "|---|---|" + echo "| built sha256 | \`${{ steps.sha.outputs.built }}\` |" + echo "| pinned sha256 | \`${{ steps.sha.outputs.pinned }}\` |" + echo "| match | ${{ steps.sha.outputs.match }} |" + echo "" + echo '```' + cat sdk-provenance.txt + echo '```' + echo "" + if [ "${{ steps.sha.outputs.match }}" = "true" ]; then + echo "Pin already matches — safe to re-run with \`publish=true\`." + else + echo "⚠️ Pin does **not** match. To publish this build, set \`ARM64_TRAY_SHA256\`" + echo "in \`cli/hooks/trayRuntime.js\` to the built sha256 above and land that" + echo "change first. Publishing without it makes every Apple Silicon client fail" + echo "checksum verification and fall back to the Rosetta binary." + fi + } >> "$GITHUB_STEP_SUMMARY" + + # Uploaded before the mismatch gate below, so a publish run that fails on a + # checksum mismatch still leaves the bytes downloadable — that is exactly + # the run where a maintainer needs them to verify the new sha256. + - uses: actions/upload-artifact@v4 + with: + name: tray_darwin_arm64 + path: | + cli/.tray-build/tray_darwin_arm64 + sdk-provenance.txt + + - name: Refuse to publish on checksum mismatch + if: ${{ inputs.publish && steps.sha.outputs.match != 'true' }} + run: | + echo "::error::publish requested but built sha256 != ARM64_TRAY_SHA256" + echo " built: ${{ steps.sha.outputs.built }}" + echo " pinned: ${{ steps.sha.outputs.pinned }}" + echo "Update cli/hooks/trayRuntime.js and land it before publishing." + exit 1 + + - name: Publish to tray-binaries release + if: ${{ inputs.publish && steps.sha.outputs.match == 'true' }} + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + # Clients fetch from the hardcoded ARM64_TRAY_URL, so publishing from a + # different repo would gate an asset stream nobody downloads and silently + # decouple the sha pin from the bytes Apple Silicon users execute. + URL_REPO=$(node -e ' + const m = require("fs").readFileSync("cli/hooks/trayRuntime.js", "utf8") + .match(/"https:\/\/github\.com\/([^\/"]+\/[^\/"]+)\/releases\/download\/tray-binaries\/tray_darwin_arm64"/); + if (!m) { console.error("::error::ARM64_TRAY_URL not found in cli/hooks/trayRuntime.js"); process.exit(1); } + process.stdout.write(m[1]); + ') + if [ "${{ github.repository }}" != "$URL_REPO" ]; then + echo "::error::publishing to ${{ github.repository }}, but ARM64_TRAY_URL points clients at $URL_REPO" + echo "Repoint ARM64_TRAY_URL in cli/hooks/trayRuntime.js at this repo, or run the publish from $URL_REPO." + exit 1 + fi + + # gh release upload does not create the release, so bootstrap it on the + # first publish run rather than failing with "release not found". + if ! gh release view tray-binaries >/dev/null 2>&1; then + echo "Release 'tray-binaries' does not exist yet — creating it" + gh release create tray-binaries --latest=false \ + --title "Native macOS tray binaries" \ + --notes "Built by .github/workflows/tray-binaries.yml. Provenance and the pinned sha256 live in that workflow and in cli/hooks/trayRuntime.js (ARM64_TRAY_SHA256)." + fi + + gh release upload tray-binaries cli/.tray-build/tray_darwin_arm64 --clobber + + echo "Uploaded. Verifying public download URL..." + URL="https://github.com/${{ github.repository }}/releases/download/tray-binaries/tray_darwin_arm64" + GOT="" + for attempt in 1 2 3; do + if curl -fsSL --max-time 60 -o /tmp/verify "$URL"; then + GOT=$(shasum -a 256 /tmp/verify | cut -d' ' -f1) + if [ "$GOT" = "${{ steps.sha.outputs.built }}" ]; then break; fi + fi + # A just-uploaded asset can 404 or serve stale bytes until the CDN catches up. + echo "attempt $attempt: got '${GOT:-}' — retrying in 15s" + sleep 15 + done + if [ "$GOT" != "${{ steps.sha.outputs.built }}" ]; then + echo "::error::downloaded asset sha256 '${GOT:-}' != built ${{ steps.sha.outputs.built }} after 3 attempts" + exit 1 + fi + echo "✅ $URL serves the expected bytes" + diff --git a/CHANGELOG.md b/CHANGELOG.md index 1d6a7d07..fb5643d3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,33 @@ +# v0.5.91 (2026-09-26) + +## Features +- **Providers**: add Token Harbor provider and four OpenAI-compatible aggregator providers (dahl, atria, agnes, bai) +- **Claude**: forward `x-claude-code-session-id` on OAuth requests; merge client `anthropic-beta` flags and forward rate-limit headers; return thinking text to OpenAI-format clients +- **Codex**: add GPT-6 Sol and Luna support +- **CLI Tools**: support multiple model profiles for Codex CLI +- **Hermes**: multi-role model config (delegation + auxiliary slots) +- **OpenCode Go**: complete the Go catalog (40 models) with auto-fetch + family endpoint regex +- **Usage**: show and redeem free limit resets for cc accounts +- **Cline**: expose the `cline-free/*` tier and price it at zero +- **Combos**: display vision adapter models in an ordered table view + +## Fixes +- **Claude**: decloak tool names when `toolNameMap` misses (#4342); update spoofed cli version to 2.1.280 to support Opus 5.5 +- **Providers API**: make POST `/api/providers` O(1) and refuse silent key overwrite (#4350) +- **Capabilities**: stop caching the catalog source per module copy (#4351) +- **OAuth**: stop Zed paste-token crash and add IDE auto-import (#4359) +- **Dashboard**: resolve combo limits with the server's capabilities (#4360); lazy-load charts and `marked`, preload in background on idle +- **Responses**: carry the streamed output items in `response.completed` (#4307) +- **STT**: dispatch live-API-only Gemini models over the Live WebSocket transport (#4006) +- **Gemini**: guard terminal model turns and unresponded functionCalls in `normalizeGeminiContents` +- **Command Code**: replay raw byte chunks to preserve all NDJSON lines +- **Translator**: stop emitting empty `` markers into OpenAI content +- **CLI Tools**: refresh Codex settings after apply (#4347); keep existing `ANTHROPIC_AUTH_TOKEN` when applying Claude settings +- **Tray**: native arm64 macOS menubar binary, no Rosetta required +- **CLI**: filter model selector by active connections and noAuth providers +- **Usage**: key live byApiKey stats by full api key to prevent team-key collision and preserve API key usage attribution +- **Tailscale**: cap enable-flow health wait at 20s + # v0.5.86 (2026-09-23) ## Features diff --git a/CLAUDE.md b/CLAUDE.md index a8d933cb..fef5a486 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -88,6 +88,7 @@ Pre-translate hooks that compress `tool_result` content in-place to cut tokens. - `custom-server.js` wraps the Next standalone server to derive client IP from the TCP socket and strip attacker-controlled `X-Forwarded-For` — trusting forwarding headers only from a loopback reverse proxy. Preserve this when touching request/IP/rate-limit code. - Security-sensitive env: `JWT_SECRET` (session cookie), `INITIAL_PASSWORD` (default `123456` — must override), `API_KEY_SECRET`, `MACHINE_ID_SALT`. Full env contract in `.env.example` and ARCHITECTURE.md's env matrix. - Binary/protobuf upstreams (kiro EventStream, cursor protobuf, commandcode NDJSON) don't round-trip through OpenAI — they're handled inside their own executor, not the translator. +- **Security-first on PRs**: Security is the top priority when reviewing or creating PRs. Audit authentication, credential/token storage & leaks, header manipulation (`X-Forwarded-For`), and SSRF risks before functional logic. Always include explicit security warnings/notes when reporting PR reviews or changes to the user. - Versioning: root and `cli/` are versioned independently; changes are logged in `CHANGELOG.md`. Commit style is Conventional Commits (`fix(translator): …`, `feat(...)`). diff --git a/cli/.gitignore b/cli/.gitignore index 55fd8c49..86262d36 100644 --- a/cli/.gitignore +++ b/cli/.gitignore @@ -1,2 +1,3 @@ app/* node_modules/* +.tray-build/ diff --git a/cli/hooks/trayRuntime.js b/cli/hooks/trayRuntime.js index dafb2154..50e6cb97 100644 --- a/cli/hooks/trayRuntime.js +++ b/cli/hooks/trayRuntime.js @@ -5,9 +5,17 @@ // // We use the maintained `systray2` fork. The original `systray@1.0.5` package // bundles a 2017 x86_64 Go binary whose Mach-O headers are rejected by modern -// dyld (macOS 14+), so the tray silently fails to register on Apple Silicon. +// dyld (macOS 14+), so it fails to load at all. +// +// Note that systray2 is NOT an Apple Silicon fix: like its predecessor it ships +// only an x86_64 `tray_darwin_release`, and picks it by process.platform with no +// process.arch branch, so there is no native slice to select. On arm64 macOS the +// tray therefore needs Rosetta 2 and dies with EBADARCH without it. We overlay +// our own arm64 build of the same upstream source on top — see ensureArm64TrayBin. const { spawnSync } = require("child_process"); +const crypto = require("crypto"); const fs = require("fs"); +const os = require("os"); const path = require("path"); const { getRuntimeDir, getRuntimeNodeModules, runNpmInstall, summarizeNpmError } = require("./sqliteRuntime"); @@ -15,6 +23,19 @@ const SYSTRAY_PKG = "systray2"; const SYSTRAY_VERSION = "2.1.4"; const LEGACY_SYSTRAY_PKG = "systray"; +// Pinned `tray-binaries` release rather than `latest`, so the URL is stable and +// the artifact can only change by a deliberate re-upload. The workflow's publish +// step re-derives this repo from the literal below and refuses to upload +// anywhere else, so the integrity gate can't drift from what clients fetch. +// +// The asset is built by .github/workflows/tray-binaries.yml on a macos-15 runner. +// cgo compiles AppKit against the runner's SDK, so this value tracks that image: +// when GitHub updates it the sha changes, the workflow refuses to publish, and +// this constant must be bumped in the same change as the re-upload. +const ARM64_TRAY_URL = "https://github.com/decolua/9router/releases/download/tray-binaries/tray_darwin_arm64"; +const ARM64_TRAY_SHA256 = "487e3c365aaa1eb6ad295bf3989711e975b52cee07505bf641c8559954881c81"; +const ARM64_RETRY_COOLDOWN_MS = 24 * 60 * 60 * 1000; + function hasSystray() { return fs.existsSync(path.join(getRuntimeNodeModules(), SYSTRAY_PKG, "package.json")); } @@ -71,6 +92,124 @@ function ensureRuntimeDir() { return dir; } +// A thin (non-fat) 64-bit Mach-O stores its magic then cputype, both LE. +// CPU_TYPE_ARM64 is CPU_TYPE_ARM | CPU_ARCH_ABI64. Fat/universal binaries use a +// different magic and are reported as "not arm64" here, which is fine: we only +// ever overlay a thin arm64 build and only need to tell it apart from x86_64. +function isArm64MachO(file) { + let fd = null; + try { + fd = fs.openSync(file, "r"); + const buf = Buffer.alloc(8); + fs.readSync(fd, buf, 0, 8, 0); + if (buf.readUInt32LE(0) !== 0xfeedfacf) return false; + return buf.readUInt32LE(4) === 0x0100000c; + } catch { + return false; + } finally { + if (fd !== null) try { fs.closeSync(fd); } catch {} + } +} + +// systray2 is constructed with copyDir:true, so what actually executes is +// ~/.cache/node-systray//tray_darwin_release, and index.js only re-copies +// when that path is absent. An overlaid binary stays invisible until this is cleared. +// Scoped to our systray2 version: the parent dir is machine-global and shared +// with any other node-systray consumer. +function bustSystrayCopyCache() { + try { + fs.rmSync(path.join(os.homedir(), ".cache", "node-systray", SYSTRAY_VERSION), { recursive: true, force: true }); + } catch {} +} + +function arm64AttemptMarker() { + return path.join(getRuntimeDir(), ".tray-arm64-attempt"); +} + +// ensureTrayRuntime runs synchronously on every `9router` start (cli.js), so a +// failed download must not re-block the next launch. Retry at most daily. +function recentlyAttemptedArm64() { + try { + const at = Number(fs.readFileSync(arm64AttemptMarker(), "utf8").trim()); + return Number.isFinite(at) && Date.now() - at < ARM64_RETRY_COOLDOWN_MS; + } catch { + return false; + } +} + +function markArm64Attempt() { + try { fs.writeFileSync(arm64AttemptMarker(), String(Date.now())); } catch {} +} + +// Cleared on success so the cooldown only ever throttles *failures*. Without +// this, anything that restores systray2's x86_64 binary later — notably a +// globally installed 9router older than this change, which shares the same +// ~/.9router/runtime — would leave the user waiting out the cooldown. +function clearArm64Attempt() { + try { fs.rmSync(arm64AttemptMarker(), { force: true }); } catch {} +} + +// Throws on any failure so the caller has a single error path. +function downloadFile(url, dest, timeoutSec) { + // darwin-only path, and curl ships with macOS, so this needs no extra dep and + // keeps the caller synchronous. + const res = spawnSync("curl", ["-fsSL", "--max-time", String(timeoutSec), "-o", dest, url], { + encoding: "utf8", + timeout: (timeoutSec + 5) * 1000 + }); + if (res.status === 0 && fs.existsSync(dest)) return; + const detail = (res.stderr || res.error?.message || `curl exit ${res.status}`).trim().split("\n").pop(); + throw new Error(detail || "download failed"); +} + +function sha256File(file) { + return crypto.createHash("sha256").update(fs.readFileSync(file)).digest("hex"); +} + +// Replace systray2's x86_64 macOS binary with a native arm64 build so Apple +// Silicon users get a tray without installing Rosetta 2. Any failure leaves the +// Intel binary untouched, which still works under Rosetta. +// +// Takes no `silent` flag on purpose: cli.js calls ensureTrayRuntime({silent:true}) +// synchronously on every start, and a stalled curl would otherwise freeze the +// launch for up to 30s with no output at all. These lines print at most once per +// 24h on failure and once ever on success, so they are worth more than the quiet. +function ensureArm64TrayBin() { + if (process.platform !== "darwin" || process.arch !== "arm64") return { skipped: true }; + + const binPath = path.join(getRuntimeNodeModules(), SYSTRAY_PKG, "traybin", "tray_darwin_release"); + if (!fs.existsSync(binPath)) return { skipped: true }; + if (isArm64MachO(binPath)) return { native: true }; + if (recentlyAttemptedArm64()) return { deferred: true }; + + markArm64Attempt(); + console.log("⏳ Downloading native Apple Silicon tray binary..."); + // pid-scoped: two concurrent starts (postinstall racing cli.js, or two + // terminals) would otherwise interleave writes to one file, fail each other's + // checksum, and delete each other's in-flight download from the catch below. + const tmp = `${binPath}.arm64.${process.pid}.tmp`; + try { + downloadFile(ARM64_TRAY_URL, tmp, 30); + const sum = sha256File(tmp); + // Integrity matters more than usual: this is an executable that runs on + // every Apple Silicon user's machine. + if (sum !== ARM64_TRAY_SHA256) throw new Error(`checksum mismatch (got ${sum.slice(0, 12)}…)`); + if (!isArm64MachO(tmp)) throw new Error("downloaded file is not an arm64 Mach-O"); + fs.chmodSync(tmp, 0o755); + fs.renameSync(tmp, binPath); + bustSystrayCopyCache(); + clearArm64Attempt(); + console.log("✅ Native Apple Silicon tray installed"); + return { native: true, installed: true }; + } catch (e) { + try { fs.rmSync(tmp, { force: true }); } catch {} + console.warn("⚠️ Native tray download failed — falling back to the Intel binary"); + console.warn(` Reason: ${e.message}`); + console.warn(" The Intel tray needs Rosetta 2: softwareupdate --install-rosetta --agree-to-license"); + return { native: false, error: e.message }; + } +} + function npmInstall(pkgs, { silent = false } = {}) { const cwd = ensureRuntimeDir(); if (!silent) console.log("⏳ Installing system tray (first run)..."); @@ -94,14 +233,20 @@ function ensureTrayRuntime({ silent = false } = {}) { if (process.platform === "win32") { return { systray: false, skipped: true }; } - if (hasSystray()) { + + let ready = hasSystray(); + if (!ready) { + ready = npmInstall([`${SYSTRAY_PKG}@${SYSTRAY_VERSION}`], { silent }) && hasSystray(); + } + if (ready) { chmodSystrayBin({ silent }); if (!silent) console.log("✅ System tray ready"); - return { systray: true }; } - const ok = npmInstall([`${SYSTRAY_PKG}@${SYSTRAY_VERSION}`], { silent }); - if (ok) chmodSystrayBin({ silent }); - return { systray: ok && hasSystray() }; + + // Runs after the ready log so a download failure doesn't read as a broken + // tray — the Intel binary still works under Rosetta. + const arm64 = ready ? ensureArm64TrayBin() : { skipped: true }; + return { systray: ready, arm64 }; } -module.exports = { ensureTrayRuntime }; +module.exports = { ensureTrayRuntime, ensureArm64TrayBin }; diff --git a/cli/package.json b/cli/package.json index f7a0c677..1a67862a 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.86", + "version": "0.5.91", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" @@ -16,6 +16,7 @@ "scripts": { "dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js", "build": "node scripts/build-cli.js", + "build:tray-arm64": "node scripts/buildTrayArm64.js", "pack:cli": "npm run build && npm pack --pack-destination ..", "publish:cli": "npm run build && npm publish", "postinstall": "node hooks/postinstall.js", @@ -29,7 +30,8 @@ "react-dom": "19.2.1" }, "comment_sqlite": "sql.js + better-sqlite3 are NOT bundled here. They are installed into ~/.9router/runtime/node_modules by hooks/postinstall.js (and re-checked at runtime by cli.js). This avoids Windows EBUSY errors when updating the global CLI, since native .node files no longer live under the locked install dir.", - "comment_systray": "systray2 is NOT bundled here. It is lazy-installed into ~/.9router/runtime/node_modules by hooks/postinstall.js on macOS/Linux only. Windows uses PowerShell NotifyIcon (zero binary). This avoids shipping unsigned Go binaries that trigger antivirus false positives (Kaspersky). We use the systray2 fork because the legacy systray@1.0.5 ships a 2017 x86_64 binary that fails on modern macOS dyld.", + "comment_systray": "systray2 is NOT bundled here. It is lazy-installed into ~/.9router/runtime/node_modules by hooks/postinstall.js on macOS/Linux only. Windows uses PowerShell NotifyIcon (zero binary). This avoids shipping unsigned Go binaries that trigger antivirus false positives (Kaspersky). We use the systray2 fork because the legacy systray@1.0.5 ships a 2017 x86_64 binary that fails to load on modern macOS dyld. Neither package ships an arm64 macOS binary, so on Apple Silicon hooks/trayRuntime.js overlays our own arm64 build from the tray-binaries GitHub release; without it the tray requires Rosetta 2.", + "comment_tray_arm64": "tray_darwin_arm64 is built by scripts/buildTrayArm64.js (npm run build:tray-arm64) from felixhao28/systray-portable — the same source systray2's binary comes from — and uploaded to the pinned 'tray-binaries' GitHub release. Its sha256 is pinned as ARM64_TRAY_SHA256 in hooks/trayRuntime.js and verified after every download; rebuild and update both together. .github/workflows/tray-binaries.yml builds the same artifact in CI but refuses to publish when the sha diverges from the pin.", "engines": { "node": ">=18.0.0" }, diff --git a/cli/scripts/buildTrayArm64.js b/cli/scripts/buildTrayArm64.js new file mode 100644 index 00000000..d32990da --- /dev/null +++ b/cli/scripts/buildTrayArm64.js @@ -0,0 +1,108 @@ +#!/usr/bin/env node + +// Rebuilds tray_darwin_arm64, the native Apple Silicon menubar binary that +// hooks/trayRuntime.js overlays on top of systray2's x86_64-only build. +// +// Must run on macOS: getlantern/systray is cgo against AppKit, so the arm64 +// slice needs a real macOS SDK. Requires Go on PATH (`mise use -g go@latest`). +// +// Output is NOT reproducible across Go versions even with -s -w, so after a +// rebuild you must re-upload the asset and update ARM64_TRAY_SHA256 in +// hooks/trayRuntime.js — this script prints both and fails if they diverge. + +const { execFileSync, spawnSync } = require("child_process"); +const crypto = require("crypto"); +const fs = require("fs"); +const os = require("os"); +const path = require("path"); + +const UPSTREAM_REPO = "https://github.com/felixhao28/systray-portable.git"; +// master as of 2021-09-15, the commit systray2@2.1.4's own binary was built from. +const UPSTREAM_COMMIT = "6eddc917bf39fcc0d95b57a0741d0c065fbd1e23"; + +const outDir = path.join(__dirname, "..", ".tray-build"); +const outFile = path.join(outDir, "tray_darwin_arm64"); +const trayRuntimePath = path.join(__dirname, "..", "hooks", "trayRuntime.js"); + +function fail(msg) { + console.error(`\n❌ ${msg}`); + process.exit(1); +} + +function run(cmd, args, opts = {}) { + const res = spawnSync(cmd, args, { stdio: "inherit", ...opts }); + if (res.status !== 0) fail(`${cmd} ${args.join(" ")} exited with ${res.status}`); +} + +if (process.platform !== "darwin") fail("must be run on macOS (cgo needs the AppKit SDK)"); + +const go = spawnSync("go", ["version"], { encoding: "utf8" }); +if (go.status !== 0) { + fail("Go toolchain not found on PATH. Install it with: mise use -g go@latest"); +} +console.log(`Go: ${go.stdout.trim()}`); + +const srcDir = fs.mkdtempSync(path.join(os.tmpdir(), "systray-portable-")); +// fail() exits through process.exit(), which does not unwind the stack, so a +// try/finally here would leak the clone on every failed build. An exit handler +// covers normal completion, fail(), and uncaught exceptions alike. +process.on("exit", () => { + try { fs.rmSync(srcDir, { recursive: true, force: true }); } catch {} +}); + +console.log(`\nCloning ${UPSTREAM_REPO} @ ${UPSTREAM_COMMIT.slice(0, 7)}`); +run("git", ["clone", "--quiet", UPSTREAM_REPO, srcDir]); +run("git", ["-C", srcDir, "checkout", "--quiet", UPSTREAM_COMMIT]); + +const head = execFileSync("git", ["-C", srcDir, "log", "-1", "--format=%H %ad %s", "--date=short"], { + encoding: "utf8" +}).trim(); +console.log(`HEAD: ${head}`); + +run("go", ["mod", "download"], { cwd: srcDir }); + +console.log("\nBuilding darwin/arm64..."); +// -trimpath strips the local build directory from the binary, so two builds +// from the same commit + Go version hash identically regardless of where they +// ran. Without it the pinned sha256 could never be regenerated. +run("go", ["build", "-trimpath", "-ldflags", "-s -w", "-o", outFile, "tray.go"], { + cwd: srcDir, + env: { ...process.env, CGO_ENABLED: "1", GOOS: "darwin", GOARCH: "arm64" } +}); + +// ── Verify ──────────────────────────────────────────────────────────────── +const buf = Buffer.alloc(8); +const fd = fs.openSync(outFile, "r"); +fs.readSync(fd, buf, 0, 8, 0); +fs.closeSync(fd); +if (buf.readUInt32LE(0) !== 0xfeedfacf || buf.readUInt32LE(4) !== 0x0100000c) { + fail("output is not a thin arm64 Mach-O"); +} + +// Apple Silicon refuses to execute an unsigned binary. Go's linker applies an +// ad-hoc signature automatically; confirm it survived. +const sig = spawnSync("codesign", ["--verify", outFile], { encoding: "utf8" }); +if (sig.status !== 0) fail(`ad-hoc signature invalid: ${(sig.stderr || "").trim()}`); + +const sha256 = crypto.createHash("sha256").update(fs.readFileSync(outFile)).digest("hex"); +const sizeMb = (fs.statSync(outFile).size / 1024 / 1024).toFixed(2); + +console.log(`\n✅ ${outFile}`); +console.log(` arch: arm64 (ad-hoc signed, verified)`); +console.log(` size: ${sizeMb} MB`); +console.log(` sha256: ${sha256}`); + +// Whitespace-tolerant: a formatter could wrap the assignment across lines, and +// a null match must fail loudly rather than silently read as "no pin". +const pinMatch = fs.readFileSync(trayRuntimePath, "utf8").match(/ARM64_TRAY_SHA256\s*=\s*"([0-9a-f]{64})"/); +if (!pinMatch) fail(`could not find ARM64_TRAY_SHA256 in ${trayRuntimePath}`); +const pinned = pinMatch[1]; +if (pinned === sha256) { + console.log(`\n Matches ARM64_TRAY_SHA256 in hooks/trayRuntime.js — no code change needed.`); +} else { + console.log(`\n⚠️ Differs from ARM64_TRAY_SHA256 in hooks/trayRuntime.js (${pinned}).`); + console.log(` Re-upload the release asset, then update that constant to the sha256 above.`); +} + +console.log(`\nNext: upload to the pinned release tag, keeping the asset name stable:`); +console.log(` gh release upload tray-binaries "${outFile}" --clobber`); diff --git a/cli/src/cli/tray/tray.js b/cli/src/cli/tray/tray.js index 6658e948..0d643f29 100644 --- a/cli/src/cli/tray/tray.js +++ b/cli/src/cli/tray/tray.js @@ -141,11 +141,14 @@ function initWindowsTray(options) { /** * macOS/Linux tray via systray binary * - * Prefers `systray2` (active fork of `systray`, ships newer - * getlantern/systray-portable binaries that work on macOS 14+ and Apple - * Silicon under Rosetta). Falls back to legacy `systray@1.0.5` if systray2 - * is not available, though that binary's Mach-O headers are rejected by - * modern dyld and the icon will not appear. + * Prefers `systray2`, the active fork of `systray`. Both ship only an x86_64 + * `tray_darwin_release` and select it by process.platform alone, so on Apple + * Silicon the tray runs under Rosetta 2 and fails with EBADARCH when Rosetta is + * absent. hooks/trayRuntime.js overlays a native arm64 build over that file to + * avoid the dependency; the fallbacks below are Intel-only. + * + * Falls back to legacy `systray@1.0.5` if systray2 is unavailable, though that + * binary's Mach-O headers are rejected by modern dyld and no icon will appear. */ function resolveSystray() { let runtimeDir = null; diff --git a/cli/src/cli/utils/modelSelector.js b/cli/src/cli/utils/modelSelector.js index a2bd3fdc..cec99deb 100644 --- a/cli/src/cli/utils/modelSelector.js +++ b/cli/src/cli/utils/modelSelector.js @@ -2,22 +2,24 @@ const api = require("../api/client"); const { prompt } = require("./input"); const { clearScreen } = require("./display"); -// Provider alias order: OAuth first, then API Key (matches ModelSelectModal) +// Provider alias order: OAuth first, then Free, then API Key const PROVIDER_ALIAS_ORDER = [ - "cc", "ag", "cx", "if", "qw", "gc", "gh", "kr", + "cc", "ag", "cx", "if", "qw", "gc", "gh", "kr", "oc", "openrouter", "glm", "kimi", "minimax", "openai", "anthropic", "gemini" ]; // Alias to display name mapping const PROVIDER_ALIAS_NAMES = { cc: "Claude Code", - ag: "Antigravity", + ag: "Antigravity", cx: "OpenAI Codex", if: "iFlow AI", qw: "Qwen Code", gc: "Gemini CLI", gh: "GitHub Copilot", kr: "Kiro AI", + oc: "OpenCode Free", + opencode: "OpenCode Free", openrouter: "OpenRouter", glm: "GLM Coding", kimi: "Kimi Coding", @@ -27,30 +29,78 @@ const PROVIDER_ALIAS_NAMES = { gemini: "Gemini" }; +const PROVIDER_ID_TO_ALIAS = { + claude: "cc", + codex: "cx", + "gemini-cli": "gc", + github: "gh", + antigravity: "ag", + iflow: "if", + qwen: "qw", + kiro: "kr", + cursor: "cu", + cline: "cline", + clinepass: "clinepass", + qoder: "qd", + "qoder-cn": "qd", + gitlab: "gitlab", + "codebuddy-cn": "cb", + "codebuddy-intl": "cbai", + kimchi: "kimchi", + "grok-cli": "grok-cli", + trae: "trae", + windsurf: "windsurf", + zed: "zed", + opencode: "oc", + "opencode-go": "ocg", + "opencode-zen": "ocz", +}; + +// Providers usable without stored credentials +const NO_AUTH_PROVIDERS = new Set(["opencode", "oc"]); + /** - * Get all available models grouped by provider + combos + * Get all available models grouped by provider + combos (filtered by active connections) * @returns {Promise<{combos: Array, groups: Object}>} */ async function getAvailableModelsGrouped() { - const result = await api.getAvailableModels(); - if (!result.success) return { combos: [], groups: {} }; - - const models = result.data?.data || []; + const [modelsResult, providersResult] = await Promise.all([ + api.getAvailableModels(), + api.getProviders() + ]); + + if (!modelsResult.success) return { combos: [], groups: {} }; + + const connections = providersResult.success ? (providersResult.data?.connections || []) : []; + const activeAliases = new Set(NO_AUTH_PROVIDERS); + + connections.forEach(conn => { + if (conn.isActive === false) return; + const p = conn.provider; + if (!p) return; + activeAliases.add(p); + const alias = conn.providerSpecificData?.prefix || PROVIDER_ID_TO_ALIAS[p] || p; + activeAliases.add(alias); + }); + + const models = modelsResult.data?.data || []; const combos = []; const groups = {}; - + models.forEach(m => { if (m.owned_by === "combo") { combos.push(m.id); } else { const provider = m.owned_by; + // Only keep connected providers or noAuth providers + if (!activeAliases.has(provider)) return; if (!groups[provider]) { groups[provider] = []; } groups[provider].push(m.id); } }); - + return { combos, groups }; } @@ -68,6 +118,19 @@ async function selectModelFromList(title, currentValue = "", options = {}) { const totalModels = combos.length + Object.values(groups).flat().length; if (totalModels === 0) { + clearScreen(); + console.log(`\n🎯 ${title}`); + console.log("=".repeat(50)); + console.log("\n No connected providers found."); + console.log(" Please connect a provider in Providers menu first.\n"); + console.log(" m. ✍️ Enter custom model ID"); + console.log(" 0. Cancel\n"); + const act = await prompt("Select option (m/0): "); + const trimmed = act.trim(); + if (trimmed.toLowerCase() === "m") { + const custom = await prompt("Enter custom model ID: "); + return custom.trim() || null; + } return null; } diff --git a/open-sse/config/appConstants.js b/open-sse/config/appConstants.js index 62ddfb6b..ed627535 100644 --- a/open-sse/config/appConstants.js +++ b/open-sse/config/appConstants.js @@ -178,7 +178,7 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C // makes the backend flag the request and answer 429 Quota Exhausted. export const ANTIGRAVITY_PROMPT_REWRITES = [ { from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" }, - { from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." }, + { from: /You are Hermes(?: Agent)?(?:,\s*(?:an intelligent AI assistant|an AI assistant|an AI agent))?(?:,?\s*(?:built|created)\s+by\s+Nous Research)?\./gi, to: "You are an AI assistant." }, // Claude Code prepends this line to its system prompt. The Claude-format translator strips it, // but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions) // pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED. diff --git a/open-sse/config/providerModels.js b/open-sse/config/providerModels.js index f4747cd4..9427b598 100644 --- a/open-sse/config/providerModels.js +++ b/open-sse/config/providerModels.js @@ -3,10 +3,13 @@ import REGISTRY from "../providers/registry/index.js"; // PROVIDER_MODELS now built from providers/registry (transport + models co-located) import { PROVIDER_MODELS } from "../providers/index.js"; import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js"; -import { CODEX_REVIEW_SUFFIX, isMuseSparkModel } from "../providers/models/helpers.js"; +import { CODEX_REVIEW_SUFFIX, isMuseSparkModel, opencodeFamilyFormats } from "../providers/models/helpers.js"; import { FORMATS } from "../translator/formats.js"; export { PROVIDER_MODELS }; +// OpenCode providers sharing the endpoint-family fallback for unknown model ids +const isOpenCodeAlias = (aliasOrId) => !aliasOrId || ["oc", "opencode", "ocg", "opencode-go", "ocz", "opencode-zen"].includes(aliasOrId); + // Helper functions export function getProviderModels(aliasOrId) { @@ -53,20 +56,29 @@ export function findModelName(aliasOrId, modelId) { } export function getModelTargetFormat(aliasOrId, modelId) { - if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go" || aliasOrId === "ocz" || aliasOrId === "opencode-zen") && isMuseSparkModel(modelId)) { + if (isOpenCodeAlias(aliasOrId) && isMuseSparkModel(modelId)) { return FORMATS.OPENAI_RESPONSES; } const models = PROVIDER_MODELS[aliasOrId]; if (!models) return null; - return modelTargetFormat(findModel(models, modelId, aliasOrId)); + const found = findModel(models, modelId, aliasOrId); + if (found) return modelTargetFormat(found); + // Family fallback keeps modelsFetcher/passthrough ids on their endpoint lane + if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.targetFormat || null; + return null; } // Declared upstream formats for a model (registry `supportedFormats`). Drives the // per-model guard on the sourceFormat-matched transport; null when undeclared. +// Unknown OpenCode ids fall back to the family regex (chat lane by default) so +// auto-fetched models never wrongly use the sourceFormat-matched transport. export function getModelSupportedFormats(aliasOrId, modelId) { const models = PROVIDER_MODELS[aliasOrId]; if (!models) return null; - return modelSupportedFormats(findModel(models, modelId, aliasOrId)); + const found = findModel(models, modelId, aliasOrId); + if (found) return modelSupportedFormats(found); + if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.supportedFormats || [FORMATS.OPENAI]; + return null; } export function getModelType(aliasOrId, modelId) { diff --git a/open-sse/executors/codex.js b/open-sse/executors/codex.js index de2af822..8b8c5f79 100644 --- a/open-sse/executors/codex.js +++ b/open-sse/executors/codex.js @@ -7,7 +7,7 @@ import { } from "../services/oauthCredentialManager.js"; import { normalizeResponsesInput } from "../translator/formats/responsesApi.js"; import { fetchImageAsBase64 } from "../translator/concerns/image.js"; -import { getModelUpstreamId } from "../config/providerModels.js"; +import { getModelUpstreamId, getProviderModels } from "../config/providerModels.js"; import { getThinkingLevels } from "../providers/thinkingLevels.js"; import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js"; import { dbg } from "../utils/debugLog.js"; @@ -25,6 +25,10 @@ const CODEX_SSE_USER_OUTPUT_PATTERNS = [ ]; const CODEX_SSE_PEEK_BYTES = 256 * 1024; const CODEX_MODEL_CAPACITY_MESSAGE = "Selected model is at capacity. Please try a different model."; +function isCodexResponsesLiteModel(model) { + const baseId = String(model || "").replace(/\([^()]+\)\s*$/, ""); + return getProviderModels("cx").some((entry) => entry.id === baseId && entry.responsesLite === true); +} // Server-generated item id prefixes that Codex /responses cannot resolve when store=false const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/; @@ -43,7 +47,7 @@ const CODEX_PASSTHROUGH_TOOL_TYPES = new Set(["custom"]); const RESPONSES_API_ALLOWLIST = new Set([ "model", "input", "instructions", "tools", "tool_choice", "stream", "store", "reasoning", "service_tier", "include", "prompt_cache_key", "client_metadata", - "text" + "text", "parallel_tool_calls" ]); // Convert role=system → role=developer in body.input (keeps content in cacheable prefix) @@ -57,13 +61,14 @@ function convertSystemToDeveloperRole(body) { } // Strip server-generated item IDs (rs_/fc_/resp_/msg_) from input — avoids 404 with store=false -function stripStoredItemReferences(body) { +function stripStoredItemReferences(body, preserveLitePrefix = false) { if (!Array.isArray(body.input)) return; body.input = body.input.filter((item) => { if (typeof item === "string" && SERVER_ID_PATTERN.test(item)) return false; if (item && typeof item === "object" && !Array.isArray(item)) { if (item.type === "item_reference") return false; - if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)) delete item.id; + if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id) + && !(preserveLitePrefix && item.role === "developer" && item.id.startsWith("msg_"))) delete item.id; } return true; }); @@ -138,6 +143,7 @@ function resolveCacheSessionId(body, credentials) { function normalizeReasoningEffort(model, value) { const supportedLevels = getThinkingLevels("codex", model); if (supportedLevels?.includes(value)) return value; + if (isCodexResponsesLiteModel(model) && (value === "none" || value === "minimal")) return "low"; if (value === "ultra" && supportedLevels?.includes("max")) return "max"; if (value === "max" || value === "ultra") return "xhigh"; return value; @@ -209,8 +215,11 @@ export class CodexExecutor extends BaseExecutor { * Override headers to add codex-specific identity headers. * transformRequest runs BEFORE buildHeaders, sets this._currentSessionId. */ - buildHeaders(credentials, stream = true) { + buildHeaders(credentials, stream = true, _url = null, model = null) { const headers = super.buildHeaders(credentials, stream); + if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model))) { + headers["x-openai-internal-codex-responses-lite"] = "true"; + } headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default"; // Identify client type to Codex backend (matches official codex CLI) if (!headers["originator"]) headers["originator"] = "codex_cli_rs"; @@ -408,6 +417,8 @@ export class CodexExecutor extends BaseExecutor { // Convert string input to array format (Codex API requires input as array) const normalized = normalizeResponsesInput(body.input); if (normalized) body.input = normalized; + const upstreamModel = getModelUpstreamId("cx", body.model || model); + const responsesLite = isCodexResponsesLiteModel(upstreamModel); // Ensure input is present and non-empty (Codex API rejects empty input) if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) { @@ -417,7 +428,7 @@ export class CodexExecutor extends BaseExecutor { // Keep system prompts in body.input as role=developer so they stay in the cacheable prefix convertSystemToDeveloperRole(body); // Strip server-generated item IDs (rs_/fc_/resp_/msg_) — Codex /responses can't resolve when store=false - stripStoredItemReferences(body); + stripStoredItemReferences(body, responsesLite); // Flatten function tools + drop unsupported types normalizeCodexTools(body); @@ -425,7 +436,7 @@ export class CodexExecutor extends BaseExecutor { body.stream = true; // If no instructions provided, inject default Codex instructions - if (!body.instructions || body.instructions.trim() === "") { + if (!responsesLite && (!body.instructions || body.instructions.trim() === "")) { body.instructions = CODEX_DEFAULT_INSTRUCTIONS; } @@ -438,7 +449,29 @@ export class CodexExecutor extends BaseExecutor { } // Map virtual Codex review models to the upstream Codex model before suffix parsing. - body.model = getModelUpstreamId("cx", body.model || model); + body.model = upstreamModel; + + if (responsesLite) { + // Codex 0.155 carries tools and instructions as input prefix items. + const input = Array.isArray(body.input) ? body.input : [body.input]; + const hasLitePrefix = input.some((item) => item?.type === "additional_tools"); + if (!hasLitePrefix) { + const instructions = typeof body.instructions === "string" && body.instructions.trim() + ? body.instructions : CODEX_DEFAULT_INSTRUCTIONS; + const prefix = [{ type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] }]; + if (instructions) { + prefix.push({ type: "message", role: "developer", content: [{ type: "input_text", text: instructions }] }); + } + input.unshift(...prefix); + } + body.input = input; + body.instructions = ""; + body.tools = null; + body.tool_choice ||= "auto"; + body.parallel_tool_calls = false; + } else { + delete body.parallel_tool_calls; + } // Extract thinking level from model name suffix // e.g., gpt-5.3-codex-high → high, gpt-5.3-codex → medium (default) @@ -455,12 +488,13 @@ export class CodexExecutor extends BaseExecutor { // Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium) if (!body.reasoning) { - const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || 'low'); - body.reasoning = { effort, summary: "auto" }; + const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || (responsesLite ? 'medium' : 'low')); + body.reasoning = responsesLite ? { effort } : { effort, summary: "auto" }; } else { body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort); - if (!body.reasoning.summary) body.reasoning.summary = "auto"; + if (!responsesLite && !body.reasoning.summary) body.reasoning.summary = "auto"; } + if (responsesLite) body.reasoning.context = "all_turns"; delete body.reasoning_effort; // Include reasoning encrypted content (required by Codex backend for reasoning models) diff --git a/open-sse/executors/commandcode.js b/open-sse/executors/commandcode.js index 21aac514..7cb5afdf 100644 --- a/open-sse/executors/commandcode.js +++ b/open-sse/executors/commandcode.js @@ -148,7 +148,7 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model) const reader = originalResponse.body.getReader(); const decoder = new TextDecoder(); let buffer = ""; - const bufferedLines = []; + const rawChunks = []; let detectedError = null; try { @@ -162,16 +162,15 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model) const parsed = JSON.parse(jsonStr); if (parsed?.type === "error") { detectedError = parsed; - } else { - bufferedLines.push(trimmed); } } catch { - bufferedLines.push(trimmed); + /* ignore */ } } break; } + rawChunks.push(value); buffer += decoder.decode(value, { stream: true }); const lines = buffer.split("\n"); buffer = lines.pop() || ""; @@ -182,7 +181,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model) if (!trimmed) continue; const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed; if (!jsonStr || jsonStr === "[DONE]") { - bufferedLines.push(trimmed); stopLoop = true; break; } @@ -191,7 +189,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model) try { event = JSON.parse(jsonStr); } catch { - bufferedLines.push(trimmed); continue; } @@ -201,8 +198,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model) break; } - bufferedLines.push(trimmed); - if ( event?.type === "text-delta" || event?.type === "reasoning-delta" || @@ -245,29 +240,18 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model) ); } - const combinedStream = createReplayedStream(bufferedLines, buffer, reader); + const combinedStream = createRawReplayedStream(rawChunks, reader); return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse); } -function createReplayedStream(bufferedLines, remainingBuffer, reader) { - const encoder = new TextEncoder(); - let replayed = false; +function createRawReplayedStream(rawChunks, reader) { + let chunkIndex = 0; return new ReadableStream({ async pull(controller) { - if (!replayed) { - replayed = true; - let prefix = bufferedLines.join("\n"); - if (prefix && remainingBuffer) { - prefix += "\n" + remainingBuffer; - } else if (remainingBuffer) { - prefix = remainingBuffer; - } else if (prefix) { - prefix += "\n"; - } - if (prefix) { - controller.enqueue(encoder.encode(prefix)); - } + if (chunkIndex < rawChunks.length) { + controller.enqueue(rawChunks[chunkIndex++]); + return; } try { diff --git a/open-sse/executors/default.js b/open-sse/executors/default.js index 64ad46d0..e5c59741 100644 --- a/open-sse/executors/default.js +++ b/open-sse/executors/default.js @@ -1,12 +1,13 @@ import { BaseExecutor } from "./base.js"; import { PROVIDERS, PROVIDER_OAUTH } from "../config/providers.js"; -import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta } from "../providers/shared.js"; +import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta, mergeAnthropicBeta } from "../providers/shared.js"; import { resolveOpenAICompatibleApiType } from "../services/provider.js"; import { OAUTH_ENDPOINTS, buildKimiHeaders } from "../config/appConstants.js"; import { buildClineHeaders } from "../shared/clineAuth.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { injectReasoningContent } from "../utils/reasoningContentInjector.js"; import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js"; +import { extractClaudeSessionIdFromUserId } from "../utils/claudeCloaking.js"; // Auth header descriptors — derived from registry transport.auth, fallback to hardcoded defaults. const BEARER = { combined: true, header: "Authorization", scheme: "bearer" }; @@ -164,9 +165,21 @@ export class DefaultExecutor extends BaseExecutor { // a node fronting Kimi or GLM answers on its own ids and never matches, so // gateways that would choke on unknown beta flags are left untouched. const isClaudeModel = typeof model === "string" && /^claude-/.test(model); + const clientBeta = credentials?.rawHeaders?.["anthropic-beta"]; if (model && (this.provider === "claude" || (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) { - headers["Anthropic-Beta"] = selectAnthropicBeta(model, body); + headers["Anthropic-Beta"] = mergeAnthropicBeta(selectAnthropicBeta(model, body), clientBeta); + } else if (this.provider === "anthropic" && clientBeta) { + headers["Anthropic-Beta"] = mergeAnthropicBeta(headers["Anthropic-Beta"], clientBeta); + } + + // Claude OAuth: align x-claude-code-session-id with metadata.user_id.session_id if missing + if (this.provider === "claude" && !headers["x-claude-code-session-id"]) { + const token = credentials?.accessToken || credentials?.apiKey || ""; + if (token.includes("sk-ant-oat")) { + const sid = extractClaudeSessionIdFromUserId(body?.metadata?.user_id); + if (sid) headers["x-claude-code-session-id"] = sid; + } } // Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js index fe90c6eb..7eb645a8 100644 --- a/open-sse/executors/opencode-go.js +++ b/open-sse/executors/opencode-go.js @@ -1,8 +1,8 @@ import crypto from "node:crypto"; import { DefaultExecutor } from "./default.js"; import { resolveSessionId } from "../utils/sessionManager.js"; -import { modelTargetFormat } from "../providers/models/schema.js"; -import { getProviderModels } from "../config/providerModels.js"; +import { getModelTargetFormat } from "../config/providerModels.js"; +import { FORMATS } from "../translator/formats.js"; import { normalizeResponsesInput, clampResponsesCallId, @@ -41,16 +41,10 @@ function translatedSession(sessionId, clientTool) { return `ses_${digest}`; } -// Strip the thinking suffix "model(level)" so checks hit the base id. -function baseModelId(model) { - return String(model || "").replace(/\([^()]+\)\s*$/, "").trim(); -} - -// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …). -// Reading the registry keeps this in sync with config — never hardcode model ids here. +// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …), +// including the family-regex fallback for passthrough ids — never hardcode model ids here. function isResponsesModel(model) { - const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model)); - return modelTargetFormat(entry) === "openai-responses"; + return getModelTargetFormat("opencode-go", model) === FORMATS.OPENAI_RESPONSES; } // Flatten Chat Completions tool declarations into the Responses flat shape and diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index 3005fde3..73cc86c2 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -9,6 +9,7 @@ import { createRequestLogger } from "../utils/requestLogger.js"; import { getModelTargetFormat, getModelSupportedFormats, getModelStrip, getModelUpstreamId, getModelType, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js"; import { PROVIDERS } from "../config/providers.js"; import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js"; +import { upstreamResponseHeaders } from "../utils/upstreamHeaders.js"; import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js"; import { handleBypassRequest } from "../utils/bypassHandler.js"; import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js"; @@ -493,7 +494,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred log.errorLine(reqTag, "✗", `ERROR ${statusCode} · ${provider}/${model} · ${Date.now() - requestStartTime}ms${urlStr}\n ${errMsg}`); } reqLogger.logError(new Error(message), finalBody || translatedBody); - return createErrorResult(statusCode, errMsg, resetsAtMs); + return createErrorResult(statusCode, errMsg, resetsAtMs, upstreamResponseHeaders(providerResponse.headers)); } const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers diff --git a/open-sse/handlers/chatCore/nonStreamingHandler.js b/open-sse/handlers/chatCore/nonStreamingHandler.js index ac7d633c..f940371f 100644 --- a/open-sse/handlers/chatCore/nonStreamingHandler.js +++ b/open-sse/handlers/chatCore/nonStreamingHandler.js @@ -4,6 +4,7 @@ import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js"; import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js"; import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js"; import { createErrorResult } from "../../utils/error.js"; +import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js"; import { HTTP_STATUS } from "../../config/runtimeConfig.js"; import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js"; import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js"; @@ -417,7 +418,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m return { success: true, response: new Response(JSON.stringify(restoreToolNames(translatedResponse, toolNameMap)), { - headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } + headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*", ...upstreamResponseHeaders(providerResponse.headers) } }) }; } diff --git a/open-sse/handlers/chatCore/streamingHandler.js b/open-sse/handlers/chatCore/streamingHandler.js index a52072b0..4eb22ca2 100644 --- a/open-sse/handlers/chatCore/streamingHandler.js +++ b/open-sse/handlers/chatCore/streamingHandler.js @@ -20,6 +20,7 @@ import { import { streamStatusForContent } from "../../utils/streamErrorPatterns.js"; import { saveRequestDetail } from "@/lib/usageDb.js"; import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js"; +import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js"; // Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat. // Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI. @@ -158,7 +159,12 @@ export async function handleStreamingResponse({ providerResponse, provider, mode return { success: true, - response: new Response(transformedBody, { headers: SSE_HEADERS }), + response: new Response(transformedBody, { + headers: { + ...SSE_HEADERS, + ...upstreamResponseHeaders(providerResponse.headers), + }, + }), }; } diff --git a/open-sse/handlers/geminiLiveStt.js b/open-sse/handlers/geminiLiveStt.js new file mode 100644 index 00000000..51837732 --- /dev/null +++ b/open-sse/handlers/geminiLiveStt.js @@ -0,0 +1,266 @@ +import { Buffer } from "node:buffer"; + +// Gemini Live API realtime STT transport. +// +// The REST generateContent path (sttCore.transcribeGemini) only transcribes +// whole files inline. The Live API's `:bidiGenerateContent` WebSocket is the +// streaming counterpart: audio goes up as realtimeInput mediaChunks and the +// server pushes incremental `serverContent.inputTranscription` events back. +// This module owns the socket lifecycle only — envelope/response shaping +// stays in sttCore so the engine's single STT exit shape is preserved. +// +// Marker contract: dispatched from sttCore's format-switch when the model +// entry carries `transport: "gemini-live"` (registry) or the caller passes a +// transport string (custom models). Never keyed on a hardcoded model id here. +// +// Transport behavior: +// - Node >= 22 global WebSocket (undici). No new dependency. +// - Live API expects low-latency PCM; other containers are forwarded with +// their declared MIME unchanged (provider-side rejection is surfaced). +// - Text accumulation is append-only over inputTranscription segments and +// ends on serverContent.turnComplete (or graceful close with partial text). +// - Transcription deltas are kept per-frame (chunks[]) so sttCore can shape +// verbose_json segments without fabricating timestamps. goAway advisements +// rotate the socket once per call: setup replay + byte-offset resume. + +const SETUP_TIMEOUT_MS = 10_000; // open → setupComplete +const TURN_TIMEOUT_MS = 60_000; // audio streamed → turnComplete +const MAX_TIMEOUT_MS = 300_000; // clamp ceiling for client-supplied lifecycle knobs +const CHUNK_BYTES = 16_384; // ~0.5s of 16-bit 16kHz mono PCM +const GOAWAY_RECONNECTS = 1; // socket rotations honoured per call + +class GeminiLiveError extends Error { + constructor(message, status) { + super(message); + this.name = "GeminiLiveError"; + this.status = status || 502; + } +} + +// REST base (https://host/v1beta/models) → Live WS base +// (wss://host/ws/api/v1beta/models), then the bidiGenerateContent endpoint. +function toLiveWsUrl(baseUrl, model, token) { + const url = new URL(baseUrl); + url.protocol = "wss:"; + if (!url.pathname.startsWith("/ws/")) url.pathname = `/ws/api${url.pathname}`; + const base = url.toString().replace(/\/+$/, ""); + return `${base}/${encodeURIComponent(model)}:bidiGenerateContent?key=${encodeURIComponent(token || "")}`; +} + +// Bind socket events supporting BOTH handler styles: addEventListener +// (browser WebSocket, undici) and onopen/onmessage property assignment +// (minimal polyfills). Whichever the implementation exposes, it works. +function bindSocket(ws, { onOpen, onMessage, onError, onClose }) { + if (typeof ws.addEventListener === "function") { + ws.addEventListener("open", onOpen); + ws.addEventListener("message", onMessage); + ws.addEventListener("error", onError); + ws.addEventListener("close", onClose); + return; + } + ws.onopen = onOpen; + ws.onmessage = onMessage; + ws.onerror = onError; + ws.onclose = onClose; +} + +function parseFrame(data) { + try { + return JSON.parse(typeof data === "string" ? data : String(data)); + } catch { + return null; // non-JSON frames carry no Live API semantics + } +} + +function firstStringField(formData, key) { + const v = typeof formData?.get === "function" ? formData.get(key) : null; + return typeof v === "string" && v.trim() ? v.trim() : ""; +} + +// Lifecycle knobs the live registry entry advertises in params[] +// (setup/turn timeouts). They ride the same formData pass-through sttCore +// gives every transport — no sttCore change needed to reach this leaf. +function firstNumberField(formData, key, fallback) { + const n = Number(firstStringField(formData, key)); + return Number.isFinite(n) && n > 0 ? Math.min(n, MAX_TIMEOUT_MS) : fallback; +} + +/** + * Transcribe an audio File via the Gemini Live bidirectional stream. + * @returns {Promise<{text: string, chunks: string[]}>} transcript plus the raw + * incremental inputTranscription deltas (sttCore shapes verbose_json from them). + * @throws {GeminiLiveError} with .status for the sttCore error envelope. + */ +export async function transcribeGeminiLive({ cfg, file, model, token, formData, mimeType }) { + const WS = globalThis.WebSocket; + if (!WS) throw new GeminiLiveError("Gemini Live transport needs global WebSocket (Node >= 22)", 502); + + const buf = Buffer.from(await file.arrayBuffer()); + if (!buf.length) throw new GeminiLiveError("Empty audio file", 400); + + const instruction = firstStringField(formData, "prompt") || "Transcribe the spoken audio verbatim."; + const language = firstStringField(formData, "language"); + const setupTimeoutMs = firstNumberField(formData, "setup_timeout_ms", SETUP_TIMEOUT_MS); + const turnTimeoutMs = firstNumberField(formData, "turn_timeout_ms", TURN_TIMEOUT_MS); + // system_instruction (registry param) overrides the built-in transcription + // directive wholesale; prompt/language only shape the default. + const instructionOverride = firstStringField(formData, "system_instruction"); + const systemText = instructionOverride + || (language ? `${instruction} Language: ${language}.` : instruction); + const wsUrl = toLiveWsUrl(cfg.baseUrl, model, token); + + return await new Promise((resolve, reject) => { + let text = ""; + const chunks = []; // raw inputTranscription deltas, shaped by sttCore + let settled = false; + let timer = null; + let goAwayTimer = null; + let ws = null; + let generation = 0; // socket identity: superseded closes never settle + let sentBytes = 0; // audio prefix already handed to the live socket + let goAwayReconnects = GOAWAY_RECONNECTS; + + const arm = (ms, message) => { + if (timer) clearTimeout(timer); + timer = setTimeout(() => fail(new GeminiLiveError(message, 504)), ms); + }; + const shutdown = () => { + if (timer) { clearTimeout(timer); timer = null; } + if (goAwayTimer) { clearTimeout(goAwayTimer); goAwayTimer = null; } + // ws is null until the first open() dials (and stays null when the + // constructor throws) — fail() runs shutdown() on that path. + if (!ws) return; + try { + if (ws.readyState === WS.OPEN || ws.readyState === WS.CONNECTING) ws.close(1000); + } catch { /* socket already dead — outcome is already settled */ } + }; + const succeed = () => { + if (settled) return; + settled = true; + shutdown(); + resolve({ text, chunks }); + }; + const fail = (err) => { + if (settled) return; + settled = true; + shutdown(); + reject(err); + }; + const send = (frame) => { + if (ws.readyState !== WS.OPEN) return false; + try { + ws.send(JSON.stringify(frame)); + } catch { + return false; // socket died mid-send — streamAudioAndPrompt maps this to a 502 + } + return true; + }; + + // Streams every byte not yet sent, then the flushing text turn. After a + // goAway rotation this resumes from sentBytes — no audio re-upload. + const streamAudioAndPrompt = () => { + for (let off = sentBytes; off < buf.length; off += CHUNK_BYTES) { + const mediaChunk = buf.subarray(off, off + CHUNK_BYTES).toString("base64"); + if (!send({ realtimeInput: { mediaChunks: [{ mimeType, data: mediaChunk }] } })) { + fail(new GeminiLiveError("Gemini Live socket closed while streaming audio", 502)); + return; + } + sentBytes = Math.min(off + CHUNK_BYTES, buf.length); + } + // Final user turn: flushes the recognizer and yields turnComplete. + send({ clientContent: { turns: [{ parts: [{ text: systemText }] }], turnComplete: true } }); + }; + + // goAway: the server names the instant it will force-close this socket. + // Graceful play = rotate BEFORE the deadline: retire the live socket, + // dial a fresh one, replay setup, resume audio from sentBytes — text and + // chunks survive the hop. Once the advisory budget is spent a later + // goAway is left to the close path, which settles on partial transcript. + const scheduleGoAwayReconnect = (goAway) => { + if (settled || goAwayTimer || goAwayReconnects <= 0) return; + const deadline = Date.parse(typeof goAway?.time === "string" ? goAway.time : ""); + const delay = Number.isFinite(deadline) + ? Math.max(0, Math.min(deadline - Date.now(), setupTimeoutMs)) + : 0; + goAwayTimer = setTimeout(() => { + goAwayTimer = null; + if (settled) return; + goAwayReconnects--; + generation++; + try { ws?.close(1000); } catch { /* deadline crossed mid-flight — re-dial anyway */ } + open(); + }, delay); + }; + + const open = () => { + const gen = ++generation; + try { + ws = new WS(wsUrl); + } catch { + fail(new GeminiLiveError("Gemini Live websocket connection failed", 502)); + return; + } + bindSocket(ws, { + onOpen: () => { + if (settled || gen !== generation) return; + arm(setupTimeoutMs, "Gemini Live timed out waiting for setupComplete"); + send({ + setup: { + model: `models/${model}`, + generationConfig: { + responseModalities: ["TEXT"], + inputAudioTranscription: {}, + }, + systemInstruction: { parts: [{ text: systemText }] }, + }, + }); + }, + onMessage: (ev) => { + if (settled || gen !== generation) return; + const frame = parseFrame(ev?.data); + if (!frame) return; + + if (frame.error) { + const e = frame.error; + fail(new GeminiLiveError(`Gemini Live error${e.status ? ` (${e.status})` : ""}: ${e.message || "unknown"}`, 502)); + return; + } + if (frame.goAway) { + scheduleGoAwayReconnect(frame.goAway); + return; + } + + const sc = frame.serverContent; + if (!sc) return; + + const delta = typeof sc.inputTranscription?.text === "string" ? sc.inputTranscription.text : ""; + // Trim before testing: a padding-only frame carries no transcript and + // must not make an empty run look like a partial success on close. + if (delta.trim()) { + text += delta; + chunks.push(delta); + } + + if (sc.setupComplete) { + arm(turnTimeoutMs, "Gemini Live transcription timed out"); + streamAudioAndPrompt(); + return; + } + if (sc.turnComplete) succeed(); + }, + onError: () => { + if (settled || gen !== generation) return; + fail(new GeminiLiveError("Gemini Live websocket connection failed", 502)); + }, + onClose: (ev) => { + if (settled || gen !== generation) return; + // Partial transcript beats a hard error on graceful close; silence is one. + if (text.trim()) succeed(); + else fail(new GeminiLiveError(`Gemini Live socket closed before completion${ev?.code ? ` (code ${ev.code})` : ""}`, 502)); + }, + }); + }; + + open(); + }); +} diff --git a/open-sse/handlers/sttCore.js b/open-sse/handlers/sttCore.js index 8127782a..21582c4d 100644 --- a/open-sse/handlers/sttCore.js +++ b/open-sse/handlers/sttCore.js @@ -1,5 +1,7 @@ import { Buffer } from "node:buffer"; import { createErrorResult } from "../utils/error.js"; +import { transcribeGeminiLive } from "./geminiLiveStt.js"; +import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js"; import { HTTP_STATUS } from "../config/runtimeConfig.js"; // Build auth headers from sttConfig + token @@ -162,11 +164,26 @@ function jsonResponse(obj) { }; } +// Model-level transport marker (registry models[].transport, e.g. the Gemini +// live STT entry's "gemini-live", or a custom model's stored transport). +// Dispatch reads the marker — never a hardcoded model id — so new realtime +// providers extend sttCore through data, not code. +function resolveModelTransport(provider, model) { + const key = PROVIDER_ID_TO_ALIAS[provider] || provider; + const models = PROVIDER_MODELS[key] || PROVIDER_MODELS[provider]; + if (!Array.isArray(models)) return null; + const entry = models.find((m) => m && m.id === model && (m.kind || "llm") === "stt"); + const marker = typeof entry?.transport === "string" ? entry.transport.trim() : ""; + return marker || null; +} + /** - * STT core handler — dispatch by sttConfig.format. + * STT core handler — dispatch by model transport marker, else sttConfig.format. + * `transport` is the caller-supplied marker override (custom models resolve + * it in the app layer; built-ins fall back to the registry entry marker). * @returns {Promise<{success, response, status?, error?}>} */ -export async function handleSttCore({ provider, model, formData, credentials, sttConfig }) { +export async function handleSttCore({ provider, model, formData, credentials, sttConfig, transport }) { const file = formData.get("file"); if (!file) return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: file"); @@ -186,8 +203,29 @@ export async function handleSttCore({ provider, model, formData, credentials, st return createErrorResult(HTTP_STATUS.UNAUTHORIZED, `No credentials for STT provider: ${provider}`); } + // Format-switch extension: an explicit caller marker wins over the registry + // marker; with neither, the provider-default sttConfig.format applies. + const marker = (typeof transport === "string" && transport.trim()) ? transport.trim() : resolveModelTransport(provider, model); + try { - switch (cfg.format) { + switch (marker || cfg.format) { + case "gemini-live": { + const live = await transcribeGeminiLive({ cfg, file, model, token, formData, mimeType: resolveAudioContentType(file) }); + // response_format parity with the OpenAI-compatible transport: default + // envelope stays {text}; verbose_json adds segments mapped from the + // Live API's incremental inputTranscription deltas. Those frames carry + // NO timestamps, so segments expose {id,text} only (id = delta order, + // Whisper-compatible 0-based) — start/end/duration are deliberately + // absent rather than fabricated as zeros, which would misrepresent + // provider data to callers diffing transports. + const fmt = typeof formData?.get === "function" + ? String(formData.get("response_format") ?? "").trim().toLowerCase() + : ""; + if (fmt === "verbose_json") { + return jsonResponse({ text: live.text, segments: live.chunks.map((segText, id) => ({ id, text: segText })) }); + } + return jsonResponse({ text: live.text }); + } case "deepgram": return await transcribeDeepgram(cfg, file, model, token, formData); case "assemblyai": return await transcribeAssemblyAI(cfg, file, model, token); case "nvidia-asr": return await transcribeNvidia(cfg, file, model, token); @@ -196,6 +234,6 @@ export async function handleSttCore({ provider, model, formData, credentials, st default: return await transcribeOpenAICompatible(cfg, file, model, token, formData); } } catch (err) { - return createErrorResult(HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed"); + return createErrorResult(err.status || HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed"); } } diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index ef331149..e7f48560 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -430,21 +430,29 @@ const TRUST_UPSTREAM_VISION = new Set(["openrouter"]); * * @param {string[]} comboModels * @param {Object|null} [comboLookup] optional map of combo name → models array for nested resolution + * @param {Function|null} [resolveCaps] optional (fullId) → caps override. The synced model + * catalog is server-only (it reads a file), so a browser-side resolution cannot see the + * limits it supplies and silently falls back to the generic patterns below. Callers that + * have the server's answer (/api/models, via useModelCaps) pass it here; it is merged over + * the local tables, so fields it does not carry (tools, pdf, audio/video, thinking*) survive. * @param {number} [_depth] internal recursion depth guard * @returns {object|null} full capabilities object, or null for empty input */ -export function aggregateComboCapabilities(comboModels, comboLookup = null, _depth = 0) { +export function aggregateComboCapabilities(comboModels, comboLookup = null, resolveCaps = null, _depth = 0) { if (!comboModels?.length || _depth > 6) return null; const allCaps = comboModels.map((fullId) => { // Nested combo: bare name (no slash) that exists in the lookup — recurse if (!fullId.includes("/") && comboLookup?.[fullId]) { - return aggregateComboCapabilities(comboLookup[fullId], comboLookup, _depth + 1) + return aggregateComboCapabilities(comboLookup[fullId], comboLookup, resolveCaps, _depth + 1) + ?? resolveCaps?.(fullId) ?? getCapabilitiesForModel(null, fullId); } const slash = fullId.indexOf("/"); const provider = slash === -1 ? null : fullId.slice(0, slash); const model = slash === -1 ? fullId : fullId.slice(slash + 1); - return getCapabilitiesForModel(provider, model); + const local = getCapabilitiesForModel(provider, model); + const override = resolveCaps?.(fullId); + return override ? { ...local, ...override } : local; }); const first = allCaps[0]; return { @@ -482,7 +490,9 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"]; // handlers (silently: the setters still "succeed"). The slots therefore live on // globalThis, which IS shared across server bundles in the same process. // Same reason the browser bundle is safe: it never calls a setter, so the slots -// stay empty and every consumer below short-circuits. +// stay empty and every consumer below short-circuits. Every read goes through +// globalThis: caching it locally would keep a reader alive in other copies after +// setCatalogSource(null). let catalogSource = null; const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= { catalog: null, // { getModalities, getLimits } — synced models.dev catalog @@ -496,15 +506,13 @@ const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= { */ export function setCatalogSource(source) { catalogSource = source || null; - SOURCE_SLOTS.catalog = source || null; + if (SOURCE_SLOTS) SOURCE_SLOTS.catalog = source || null; if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source || null; } function getCatalogSource() { - if (catalogSource) return catalogSource; - if (SOURCE_SLOTS.catalog) return (catalogSource = SOURCE_SLOTS.catalog); - if (typeof globalThis === "undefined") return null; - return (catalogSource = globalThis.__9rCatalogSource || null); + if (typeof globalThis === "undefined") return catalogSource; + return SOURCE_SLOTS?.catalog || globalThis.__9rCatalogSource || null; } // Capabilities the user asserted per provider+model (dashboard "Add/Edit Model" diff --git a/open-sse/providers/models/helpers.js b/open-sse/providers/models/helpers.js index c66d662b..cf8076bd 100644 --- a/open-sse/providers/models/helpers.js +++ b/open-sse/providers/models/helpers.js @@ -1,3 +1,5 @@ +import { FORMATS } from "../../translator/formats.js"; + // Codex auto-generates a "-review" variant for each llm model (review quota family) export const CODEX_REVIEW_SUFFIX = "-review"; @@ -25,3 +27,20 @@ export function isMuseSparkModel(modelId) { const base = clean.includes("/") ? clean.split("/").pop() : clean; return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base); } + +// Endpoint families for OpenCode models outside the curated registry (modelsFetcher / +// passthrough ids) — regex keeps auto-fetched models on the right endpoint: +// /responses (gpt/grok/muse-spark), /messages (minimax/qwen), /chat/completions (rest). +// Curated registry entries always win; this is the unknown-id fallback only. +const OPENCODE_FAMILIES = [ + { match: /^(grok|gpt|muse[-_]?spark)/i, supportedFormats: [FORMATS.OPENAI_RESPONSES], targetFormat: FORMATS.OPENAI_RESPONSES }, + { match: /^deepseek-v4-(pro|flash)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE, FORMATS.OPENAI_RESPONSES] }, + { match: /^(minimax|qwen)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE] }, + { match: /^claude-/i, supportedFormats: [FORMATS.CLAUDE] }, +]; + +export function opencodeFamilyFormats(modelId) { + if (!modelId || typeof modelId !== "string") return null; + const base = modelId.replace(/\([^()]+\)\s*$/, "").trim(); + return OPENCODE_FAMILIES.find((f) => f.match.test(base)) || null; +} diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index d09452e2..0cdd6844 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -2,8 +2,28 @@ // // Fallback order (first match wins): // 1. PROVIDER_PRICING[provider][model] — provider-specific override -// 2. MODEL_PRICING[model] — canonical model price (provider-agnostic) -// 3. PATTERN_PRICING — glob pattern match (e.g. "codex-*") +// 2. FREE_MODEL_NAMESPACES — upstream bills these at $0 +// 3. MODEL_PRICING[model] — canonical model price (provider-agnostic) +// 4. PATTERN_PRICING — glob pattern match (e.g. "codex-*") + +/** + * Namespaces upstream meters at $0. A free model must never inherit a paid + * rate: the vendor-prefix strip in getPricingForModel() would turn + * "cline-free/deepseek-v4.1-flash" into "deepseek-v4.1-flash" and match + * MODEL_PRICING, so the namespace is checked before both fallbacks. + */ +export const FREE_MODEL_NAMESPACES = ["cline-free/"]; + +export const ZERO_PRICING = { + input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0, +}; + +/** True when the model id sits in a namespace upstream bills at $0. */ +export function isFreeModel(model) { + if (!model) return false; + const lower = String(model).toLowerCase(); + return FREE_MODEL_NAMESPACES.some((ns) => lower.startsWith(ns)); +} /** * Canonical model pricing — provider-agnostic. @@ -361,10 +381,11 @@ export function matchPattern(pattern, model) { } /** - * Resolve pricing for a model using the 3-step fallback chain: + * Resolve pricing for a model using the 4-step fallback chain: * 1. PROVIDER_PRICING[provider][model] - * 2. MODEL_PRICING[model] - * 3. PATTERN_PRICING (glob match) + * 2. free namespace (upstream bills $0) + * 3. MODEL_PRICING[model] + * 4. PATTERN_PRICING (glob match) * * @param {string} provider * @param {string} model @@ -378,12 +399,15 @@ export function getPricingForModel(provider, model) { return PROVIDER_PRICING[provider][model]; } - // 2. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat") + // 2. Free namespaces bill $0 regardless of the model name behind them. + if (isFreeModel(model)) return ZERO_PRICING; + + // 3. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat") const baseModel = model.includes("/") ? model.split("/").pop() : model; if (MODEL_PRICING[baseModel]) return MODEL_PRICING[baseModel]; if (MODEL_PRICING[model]) return MODEL_PRICING[model]; - // 3. Pattern match + // 4. Pattern match for (const { pattern, pricing } of PATTERN_PRICING) { if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) { return pricing; diff --git a/open-sse/providers/registry/agnes.js b/open-sse/providers/registry/agnes.js new file mode 100644 index 00000000..4cf2a88e --- /dev/null +++ b/open-sse/providers/registry/agnes.js @@ -0,0 +1,29 @@ +export default { + id: "agnes", + priority: 120, + alias: "agnes", + aliases: [ + "agnes-ai", + ], + uiAlias: "agnes", + display: { + name: "Agnes AI", + icon: "auto_awesome", + color: "#7C3AED", + textIcon: "AG", + website: "https://agnes-ai.com", + notice: { + text: "OpenAI-compatible gateway from Agnes AI, offering free API credits on sign-up. Accepts a bearer token or an x-api-key header.", + apiKeyUrl: "https://platform.agnes-ai.com", + }, + }, + category: "freeTier", + authType: "apikey", + transport: { + baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions", + validateUrl: "https://apihub.agnes-ai.com/v1/models", + }, + // No model ids could be verified without a key, so discovery is left to the + // live endpoint and any id is accepted through passthroughModels. + passthroughModels: true, +}; diff --git a/open-sse/providers/registry/atria.js b/open-sse/providers/registry/atria.js new file mode 100644 index 00000000..ed9de421 --- /dev/null +++ b/open-sse/providers/registry/atria.js @@ -0,0 +1,32 @@ +export default { + id: "atria", + priority: 120, + alias: "atria", + aliases: [ + "atria-asi", + ], + uiAlias: "atria", + display: { + name: "Atria Dawn", + icon: "flare", + color: "#C2410C", + textIcon: "AD", + website: "https://atria-asi.ai", + notice: { + text: "OpenAI-compatible endpoint from Atria Dawn (AtomInnoLab). Currently a research preview offering a single text model, Atria-Dawn-Preview.", + apiKeyUrl: "https://api.atria-asi.ai/dashboard", + }, + }, + category: "apikey", + authType: "apikey", + transport: { + baseUrl: "https://api.atria-asi.ai/v1/chat/completions", + validateUrl: "https://api.atria-asi.ai/v1/models", + }, + // Docs pin the model field to one case-sensitive id. Text-only for now: the + // service ships a hook that blocks image/PDF input, so no vision is claimed. + models: [ + { id: "Atria-Dawn-Preview", name: "Atria Dawn Preview" }, + ], + passthroughModels: true, +}; diff --git a/open-sse/providers/registry/bai.js b/open-sse/providers/registry/bai.js new file mode 100644 index 00000000..84eb2797 --- /dev/null +++ b/open-sse/providers/registry/bai.js @@ -0,0 +1,30 @@ +export default { + id: "bai", + priority: 120, + alias: "bai", + aliases: [ + "b-ai", + ], + uiAlias: "bai", + display: { + name: "B.AI", + icon: "account_balance", + color: "#0369A1", + textIcon: "BA", + website: "https://b.ai", + notice: { + text: "OpenAI-compatible gateway with one of the larger catalogues here. Accepts a bearer token or an x-api-key header. Model ids are fetched live from the provider.", + apiKeyUrl: "https://b.ai", + }, + }, + category: "apikey", + authType: "apikey", + transport: { + baseUrl: "https://api.b.ai/v1/chat/completions", + validateUrl: "https://api.b.ai/v1/models", + }, + // No ids hardcoded: the catalogue is large and rotates, so the live endpoint + // is the source of truth and any id is accepted via passthroughModels. + modelsFetcher: { url: "https://api.b.ai/v1/models", type: "openai" }, + passthroughModels: true, +}; diff --git a/open-sse/providers/registry/claude.js b/open-sse/providers/registry/claude.js index d0bc9302..3677ffb5 100644 --- a/open-sse/providers/registry/claude.js +++ b/open-sse/providers/registry/claude.js @@ -54,6 +54,8 @@ export default { oauthUrl: "https://api.anthropic.com/api/oauth/usage", orgUrl: "https://api.anthropic.com/v1/organizations/{org_id}/usage", settingsUrl: "https://api.anthropic.com/v1/settings", + profileUrl: "https://api.anthropic.com/api/oauth/profile", + resetUrl: "https://api.anthropic.com/api/organizations/{org_id}/reset_rate_limits", }, }, models: [ diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 710cbc27..c7e192e9 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -2,7 +2,8 @@ import { withCodexReviewModels } from "../models/helpers.js"; // Codex CLI version seen by OpenAI's backend — single source for the Version / // User-Agent identity headers. Bump when the installed codex CLI is upgraded. -const CODEX_CLI_VERSION = "0.154.0"; +const CODEX_CLI_VERSION = "0.155.0"; +const GPT_6_LITE_THINKING_LEVELS = ["low", "medium", "high", "xhigh", "max"]; export default { id: "codex", @@ -42,6 +43,7 @@ export default { headers: { originator: "codex_cli_rs", "User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`, + version: CODEX_CLI_VERSION, }, usage: { url: "https://chatgpt.com/backend-api/wham/usage", @@ -51,6 +53,8 @@ export default { }, models: [ { id: "gpt-6-astra", name: "GPT 6.0 Astra" }, + { id: "gpt-6-sol", name: "GPT 6.0 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, + { id: "gpt-6-luna", name: "GPT 6.0 Luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, { id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, diff --git a/open-sse/providers/registry/dahl.js b/open-sse/providers/registry/dahl.js new file mode 100644 index 00000000..8efc0189 --- /dev/null +++ b/open-sse/providers/registry/dahl.js @@ -0,0 +1,35 @@ +export default { + id: "dahl", + priority: 120, + alias: "dahl", + aliases: [ + "dahl-inference", + ], + uiAlias: "dahl", + display: { + name: "Dahl Inference", + icon: "hub", + color: "#1E40AF", + textIcon: "DH", + website: "https://dahl.global", + notice: { + text: "OpenAI-compatible Gonka inference node. Small, fixed catalogue (GLM-5.3-Flash, DeepSeek-V4-Flash, MiniMax-M2.7) at a flat per-token rate.", + apiKeyUrl: "https://dahl.global/dashboard", + }, + }, + category: "apikey", + authType: "apikey", + transport: { + baseUrl: "https://inference.dahl.global/v1/chat/completions", + validateUrl: "https://inference.dahl.global/v1/models", + }, + // The live catalogue is public (no auth), so modelsFetcher works without a key + // and the ids below are a convenience seed rather than an exhaustive list. + models: [ + { id: "zai-org/GLM-5.3-Flash", name: "GLM-5.3 Flash" }, + { id: "deepseek-ai/DeepSeek-V4-Flash-0731", name: "DeepSeek V4 Flash 0731" }, + { id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7" }, + ], + modelsFetcher: { url: "https://inference.dahl.global/v1/models", type: "openai" }, + passthroughModels: true, +}; diff --git a/open-sse/providers/registry/gemini.js b/open-sse/providers/registry/gemini.js index 590bf48e..e0fb109d 100644 --- a/open-sse/providers/registry/gemini.js +++ b/open-sse/providers/registry/gemini.js @@ -58,6 +58,7 @@ export default { { id: "gemini-2.5-flash", name: "Gemini 2.5 Flash", params: ["language","prompt"], kind: "stt" }, { id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash Lite (Cheapest)", params: ["language","prompt"], kind: "stt" }, { id: "gemini-2.0-flash", name: "Gemini 2.0 Flash", params: ["language","prompt"], kind: "stt" }, + { id: "gemini-2.5-flash-native-audio-preview-09-17", name: "Gemini Live Transcription (Realtime)", params: ["language","prompt","system_instruction","setup_timeout_ms","turn_timeout_ms"], kind: "stt", transport: "gemini-live" }, { id: "gemini-3.1-flash-tts-preview", name: "Gemini 3.1 Flash TTS", kind: "tts" }, { id: "gemini-2.5-flash-preview-tts", name: "Gemini 2.5 Flash TTS", kind: "tts" }, { id: "gemini-2.5-pro-preview-tts", name: "Gemini 2.5 Pro TTS", kind: "tts" }, diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 0e8dbe2e..07b71d2c 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -125,6 +125,11 @@ import p119 from "./selfhosted-embedding.js"; import p120 from "./fish-audio.js"; import p121 from "./alitp-intl.js"; import p122 from "./xquik.js"; +import p125 from "./tokenharbor.js"; +import p126 from "./dahl.js"; +import p127 from "./atria.js"; +import p129 from "./agnes.js"; +import p130 from "./bai.js"; export default [ p0, p1, @@ -250,4 +255,9 @@ export default [ p120, p121, p122, + p125, + p126, + p127, + p129, + p130, ]; diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index d65d4c15..7d161270 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -35,37 +35,56 @@ export default { ], // supportedFormats follow the endpoint table in https://opencode.ai/docs/go/ models: [ - { id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] }, + { id: "deepseek-flash", name: "DeepSeek Flash", supportedFormats: ["openai"] }, { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] }, { id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] }, { id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] }, { id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] }, + { id: "glm-5", name: "GLM 5", supportedFormats: ["openai"] }, { id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] }, { id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] }, + { id: "kimi-k2.5", name: "Kimi K2.5", supportedFormats: ["openai"] }, { id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] }, { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] }, + { id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] }, + { id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", supportedFormats: ["openai"] }, + { id: "mimo-v2.6-pro", name: "MiMo V2.6 Pro", supportedFormats: ["openai"] }, { id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] }, { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] }, + { id: "mimo-v2-pro", name: "MiMo V2 Pro", supportedFormats: ["openai"] }, + { id: "mimo-v2-omni", name: "MiMo V2 Omni", supportedFormats: ["openai"] }, { id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] }, { id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] }, { id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] }, + { id: "space-bunny-free", name: "Space Bunny Free", supportedFormats: ["openai", "claude"] }, { id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] }, { id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, + { id: "qwen3.5-plus", name: "Qwen 3.5 Plus", supportedFormats: ["openai", "claude"] }, { id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] }, { id: "hy3", name: "Hy3", supportedFormats: ["openai"] }, + { id: "hy3-preview", name: "Hy3 Preview", supportedFormats: ["openai"] }, + // In /zen/go/v1/models but absent from the docs endpoint table — chat lane is the fallback guess + { id: "omen-alpha", name: "Omen Alpha", supportedFormats: ["openai"] }, // Served by /zen/go/v1/responses only — the responses-only entry forces chatCore // past the sourceFormat-matched transports into translation (see chatCore guard). + { id: "grok-4.7", name: "Grok 4.7", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "grok-4.5", name: "Grok 4.5", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "gpt-6-luna", name: "GPT 6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, { id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, ], + // Live catalogue; ids outside this curated list get their endpoint lane from the + // family regex in providers/models/helpers.js (opencodeFamilyFormats). + modelsFetcher: { url: "https://opencode.ai/zen/go/v1/models", type: "opencode-go" }, + passthroughModels: true, features: { usage: true, usageApikey: true, diff --git a/open-sse/providers/registry/tokenharbor.js b/open-sse/providers/registry/tokenharbor.js new file mode 100644 index 00000000..2ef04347 --- /dev/null +++ b/open-sse/providers/registry/tokenharbor.js @@ -0,0 +1,49 @@ +export default { + id: "tokenharbor", + priority: 120, + alias: "tokenharbor", + aliases: [ + "th", + "thh", + ], + uiAlias: "tokenharbor", + display: { + name: "Token Harbor", + icon: "anchor", + color: "#0F766E", + textIcon: "TH", + website: "https://tokenharbor.ai", + notice: { + text: "OpenAI-compatible aggregator. One API key reaches every model, billed per-token from a prepaid wallet. Model ids are bare (e.g. claude-opus-5.5, gpt-6-astra, deepseek-v4.1-flash:free) and are fetched live from the provider.", + apiKeyUrl: "https://tokenharbor.ai/dashboard", + }, + }, + category: "apikey", + authType: "apikey", + transport: { + // OpenAI-compatible. `format` is left at the shared "openai" default and + // `thinkingFormat` is deliberately NOT declared: Token Harbor forwards + // requests verbatim, so each model must resolve its own thinking wire + // format through providers/capabilities.js. Setting a provider-wide value + // would force one format (e.g. claude-adaptive) onto every model. + baseUrl: "https://tokenharbor.ai/v1/chat/completions", + validateUrl: "https://tokenharbor.ai/v1/models", + retry: { + 429: 2, + }, + }, + // Curated seed; the live catalogue is fetched via modelsFetcher and any other + // id is accepted via passthroughModels. Their catalogue rotates (the :free set + // in particular), so this stays deliberately small and is only the offline + // fallback. Ids are bare — Token Harbor does not prefix them by upstream vendor. + models: [ + { id: "claude-opus-5.5", name: "Claude Opus 5.5" }, + { id: "claude-sonnet-5", name: "Claude Sonnet 5" }, + { id: "gpt-6-astra", name: "GPT-6 Astra" }, + { id: "gpt-6-sol", name: "GPT-6 Sol" }, + { id: "deepseek-v4.1-flash:free", name: "DeepSeek V4.1 Flash (Free)" }, + { id: "grok-4.7", name: "Grok 4.7" }, + ], + modelsFetcher: { url: "https://tokenharbor.ai/v1/models", type: "openai" }, + passthroughModels: true, +}; diff --git a/open-sse/providers/shared.js b/open-sse/providers/shared.js index 176c82e9..def3171d 100644 --- a/open-sse/providers/shared.js +++ b/open-sse/providers/shared.js @@ -77,6 +77,11 @@ export function selectAnthropicBeta(model = "", body = null) { return flags.join(","); } +export function mergeAnthropicBeta(...values) { + const flags = values.flatMap((v) => (typeof v === "string" ? v.split(",") : [])).map((f) => f.trim()).filter(Boolean); + return [...new Set(flags)].join(","); +} + // Shared baseUrls export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages"; diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index 7589b9e6..63850064 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -3,6 +3,7 @@ import { getCapabilitiesForModel } from "./capabilities.js"; import { matchPattern } from "./pricing.js"; import { resolveKiroEffortPath } from "../config/kiroConstants.js"; +import { getProviderModels } from "../config/providerModels.js"; // Shared level sets (deduped) — verified against provider docs + wire in thinkingUnified.applyFormat. const L = { @@ -42,6 +43,8 @@ const PATTERN_THINKING = [ { provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS }, { pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking { pattern: "*mimo*v2.6*", levels: ["none", "low", "medium", "high", "xhigh"] }, + // mimo-v2.5-pro on opencode-go rejects reasoning_effort "max" (probed live); v2.5 accepts it. + { pattern: "*mimo*v2.5-pro*", levels: ["none", "low", "medium", "high", "xhigh"] }, // DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max // all 200 via output_config.effort; "none" is a 400 on the anthropic route // (disable thinking instead). none kept for the picker = disable. @@ -73,10 +76,14 @@ export function getThinkingLevels(provider, model) { if (provider === "kiro" && resolveKiroEffortPath(model) === null) return null; const caps = getCapabilitiesForModel(provider, model); if (!caps.reasoning) return null; + const baseId = String(model || "").replace(/\([^()]+\)\s*$/, ""); + const modelLevels = provider === "codex" + ? getProviderModels("cx").find((entry) => entry.id === baseId)?.thinkingLevels + : null; const hit = PATTERN_THINKING.find((entry) => (!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model) ); - let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base; + let levels = modelLevels || hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base; if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none"); return levels; } diff --git a/open-sse/services/clinepassModels.js b/open-sse/services/clinepassModels.js index 0aa96ffe..8ba86147 100644 --- a/open-sse/services/clinepassModels.js +++ b/open-sse/services/clinepassModels.js @@ -1,6 +1,12 @@ import { buildClineHeaders } from "../shared/clineAuth.js"; const CLINEPASS_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/models"; +// Cline's free tier is published here, not in /api/v1/models: the catalog +// endpoint carries no `cline-free/*` ids at all. Cline's own SDK calls this +// feed unauthenticated (sdk/packages/core/src/services/llms/cline-recommended-models.ts), +// so no Authorization header is sent — adding one would only make the request +// fail on a header the endpoint ignores. +const CLINE_RECOMMENDED_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/ai/cline/recommended-models"; const FETCH_TIMEOUT_MS = 5000; /** @@ -72,6 +78,40 @@ export async function resolveClinepassModels(credentials) { return models.length ? { models } : null; } +/** + * Fetch Cline's recommended-models feed and return only its `free[]` tier. + * Returns null on any failure — the free tier is additive, so a dead feed must + * never take the /api/v1/models catalog down with it. + * @param {{accessToken?: string, apiKey?: string}} credentials + * @returns {Promise<{id: string, name: string}[] | null>} + */ +async function fetchClineFreeTierModels() { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS); + + try { + const response = await fetch(CLINE_RECOMMENDED_MODELS_ENDPOINT, { + method: "GET", + headers: { Accept: "application/json" }, + signal: controller.signal, + }); + + if (!response.ok) return null; + + const json = await response.json(); + const free = Array.isArray(json?.free) ? json.free : []; + if (!free.length) return null; + + return free + .filter((m) => typeof m?.id === "string" && m.id.trim() !== "") + .map((m) => ({ id: m.id, name: m.name || m.id })); + } catch { + return null; + } finally { + clearTimeout(timer); + } +} + /** * Fetch Cline live model catalog from Cline's /models endpoint. * Unlike resolveClinepassModels, this returns ALL models (including @@ -91,5 +131,15 @@ export async function resolveClineModels(credentials) { name: m.name || m.id, })); - return models.length ? { models } : null; + // Free tier: /api/v1/models lists no `cline-free/*` ids, so merge the feed's + // free[] in. First writer wins on a shared id, keeping the catalog's entry + // for anything the two sources agree on. + const freeTier = await fetchClineFreeTierModels(); + const byId = new Map(models.map((m) => [m.id, m])); + for (const m of freeTier || []) { + if (!byId.has(m.id)) byId.set(m.id, m); + } + const merged = Array.from(byId.values()); + + return merged.length ? { models: merged } : null; } diff --git a/open-sse/services/usage.js b/open-sse/services/usage.js index cf7d4c24..05fe5d78 100644 --- a/open-sse/services/usage.js +++ b/open-sse/services/usage.js @@ -4,10 +4,10 @@ import { getGitHubUsage } from "./usage/github.js"; import { getGeminiUsage, getAntigravityUsage } from "./usage/google.js"; -import { getClaudeUsage } from "./usage/claude.js"; +import { getClaudeUsage, consumeClaudeResetGrant } from "./usage/claude.js"; import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits } from "./usage/codex.js"; -export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits }; +export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits, consumeClaudeResetGrant }; import { getKiroUsage } from "./usage/kiro.js"; import { getMiniMaxUsage } from "./usage/minimax.js"; import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js"; diff --git a/open-sse/services/usage/claude.js b/open-sse/services/usage/claude.js index d93a0682..13665201 100644 --- a/open-sse/services/usage/claude.js +++ b/open-sse/services/usage/claude.js @@ -3,7 +3,7 @@ */ import { proxyAwareFetch } from "../../utils/proxyFetch.js"; -import { ANTHROPIC_API_VERSION } from "../../providers/shared.js"; +import { ANTHROPIC_API_VERSION, CLAUDE_CLI_VERSION } from "../../providers/shared.js"; import { U, parseResetTime } from "./shared.js"; // Claude API config (urls from registry, apiVersion is header logic kept here) @@ -11,7 +11,11 @@ const CLAUDE_CONFIG = { oauthUsageUrl: U("claude").oauthUrl, usageUrl: U("claude").orgUrl, settingsUrl: U("claude").settingsUrl, + profileUrl: U("claude").profileUrl, + resetUrl: U("claude").resetUrl, apiVersion: ANTHROPIC_API_VERSION, + // Reset grants are gated by surface: only "(external, cli)" UA is eligible + userAgent: `claude-cli/${CLAUDE_CLI_VERSION} (external, cli)`, }; // OAuth usage endpoint rate-limits (429); cool down per-token to stop hammering it. @@ -64,12 +68,14 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) { } // Primary: OAuth usage endpoint (Claude Code consumer OAuth tokens) - const oauthResponse = await proxyAwareFetch(CLAUDE_CONFIG.oauthUsageUrl, { + // cedar_ember=1 adds the "limit reset" grant block (same flag Claude Code sends) + const oauthResponse = await proxyAwareFetch(`${CLAUDE_CONFIG.oauthUsageUrl}?cedar_ember=1`, { method: "GET", headers: { "Authorization": `Bearer ${accessToken}`, "anthropic-beta": "oauth-2025-04-20", "anthropic-version": CLAUDE_CONFIG.apiVersion, + "User-Agent": CLAUDE_CONFIG.userAgent, }, }, proxyOptions); @@ -129,6 +135,7 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) { return { plan: "Claude Code", extraUsage: data.extra_usage ?? null, + resetCredits: parseClaudeResetGrants(data.cedar_ember), quotas, }; } @@ -146,6 +153,70 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) { } } +// Free "limit reset" grants (Anthropic program id "cedar_ember"). +// Shape: { eligible, next_grant_id, grants: [{ id, resets_left, ends_at, paused, clears }] } +export function parseClaudeResetGrants(block) { + if (!block?.eligible || !Array.isArray(block.grants)) return null; + const grants = block.grants.filter((g) => g?.id && !g.paused && Number(g.resets_left) > 0); + const next = grants.find((g) => g.id === block.next_grant_id) || grants[0] || null; + return { + availableCount: grants.reduce((sum, g) => sum + Number(g.resets_left), 0), + nextGrantId: next?.id || null, + expiresAt: next?.ends_at || null, + clears: next?.clears || [], + cooldownUntil: block.cooldown_until || null, + weeklyResetsAt: block.weekly_resets_at || null, + grants: block.grants.filter((g) => g?.id).map((g) => ({ + id: g.id, + label: g.label || "", + resetsLeft: Number(g.resets_left) || 0, + resetsTotal: Number(g.resets_total) || 0, + startsAt: g.starts_at || null, + endsAt: g.ends_at || null, + clears: Array.isArray(g.clears) ? g.clears : [], + paused: g.paused === true, + usableNow: g.usable_now === true, + useRequiresLimit: g.use_requires_limit !== false, + })), + }; +} + +// Spend one reset grant: refills the limits listed in grant.clears. Irreversible. +export async function consumeClaudeResetGrant(accessToken, grantId, proxyOptions = null) { + if (!accessToken) throw new Error("No Claude access token available. Please re-authorize the connection."); + if (!/^[a-z0-9_-]{1,40}$/.test(grantId || "")) throw new Error("Invalid reset grant id."); + + const headers = { + "Authorization": `Bearer ${accessToken}`, + "anthropic-beta": "oauth-2025-04-20", + "anthropic-version": CLAUDE_CONFIG.apiVersion, + "User-Agent": CLAUDE_CONFIG.userAgent, + "Content-Type": "application/json", + }; + + const profileRes = await proxyAwareFetch(CLAUDE_CONFIG.profileUrl, { method: "GET", headers }, proxyOptions); + const profile = await profileRes.json().catch(() => null); + const orgId = profile?.organization?.uuid; + if (!profileRes.ok || !orgId) throw new Error(`Cannot resolve Claude organization (${profileRes.status}).`); + + const res = await proxyAwareFetch(CLAUDE_CONFIG.resetUrl.replace("{org_id}", orgId), { + method: "POST", + headers, + body: JSON.stringify({ program: "cedar_ember", grant_id: grantId, request_id: crypto.randomUUID() }), + }, proxyOptions); + const data = await res.json().catch(() => null); + + usageCache.delete(accessToken); // next read must show refilled limits + return { + ok: res.ok && data?.result === "reset", + status: res.status, + result: data?.result || null, + reason: data?.reason || null, + resetsLeft: data?.resets_left ?? null, + message: data?.error?.message || null, + }; +} + /** * Legacy Claude usage for API key / org admin users */ diff --git a/open-sse/transformer/responsesTransformer.js b/open-sse/transformer/responsesTransformer.js index ac84db20..8b77eaf2 100644 --- a/open-sse/transformer/responsesTransformer.js +++ b/open-sse/transformer/responsesTransformer.js @@ -333,6 +333,9 @@ export function createResponsesApiTransformStream(logger = null) { // Regular text content if (content) { + // The answer starts, so thinking is over. Upstreams that send reasoning via + // reasoning_content never emit "", so close it here rather than at finish. + closeReasoning(controller); if (!state.msgItemAdded[idx]) { state.msgItemAdded[idx] = true; const msgId = `msg_${state.responseId}_${idx}`; @@ -372,6 +375,7 @@ export function createResponsesApiTransformStream(logger = null) { // Handle tool_calls if (delta.tool_calls) { + closeReasoning(controller); closeMessage(controller, idx); for (const tc of delta.tool_calls) { diff --git a/open-sse/translator/concerns/thinkingUnified.js b/open-sse/translator/concerns/thinkingUnified.js index 4bc9601f..5ce73b32 100644 --- a/open-sse/translator/concerns/thinkingUnified.js +++ b/open-sse/translator/concerns/thinkingUnified.js @@ -102,9 +102,28 @@ export function extractThinking(body) { return null; } -// Capture thinking intent from a body. Alias of extractThinking, named for clarity -// at the call-site where intent is snapshotted before format translation. -export const captureThinking = extractThinking; +// Capture thinking intent from a body before format translation strips it. +// Besides the effort, records whether an OpenAI-shaped client wants the thinking +// text itself: Claude returns it only with thinking.display "summarized", a field +// OpenAI has no equivalent for, so the intent cannot survive translation on its own. +export function captureThinking(body) { + const cfg = extractThinking(body); + if (!cfg || cfg.mode === "none") return cfg; + const display = openAIThinkingDisplay(body); + return display ? { ...cfg, display } : cfg; +} + +function openAIThinkingDisplay(body) { + // Responses API: reasoning.summary is the explicit request for reasoning text. + if (body.reasoning && typeof body.reasoning === "object") { + const summary = body.reasoning.summary; + return typeof summary === "string" && summary && summary !== "none" ? "summarized" : undefined; + } + // Chat Completions has no summary knob. A client setting reasoning_effort is + // asking for reasoning, and reasoning_content is how it would receive it. + if (typeof body.reasoning_effort === "string") return "summarized"; + return undefined; +} const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]); @@ -302,9 +321,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) { case "deepseek": { if (none && canDisable) { body.thinking = { type: "disabled" }; break; } body.thinking = { type: "enabled" }; - // DeepSeek: low/medium→high, xhigh/max→max. + // DeepSeek: low/medium→high, xhigh/max→max. Some backends (mimo v2.5-pro/v2.6 + // on opencode-go, probed live) 400 on "max" — clamp to high when the declared + // levels exclude it. const level = toLevel(eff); - body.reasoning_effort = level === "xhigh" || level === "max" ? "max" : "high"; + const want = level === "xhigh" || level === "max" ? "max" : "high"; + body.reasoning_effort = want === "max" && supportedLevels && !supportedLevels.includes("max") ? "high" : want; break; } case "kimi": { @@ -380,7 +402,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent const supportedLevels = getThinkingLevels(provider, cleanModel); // Anthropic's `display` (summarized | omitted) decides whether thinking text // comes back at all; keep what the client asked for instead of resetting it. - const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined; + // An OpenAI-shaped client's ask arrives via the captured intent instead. + const display = typeof body.thinking?.display === "string" ? body.thinking.display : intent?.display; stripAll(body); applyFormat(fmt, body, cfg, caps, supportedLevels, display); return body; diff --git a/open-sse/translator/formats/gemini.js b/open-sse/translator/formats/gemini.js index 729e4c10..412fadcc 100644 --- a/open-sse/translator/formats/gemini.js +++ b/open-sse/translator/formats/gemini.js @@ -432,7 +432,7 @@ export function cleanJSONSchemaForAntigravity(schema) { return cleaned; } -// Merge adjacent same-role messages, strip empty parts, ensure initial user turn +// Merge adjacent same-role messages, strip empty parts, ensure initial and terminal user turns export function normalizeGeminiContents(contents) { const out = []; for (const c of contents || []) { @@ -446,6 +446,23 @@ export function normalizeGeminiContents(contents) { if (out.length > 0 && out[0].role !== "user") { out.unshift({ role: "user", parts: [{ text: "..." }] }); } + if (out.length > 0 && out.at(-1).role === "model") { + const fnCalls = (out.at(-1).parts || []).filter(p => p && p.functionCall); + if (fnCalls.length > 0) { + const responses = fnCalls.map(p => { + const call = p.functionCall || {}; + const fr = { + name: call.name || "tool", + response: { result: "Continue." } + }; + if (call.id) fr.id = call.id; + return { functionResponse: fr }; + }); + out.push({ role: "user", parts: responses }); + } else { + out.push({ role: "user", parts: [{ text: "Continue." }] }); + } + } return out; } diff --git a/open-sse/translator/response/claude-to-openai.js b/open-sse/translator/response/claude-to-openai.js index 80de2f18..42a82a66 100644 --- a/open-sse/translator/response/claude-to-openai.js +++ b/open-sse/translator/response/claude-to-openai.js @@ -59,10 +59,6 @@ export function claudeToOpenAIResponse(chunk, state) { } if (block?.type === CLAUDE_BLOCK.TEXT) { state.textBlockStarted = true; - } else if (block?.type === CLAUDE_BLOCK.THINKING) { - state.inThinkingBlock = true; - state.currentBlockIndex = chunk.index; - results.push(createChunk(state, { content: "" })); } else if (block?.type === CLAUDE_BLOCK.TOOL_USE) { const toolCallIndex = state.toolCallIndex++; // Restore original tool name from mapping (Claude OAuth) @@ -89,6 +85,8 @@ export function claudeToOpenAIResponse(chunk, state) { if (delta?.type === "text_delta" && delta.text) { results.push(createChunk(state, { content: delta.text })); } else if (delta?.type === "thinking_delta" && delta.thinking) { + // Thinking travels only in reasoning_content. No "" markers in + // content: OpenAI-format clients render them as literal text. results.push(createChunk(state, reasoningDelta(delta.thinking))); } else if (delta?.type === "input_json_delta" && delta.partial_json) { const toolCall = state.toolCalls.get(chunk.index); @@ -112,10 +110,6 @@ export function claudeToOpenAIResponse(chunk, state) { state.serverToolBlockIndex = -1; break; } - if (state.inThinkingBlock && chunk.index === state.currentBlockIndex) { - results.push(createChunk(state, { content: "" })); - state.inThinkingBlock = false; - } state.textBlockStarted = false; state.thinkingBlockStarted = false; break; diff --git a/open-sse/translator/response/openai-responses.js b/open-sse/translator/response/openai-responses.js index c3d16448..6a865d1a 100644 --- a/open-sse/translator/response/openai-responses.js +++ b/open-sse/translator/response/openai-responses.js @@ -129,12 +129,16 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { } if (content) { + // The answer starts, so thinking is over. Upstreams that send reasoning via + // reasoning_content never emit "", so close it here rather than at finish. + closeReasoning(state, emit); emitTextContent(state, emit, idx, content); } } // Handle tool_calls (empty array is truthy; require a real call) if (delta.tool_calls && delta.tool_calls.length) { + closeReasoning(state, emit); closeMessage(state, emit, idx); for (const tc of delta.tool_calls) { emitToolCall(state, emit, tc); @@ -219,15 +223,19 @@ function closeReasoning(state, emit) { part: { type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf } }); + const item = { + id: state.reasoningId, + type: RESPONSES_ITEM.REASONING, + summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }] + }; + emit("response.output_item.done", { type: "response.output_item.done", output_index: state.reasoningIndex, - item: { - id: state.reasoningId, - type: RESPONSES_ITEM.REASONING, - summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }] - } + item }); + + recordCompletedOutputItem(state, state.reasoningIndex, item); } } @@ -291,16 +299,20 @@ function closeMessage(state, emit, idx) { part: { type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText } }); + const item = { + id: msgId, + type: RESPONSES_ITEM.MESSAGE, + content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }], + role: ROLE.ASSISTANT + }; + emit("response.output_item.done", { type: "response.output_item.done", output_index: parseInt(idx), - item: { - id: msgId, - type: RESPONSES_ITEM.MESSAGE, - content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }], - role: ROLE.ASSISTANT - } + item }); + + recordCompletedOutputItem(state, parseInt(idx), item); } } @@ -394,23 +406,51 @@ function closeToolCall(state, emit, idx) { }); } + const item = { + id: `${custom ? "ctc" : "fc"}_${callId}`, + type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL, + ...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }), + call_id: callId, + name: state.funcNames[idx] || "" + }; + emit("response.output_item.done", { type: "response.output_item.done", output_index: parseInt(idx), - item: { - id: `${custom ? "ctc" : "fc"}_${callId}`, - type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL, - ...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }), - call_id: callId, - name: state.funcNames[idx] || "" - } + item }); + recordCompletedOutputItem(state, parseInt(idx), item); + state.funcItemDone[idx] = true; state.funcArgsDone[idx] = true; } } +// response.completed carries the finished Response object, so response.output has +// to repeat the items already delivered in response.output_item.done. Clients that +// build their final result from the terminal event (GitHub Copilot CLI, the OpenAI +// SDK "final response" helpers) otherwise treat the turn as empty even though the +// text was streamed - see issue #4307. +// +// Keyed by output_index so a repeated close overwrites rather than duplicating the +// item, and ordered by output_index so response.output matches the order the items +// were emitted in. Lazily created because stream.js can hand us a state it built +// itself rather than one from initState(). +function recordCompletedOutputItem(state, outputIndex, item) { + state.completedOutputItems ??= new Map(); + const index = Number.isInteger(outputIndex) ? outputIndex : Number.parseInt(outputIndex, 10) || 0; + state.completedOutputItems.set(index, item); +} + +function collectCompletedOutputItems(state) { + const recorded = state.completedOutputItems; + if (!(recorded instanceof Map) || recorded.size === 0) return []; + return [...recorded.entries()] + .sort((left, right) => left[0] - right[0]) + .map(([, item]) => item); +} + function sendCompleted(state, emit) { if (!state.completedSent) { state.completedSent = true; @@ -423,6 +463,7 @@ function sendCompleted(state, emit) { status: "completed", background: false, error: null, + output: collectCompletedOutputItems(state), ...(state.responsesUsage ? { usage: state.responsesUsage } : {}) } }); diff --git a/open-sse/utils/claudeCloaking.js b/open-sse/utils/claudeCloaking.js index 7e3790df..60e40b79 100644 --- a/open-sse/utils/claudeCloaking.js +++ b/open-sse/utils/claudeCloaking.js @@ -25,10 +25,25 @@ function deriveUuid(seed) { function generateFakeUserID(sessionId, apiKey) { const deviceId = apiKey ? createHash("sha256").update(`device:${apiKey}`).digest("hex") : randomBytes(32).toString("hex"); const accountUuid = apiKey ? deriveUuid(`account:${apiKey}`) : randomUUID(); - const sessionUuid = sessionId || randomUUID(); + const cleanSessionId = typeof sessionId === "string" ? sessionId.replace(/^claude:/i, "").trim() : null; + const sessionUuid = cleanSessionId || randomUUID(); return `{"device_id":"${deviceId}","account_uuid":"${accountUuid}","session_id":"${sessionUuid}"}`; } +export function extractClaudeSessionIdFromUserId(userId) { + if (typeof userId !== "string" || !userId) return null; + if (userId[0] === "{") { + try { + const sid = JSON.parse(userId)?.session_id; + return typeof sid === "string" && sid ? sid.replace(/^claude:/i, "").trim() || null : null; + } catch { + return null; + } + } + const clean = userId.replace(/^claude:/i, "").trim(); + return clean || null; +} + /** * Cloak tools before sending to Claude provider (anti-ban): * - Rename client tools with the CLAUDE_TOOL_SUFFIX ("_ide") in tools[] and messages[] @@ -89,14 +104,29 @@ export function cloakClaudeTools(body) { }; } +// Strip a trailing CLAUDE_TOOL_SUFFIX from a cloaked name as a last-resort +// fallback when the name isn't in toolNameMap (e.g. map lost across a retry/ +// reconnect). Never strips decoy names — those are meant to reach the client +// unresolved so it can see "tool unavailable" instead of silently no-oping. +function stripCloakSuffix(name) { + if (typeof name !== "string" || !name.endsWith(CLAUDE_TOOL_SUFFIX)) return null; + if (CC_DEFAULT_TOOLS.has(name)) return null; + const original = name.slice(0, -CLAUDE_TOOL_SUFFIX.length); + return original.length > 0 ? original : null; +} + // Decloak tool_use names in non-streaming Claude response body (INPUT side) export function decloakToolNames(body, toolNameMap) { - if (!toolNameMap?.size || !Array.isArray(body?.content)) return body; + if (!Array.isArray(body?.content)) return body; const content = body.content.map(block => { - if (block?.type === "tool_use" && toolNameMap.has(block.name)) { + if (block?.type !== "tool_use") return block; + if (toolNameMap?.has(block.name)) { return { ...block, name: toolNameMap.get(block.name) }; } - return block; + // toolNameMap missing/stale for this name — fall back to suffix stripping + // rather than forwarding an unresolvable "_ide" name to the client. + const fallback = stripCloakSuffix(block.name); + return fallback ? { ...block, name: fallback } : block; }); return { ...body, content }; } @@ -111,19 +141,21 @@ export function decloakToolNames(body, toolNameMap) { * name appears exactly once per call — on the content_block_start event of * a tool_use block; argument deltas carry no name. * - * Unknown names (e.g. a CC decoy tool the model called anyway) pass through - * unchanged, matching the non-streaming decloak behavior. + * Falls back to stripping the literal CLAUDE_TOOL_SUFFIX when the name isn't + * in toolNameMap (map lost across a retry/reconnect), matching the + * non-streaming decloak behavior. Decoy tool names (real CC tool names) and + * anything else pass through unchanged. * * @param {object|null} chunk - Parsed SSE event (may be null on stream flush) * @param {Map|null} toolNameMap - Suffixed → original name map from cloakClaudeTools() * @returns {object|null} The chunk, with the tool_use name restored when cloaked */ export function decloakStreamChunk(chunk, toolNameMap) { - if (!toolNameMap?.size || !chunk || typeof chunk !== "object") return chunk; + if (!chunk || typeof chunk !== "object") return chunk; if (chunk.type !== "content_block_start") return chunk; const block = chunk.content_block; if (block?.type !== "tool_use" || typeof block.name !== "string") return chunk; - const original = toolNameMap.get(block.name); + const original = toolNameMap?.get(block.name) || stripCloakSuffix(block.name); if (!original) return chunk; return { ...chunk, content_block: { ...block, name: original } }; } diff --git a/open-sse/utils/error.js b/open-sse/utils/error.js index 315723e3..6992d4a4 100644 --- a/open-sse/utils/error.js +++ b/open-sse/utils/error.js @@ -27,12 +27,13 @@ export function buildErrorBody(statusCode, message) { * @param {string} message - Error message * @returns {Response} HTTP Response object */ -export function errorResponse(statusCode, message) { +export function errorResponse(statusCode, message, extraHeaders = null) { return new Response(JSON.stringify(buildErrorBody(statusCode, message)), { status: statusCode, headers: { "Content-Type": "application/json", - "Access-Control-Allow-Origin": "*" + "Access-Control-Allow-Origin": "*", + ...extraHeaders } }); } @@ -95,13 +96,13 @@ export async function parseUpstreamError(response, executor = null) { * @param {number} [resetsAtMs] - Optional precise cooldown expiry (ms epoch) for provider-specific quota errors * @returns {{ success: false, status: number, error: string, response: Response, resetsAtMs?: number }} */ -export function createErrorResult(statusCode, message, resetsAtMs) { +export function createErrorResult(statusCode, message, resetsAtMs, extraHeaders = null) { return { success: false, status: statusCode, error: message, resetsAtMs, - response: errorResponse(statusCode, message) + response: errorResponse(statusCode, message, extraHeaders) }; } @@ -113,7 +114,7 @@ export function createErrorResult(statusCode, message, resetsAtMs) { * @param {string} retryAfterHuman - Human-readable retry info e.g. "reset after 30s" * @returns {Response} */ -export function unavailableResponse(statusCode, message, retryAfter, retryAfterHuman) { +export function unavailableResponse(statusCode, message, retryAfter, retryAfterHuman, extraHeaders = null) { const retryAfterSec = Math.max(Math.ceil((new Date(retryAfter).getTime() - Date.now()) / 1000), 1); const msg = `${message} (${retryAfterHuman})`; return new Response( @@ -121,8 +122,10 @@ export function unavailableResponse(statusCode, message, retryAfter, retryAfterH { status: statusCode, headers: { + ...extraHeaders, "Content-Type": "application/json", - "Retry-After": String(retryAfterSec) + // Intentionally mis-cased to prevent duplicate headers + "retry-after": String(retryAfterSec) } } ); diff --git a/open-sse/utils/upstreamHeaders.js b/open-sse/utils/upstreamHeaders.js new file mode 100644 index 00000000..49ec3188 --- /dev/null +++ b/open-sse/utils/upstreamHeaders.js @@ -0,0 +1,12 @@ +const FORWARDED = new Set(["retry-after", "x-should-retry"]); +const FORWARDED_PREFIX = "anthropic-ratelimit-"; + +export function upstreamResponseHeaders(headers) { + const out = {}; + if (typeof headers?.forEach !== "function") return out; + headers.forEach((value, name) => { + const key = name.toLowerCase(); + if (FORWARDED.has(key) || key.startsWith(FORWARDED_PREFIX)) out[key] = value; + }); + return out; +} diff --git a/package.json b/package.json index d7d4013b..037dcab7 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "9router-app", - "version": "0.5.86", + "version": "0.5.91", "description": "9Router web dashboard", "private": true, "scripts": { diff --git a/public/providers/agnes.png b/public/providers/agnes.png new file mode 100644 index 00000000..8615a313 Binary files /dev/null and b/public/providers/agnes.png differ diff --git a/public/providers/atria.png b/public/providers/atria.png new file mode 100644 index 00000000..8ef034e3 Binary files /dev/null and b/public/providers/atria.png differ diff --git a/public/providers/bai.png b/public/providers/bai.png new file mode 100644 index 00000000..878045c5 Binary files /dev/null and b/public/providers/bai.png differ diff --git a/public/providers/dahl.png b/public/providers/dahl.png new file mode 100644 index 00000000..fbcf4270 Binary files /dev/null and b/public/providers/dahl.png differ diff --git a/public/providers/hermes.png b/public/providers/hermes.png index d108d0c3..1854f967 100644 Binary files a/public/providers/hermes.png and b/public/providers/hermes.png differ diff --git a/public/providers/tokenharbor.png b/public/providers/tokenharbor.png new file mode 100644 index 00000000..7a766a64 Binary files /dev/null and b/public/providers/tokenharbor.png differ diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/BaseUrlSelect.js b/src/app/(dashboard)/dashboard/cli-tools/components/BaseUrlSelect.js index 52ca8b09..cfea9c95 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/BaseUrlSelect.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/BaseUrlSelect.js @@ -57,6 +57,7 @@ export default function BaseUrlSelect({ const [mode, setMode] = useState(""); const [customInput, setCustomInput] = useState(""); const initializedRef = useRef(false); + const currentUrlRef = useRef(""); const customInputRef = useRef(""); useEffect(() => { @@ -85,23 +86,34 @@ export default function BaseUrlSelect({ [requiresExternalUrl, tunnelEnabled, tunnelPublicUrl, tailscaleEnabled, tailscaleUrl, cloudEnabled, cloudUrl, savedPresets, withV1] ); - // Prefer a saved preset matching the currently configured URL, else first option + // Sync the active config URL without replacing edits unless the config itself changes. useEffect(() => { - if (initializedRef.current) return; if (!presetsLoaded || options.length === 0) return; + const normalizeUrl = (url) => (withV1 ? ensureV1(url) : stripSlash(url)); + const current = normalizeUrl(currentUrl); + if (initializedRef.current && currentUrlRef.current === current) return; initializedRef.current = true; - const current = stripSlash(currentUrl); + currentUrlRef.current = current; const matched = current - ? options.find((o) => o.saved && stripSlash(o.url) === current) + ? options.find((o) => o.value !== CUSTOM_VALUE && normalizeUrl(o.url) === current) : null; - const target = matched || options.find((o) => o.value !== CUSTOM_VALUE); - if (target) { + if (matched) { + setCustomInput(""); + customInputRef.current = ""; + setMode(matched.value); + onChange(matched.url); + } else if (current) { + setCustomInput(current); + customInputRef.current = current; + setMode(CUSTOM_VALUE); + onChange(current); + } else { + const target = options.find((o) => o.value !== CUSTOM_VALUE); + if (!target) return; setMode(target.value); onChange(target.url); - } else { - setMode(CUSTOM_VALUE); } - }, [presetsLoaded, options, onChange, currentUrl]); + }, [presetsLoaded, options, onChange, currentUrl, withV1]); const handleSelect = (e) => { const next = e.target.value; diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/CodexToolCard.js b/src/app/(dashboard)/dashboard/cli-tools/components/CodexToolCard.js index 6fed52de..f0cdd32b 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/CodexToolCard.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/CodexToolCard.js @@ -3,10 +3,12 @@ import { useState, useEffect } from "react"; import { Card, Button, ModelSelectModal, ManualConfigModal } from "@/shared/components"; import Image from "next/image"; +import ProviderIcon from "@/shared/components/ProviderIcon"; import BaseUrlSelect from "./BaseUrlSelect"; import ApiKeySelect from "./ApiKeySelect"; import { matchKnownEndpoint } from "./cliEndpointMatch"; import { rememberEndpoint } from "./cliEndpointPresets"; +import { getCurrentCodexProviderSettings, deriveProfileNameFromModel } from "./codexConfig"; export default function CodexToolCard({ tool, isExpanded, onToggle, baseUrl, apiKeys, activeProviders, cloudEnabled, initialStatus, tunnelEnabled, tunnelPublicUrl, tailscaleEnabled, tailscaleUrl }) { const [codexStatus, setCodexStatus] = useState(initialStatus || null); @@ -23,12 +25,23 @@ export default function CodexToolCard({ tool, isExpanded, onToggle, baseUrl, api const [modelAliases, setModelAliases] = useState({}); const [showManualConfigModal, setShowManualConfigModal] = useState(false); const [customBaseUrl, setCustomBaseUrl] = useState(""); + const [profiles, setProfiles] = useState([]); + const [aliasInput, setAliasInput] = useState(""); + const [modelInput, setModelInput] = useState(""); + const [profileModalOpen, setProfileModalOpen] = useState(false); + const [creatingProfile, setCreatingProfile] = useState(false); + const [deletingProfile, setDeletingProfile] = useState(null); + const [copiedCommand, setCopiedCommand] = useState(""); useEffect(() => { - if (apiKeys?.length > 0 && !selectedApiKey) { + fetchProfiles(); + }, []); + + useEffect(() => { + if (apiKeys?.length > 0 && !selectedApiKey && !codexStatus?.config) { setSelectedApiKey(apiKeys[0].key); } - }, [apiKeys, selectedApiKey]); + }, [apiKeys, selectedApiKey, codexStatus?.config]); useEffect(() => { if (initialStatus) setCodexStatus(initialStatus); @@ -38,6 +51,7 @@ export default function CodexToolCard({ tool, isExpanded, onToggle, baseUrl, api if (isExpanded) { if (!codexStatus) checkCodexStatus(); fetchModelAliases(); + fetchProfiles(); } }, [isExpanded]); @@ -51,24 +65,101 @@ export default function CodexToolCard({ tool, isExpanded, onToggle, baseUrl, api } }; - // Parse model and subagent settings from config content + const fetchProfiles = async () => { + try { + const res = await fetch("/api/cli-tools/codex-profiles"); + const data = await res.json(); + if (res.ok) setProfiles(data.profiles || []); + } catch (error) { + console.log("Error fetching codex profiles:", error); + } + }; + + const handleModelSelectForAlias = (model) => { + setProfileModalOpen(false); + const selectedModelId = model?.value || model?.id; + if (!selectedModelId) return; + + setModelInput(selectedModelId); + const providerName = model?.provider || (selectedModelId.includes("/") ? selectedModelId.split("/")[0] : selectedModelId); + const existingNames = profiles.map((p) => p.name); + setAliasInput(deriveProfileNameFromModel(providerName, existingNames)); + }; + + const handleAddProfileWithAlias = async () => { + const cleanAlias = aliasInput.trim().toLowerCase().replace(/[^a-z0-9_-]/g, ""); + const cleanModel = modelInput.trim(); + if (!cleanAlias || !cleanModel) return; + + setCreatingProfile(true); + try { + const res = await fetch("/api/cli-tools/codex-profiles", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + name: cleanAlias, + model: cleanModel, + }), + }); + const data = await res.json(); + if (res.ok) { + setAliasInput(""); + setModelInput(""); + fetchProfiles(); + } else { + setMessage({ type: "error", text: data.error || "Failed to add model" }); + } + } catch (error) { + setMessage({ type: "error", text: error.message }); + } finally { + setCreatingProfile(false); + } + }; + + const handleDeleteProfile = async (name) => { + setDeletingProfile(name); + try { + const res = await fetch("/api/cli-tools/codex-profiles", { + method: "DELETE", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ name }), + }); + if (res.ok) fetchProfiles(); + } catch (error) { + console.log("Error deleting codex profile:", error); + } finally { + setDeletingProfile(null); + } + }; + + const handleCopyCommand = async (cmd) => { + try { + await navigator.clipboard.writeText(cmd); + setCopiedCommand(cmd); + setTimeout(() => setCopiedCommand(""), 2000); + } catch (e) { + console.log("Copy failed", e); + } + }; + + // Sync only when config content changes so local form edits are retained. useEffect(() => { - if (codexStatus?.config) { - const modelMatch = codexStatus.config.match(/^model\s*=\s*"([^"]+)"/m); + const config = codexStatus?.config; + if (config) { + const { baseUrl, apiKey } = getCurrentCodexProviderSettings(config); + setCustomBaseUrl(baseUrl); + setSelectedApiKey(apiKey); + + const modelMatch = config.match(/^model\s*=\s*"([^"]+)"/m); if (modelMatch) setSelectedModel(modelMatch[1]); // Parse subagent settings - const subagentModelMatch = codexStatus.config.match(/^default_subagent_model\s*=\s*"([^"]+)"/m); + const subagentModelMatch = config.match(/^default_subagent_model\s*=\s*"([^"]+)"/m); if (subagentModelMatch) setSubagentModel(subagentModelMatch[1]); } - }, [codexStatus]); + }, [codexStatus?.config]); - const getCurrentBaseUrl = () => { - const parsed = codexStatus?.config?.match(/base_url\s*=\s*"([^"]+)"/); - return parsed ? parsed[1] : ""; - }; - - const currentBaseUrl = getCurrentBaseUrl(); + const currentBaseUrl = getCurrentCodexProviderSettings(codexStatus?.config).baseUrl; const getConfigStatus = () => { if (!codexStatus?.installed) return null; @@ -79,7 +170,7 @@ export default function CodexToolCard({ tool, isExpanded, onToggle, baseUrl, api const configStatus = getConfigStatus(); const getEffectiveBaseUrl = () => { - const url = customBaseUrl || `${baseUrl}/v1`; + const url = (customBaseUrl || `${baseUrl}/v1`).replace(/\/+$/, ""); // Ensure URL ends with /v1 return url.endsWith("/v1") ? url : `${url}/v1`; }; @@ -89,7 +180,7 @@ export default function CodexToolCard({ tool, isExpanded, onToggle, baseUrl, api const checkCodexStatus = async () => { setCheckingCodex(true); try { - const res = await fetch("/api/cli-tools/codex-settings"); + const res = await fetch("/api/cli-tools/codex-settings", { cache: "no-store" }); const data = await res.json(); setCodexStatus(data); } catch (error) { @@ -366,6 +457,148 @@ default_subagent_model = "${effectiveSubagentModel}" content_copyManual Config + + {/* Additional Models */} +
+
+
+ + layers + Additional Models + + {profiles.length > 0 && ( + + {profiles.length} + + )} +
+
+ + {/* Quick Add Model bar */} +
+
+ setAliasInput(e.target.value.toLowerCase().replace(/[^a-z0-9_-]/g, ""))} + placeholder="Alias (e.g. claude)" + className="w-full min-w-0 px-2.5 py-1.5 bg-surface rounded border border-border text-xs font-mono focus:outline-none focus:ring-1 focus:ring-primary/50" + onKeyDown={(e) => e.key === "Enter" && handleAddProfileWithAlias()} + /> +
+ { + const val = e.target.value; + setModelInput(val); + const provider = val.includes("/") ? val.split("/")[0] : val; + setAliasInput(deriveProfileNameFromModel(provider, profiles.map((p) => p.name))); + }} + placeholder="provider/model-id" + className="w-full min-w-0 pl-2.5 pr-7 py-1.5 bg-surface rounded border border-border text-xs focus:outline-none focus:ring-1 focus:ring-primary/50" + onKeyDown={(e) => e.key === "Enter" && handleAddProfileWithAlias()} + /> + {modelInput && ( + + )} +
+ + +
+
+ + {profiles.length === 0 ? ( +
+ + terminal + +

+ Only the main model is active. Add an alias above to configure more models for Codex CLI. +

+
+ ) : ( +
+ {profiles.map((p) => { + const providerId = p.model.includes("/") ? p.model.split("/")[0] : p.name; + const isCopied = copiedCommand === p.command; + return ( +
+
+
+ +
+
+ + {p.name} + + + {p.model} + +
+
+ +
+ + + +
+
+ ); + })} +
+ )} +
)} @@ -395,6 +628,18 @@ default_subagent_model = "${effectiveSubagentModel}" /> )} + {profileModalOpen && ( + setProfileModalOpen(false)} + onSelect={handleModelSelectForAlias} + selectedModel={modelInput} + activeProviders={activeProviders} + modelAliases={modelAliases} + title="Select Model for Codex CLI" + /> + )} + setShowManualConfigModal(false)} diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/HermesToolCard.js b/src/app/(dashboard)/dashboard/cli-tools/components/HermesToolCard.js index bf5c25d1..74758980 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/HermesToolCard.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/HermesToolCard.js @@ -7,8 +7,10 @@ import BaseUrlSelect from "./BaseUrlSelect"; import { rememberEndpoint } from "./cliEndpointPresets"; import ApiKeySelect from "./ApiKeySelect"; import { matchKnownEndpoint } from "./cliEndpointMatch"; +import { CLI_TOOLS } from "@/shared/constants/cliTools"; const ENDPOINT = "/api/cli-tools/hermes-settings"; +const HERMES_ROLES = CLI_TOOLS.hermes?.roles || []; export default function HermesToolCard({ tool, @@ -32,6 +34,8 @@ export default function HermesToolCard({ const [message, setMessage] = useState(null); const [selectedApiKey, setSelectedApiKey] = useState(""); const [selectedModel, setSelectedModel] = useState(""); + const [roleModels, setRoleModels] = useState({}); + const [modalTarget, setModalTarget] = useState("default"); const [modalOpen, setModalOpen] = useState(false); const [modelAliases, setModelAliases] = useState({}); const [showManualConfigModal, setShowManualConfigModal] = useState(false); @@ -82,6 +86,12 @@ export default function HermesToolCard({ hasInitializedModel.current = true; const cfg = hermesStatus.settings?.model; if (cfg?.default) setSelectedModel(cfg.default); + const initial = {}; + if (hermesStatus.settings?.delegation?.model) initial.delegation = hermesStatus.settings.delegation.model; + for (const [role, rcfg] of Object.entries(hermesStatus.settings?.auxiliary || {})) { + if (rcfg?.model) initial[role] = rcfg.model; + } + setRoleModels(initial); } }, [hermesStatus]); @@ -126,7 +136,12 @@ export default function HermesToolCard({ body: JSON.stringify({ baseUrl: getEffectiveBaseUrl(), apiKey: keyToUse, - model: selectedModel, + selections: [ + { role: "default", model: selectedModel }, + ...Object.entries(roleModels) + .filter(([, model]) => model?.trim()) + .map(([role, model]) => ({ role, model: model.trim() })), + ], }), }); const data = await res.json(); @@ -154,6 +169,7 @@ export default function HermesToolCard({ if (res.ok) { setMessage({ type: "success", text: "Settings reset successfully!" }); setSelectedModel(""); + setRoleModels({}); checkStatus(); } else { setMessage({ type: "error", text: data.error || "Failed to reset settings" }); @@ -166,16 +182,35 @@ export default function HermesToolCard({ }; const handleModelSelect = (model) => { - setSelectedModel(model.value); + if (modalTarget === "default") { + setSelectedModel(model.value); + } else { + setRoleModels((prev) => ({ ...prev, [modalTarget]: model.value })); + } setModalOpen(false); }; + const openModelModal = (target) => { + setModalTarget(target); + setModalOpen(true); + }; + const getManualConfigs = () => { const keyToUse = (selectedApiKey && selectedApiKey.trim()) ? selectedApiKey : (!cloudEnabled ? "sk_9router" : ""); - const yamlContent = `model:\n default: "${selectedModel || "provider/model-id"}"\n provider: "custom"\n base_url: "${getEffectiveBaseUrl()}"\n api_key: \${OPENAI_API_KEY}\n`; + const base = getEffectiveBaseUrl(); + let yamlContent = `model:\n default: "${selectedModel || "provider/model-id"}"\n provider: "custom"\n base_url: "${base}"\n api_key: \${OPENAI_API_KEY}\n`; + if (roleModels.delegation?.trim()) { + yamlContent += `delegation:\n model: "${roleModels.delegation.trim()}"\n provider: "custom"\n base_url: "${base}"\n api_key: \${OPENAI_API_KEY}\n`; + } + const auxRoles = Object.entries(roleModels).filter(([role, model]) => role !== "delegation" && model?.trim()); + if (auxRoles.length > 0) { + yamlContent += `auxiliary:\n${auxRoles.map(([role, model]) => + ` ${role}:\n provider: "custom"\n model: "${model.trim()}"\n base_url: "${base}"\n api_key: \${OPENAI_API_KEY}\n` + ).join("")}`; + } const envContent = `OPENAI_API_KEY=${keyToUse}\n`; return [ @@ -274,8 +309,49 @@ export default function HermesToolCard({ setSelectedModel(e.target.value)} placeholder="provider/model-id" className="w-full min-w-0 pl-2 pr-7 py-2 bg-surface rounded border border-border text-xs focus:outline-none focus:ring-1 focus:ring-primary/50 sm:py-1.5" /> {selectedModel && } - + + +
+ + chevron_right + Model Roles (optional) + +
+ {HERMES_ROLES.map((role) => ( +
+ {role.label} + arrow_forward +
+ setRoleModels((prev) => ({ ...prev, [role.id]: e.target.value }))} + placeholder="inherit default" + className="w-full min-w-0 pl-2 pr-7 py-2 bg-surface rounded border border-border text-xs focus:outline-none focus:ring-1 focus:ring-primary/50 sm:py-1.5" + /> + {roleModels[role.id] && ( + + )} +
+ +
+ ))} +

Empty roles inherit the default model.

+
+
{message && ( @@ -306,10 +382,10 @@ export default function HermesToolCard({ isOpen={modalOpen} onClose={() => setModalOpen(false)} onSelect={handleModelSelect} - selectedModel={selectedModel} + selectedModel={modalTarget === "default" ? selectedModel : roleModels[modalTarget] || ""} activeProviders={activeProviders} modelAliases={modelAliases} - title="Select Model for Hermes Agent" + title={`Select Model for Hermes Agent${modalTarget !== "default" ? ` — ${HERMES_ROLES.find((r) => r.id === modalTarget)?.label || modalTarget}` : ""}`} /> )} diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/codexConfig.js b/src/app/(dashboard)/dashboard/cli-tools/components/codexConfig.js new file mode 100644 index 00000000..2856e8a1 --- /dev/null +++ b/src/app/(dashboard)/dashboard/cli-tools/components/codexConfig.js @@ -0,0 +1,86 @@ +const parseTomlString = (line, key) => { + const match = line.match(new RegExp(`^\\s*${key}\\s*=\\s*(["'])([^\\n]*?)\\1\\s*(?:#.*)?$`)); + return match ? match[2] : ""; +}; + +// Only inspect the active provider tables so other providers cannot affect the form. +export function getCurrentCodexProviderSettings(config) { + if (typeof config !== "string") return { baseUrl: "", apiKey: "" }; + + const lines = config.split(/\r?\n/); + let modelProvider = ""; + let inRootTable = true; + + for (const line of lines) { + if (/^\s*\[/.test(line)) { + inRootTable = false; + continue; + } + if (inRootTable) { + modelProvider = parseTomlString(line, "model_provider") || modelProvider; + } + } + + if (!modelProvider) return { baseUrl: "", apiKey: "" }; + + const activeTable = `model_providers.${modelProvider}`; + let inActiveProviderTable = false; + let inActiveHeadersTable = false; + let baseUrl = ""; + let apiKey = ""; + + for (const line of lines) { + const tableMatch = line.match(/^\s*\[\s*([^\]]+?)\s*\]\s*(?:#.*)?$/); + if (tableMatch) { + inActiveProviderTable = tableMatch[1] === activeTable; + inActiveHeadersTable = tableMatch[1] === `${activeTable}.http_headers`; + continue; + } + if (inActiveProviderTable) { + baseUrl = parseTomlString(line, "base_url") || baseUrl; + } + if (inActiveHeadersTable) { + const authorization = parseTomlString(line, "Authorization"); + const bearerMatch = authorization.match(/^Bearer\s+(.+)$/i); + apiKey = bearerMatch ? bearerMatch[1] : apiKey; + } + } + + return { baseUrl, apiKey }; +} + +export function getCurrentCodexProviderBaseUrl(config) { + return getCurrentCodexProviderSettings(config).baseUrl; +} + +export function deriveProfileNameFromModel(modelId, existingNames = []) { + if (!modelId || typeof modelId !== "string") return "model"; + let base = ""; + const trimmed = modelId.trim(); + if (trimmed.includes("/")) { + base = trimmed.split("/")[0].toLowerCase().replace(/[^a-z0-9_-]/g, ""); + } else { + base = trimmed.toLowerCase().replace(/[^a-z0-9_-]/g, ""); + } + base = base.replace(/^[-_]+|[-_]+$/g, "") || "model"; + if (base === "config") base = "model-config"; + + let candidate = base; + let counter = 2; + while (existingNames.includes(candidate)) { + candidate = `${base}-${counter}`; + counter++; + } + return candidate; +} + +export function buildCodexProfileToml({ name, model }) { + return `# codex -p ${name}\nmodel = "${model}"\nmodel_provider = "9router"\n`; +} + +export function parseCodexProfileModel(content) { + if (typeof content !== "string") return ""; + const match = content.match(/^\s*model\s*=\s*(["'])([^"\n]+)\1/m); + return match ? match[2] : ""; +} + diff --git a/src/app/(dashboard)/dashboard/combos/page.js b/src/app/(dashboard)/dashboard/combos/page.js index 498a772d..2d26eccc 100644 --- a/src/app/(dashboard)/dashboard/combos/page.js +++ b/src/app/(dashboard)/dashboard/combos/page.js @@ -687,7 +687,10 @@ function ComboCard({ combo, getCaps, comboByName = {}, activeProviders = [], cop const current = strategy.fallbackStrategy || globalStrategy || "fallback"; const judge = strategy.judgeModel || ""; const isFusion = current === "fusion"; - const comboCaps = aggregateComboCapabilities(combo.models, comboByName); + // The synced catalog is server-only, so resolving here would fall back to the + // generic patterns and under-report the limits. getCaps carries the server's + // answer for /api/models. + const comboCaps = aggregateComboCapabilities(combo.models, comboByName, getCaps); return ( @@ -718,7 +721,7 @@ function ComboCard({ combo, getCaps, comboByName = {}, activeProviders = [], cop {model} @@ -894,8 +897,15 @@ function CapacityAdapterCap({ cap, entry, onChange, activeProviders, getCaps }) const patch = (p) => onChange({ ...entry, ...p }); const handleAdd = (model) => { - if (models.includes(model.value)) return; - patch({ models: [...models, model.value] }); + const value = model?.value || model?.name || model; + if (!value || models.includes(value)) return; + patch({ models: [...models, value] }); + }; + + const handleDeselect = (model) => { + const value = model?.value || model?.name || model; + const next = models.filter((m) => m !== value); + patch({ models: next.length === 0 ? [DEFAULT_FALLBACK_MODEL] : next }); }; const handleRemove = (index) => { @@ -914,7 +924,7 @@ function CapacityAdapterCap({ cap, entry, onChange, activeProviders, getCaps }) return (
- {/* Master toggle + icon + label + chips */} + {/* Master toggle + icon + label */}
{cap.label} — {cap.desc}
-
- {models.length === 0 ? ( - No models - ) : ( - models.slice(0, 3).map((model, index) => ( - - {model} - - - - - - )) - )} - {models.length > 3 && ( - +{models.length - 3} more - )} -
@@ -983,11 +966,97 @@ function CapacityAdapterCap({ cap, entry, onChange, activeProviders, getCaps }) + {/* Model pool list/table */} + {models.length === 0 ? ( +
+ No models in pool (will fallback to {DEFAULT_FALLBACK_MODEL}) +
+ ) : ( +
+ + + + + + + + + + + {models.map((model, index) => ( + + + + + + + ))} + +
#ModelOrder
+ #{index + 1} + +
+ {model} + + {model === DEFAULT_FALLBACK_MODEL && ( + + free default + + )} +
+
+
+ + +
+
+ +
+
+ )} + {showModelSelect && ( setShowModelSelect(false)} onSelect={handleAdd} + onDeselect={handleDeselect} activeProviders={activeProviders} title={`Add ${cap.label} Model`} addedModelValues={models} diff --git a/src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.js b/src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.js index 21017449..989aaf22 100644 --- a/src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.js +++ b/src/app/(dashboard)/dashboard/endpoint/EndpointPageClient.js @@ -13,6 +13,7 @@ import { CLIENT_PING_FAST_MS, } from "./endpointConstants"; import { clientPingUrl, clientPingAny } from "./endpointPing"; +import useSettingsStore from "@/store/settingsStore"; import EndpointRow from "./components/EndpointRow"; import StatusAlert from "./components/StatusAlert"; import Tooltip from "./components/Tooltip"; @@ -211,16 +212,15 @@ export default function APIPageClient({ machineId }) { const loadSettings = async () => { setTunnelChecking(true); try { - const [settingsRes, statusRes] = await Promise.all([ - fetch("/api/settings"), + const [settingsData, statusRes] = await Promise.all([ + useSettingsStore.getState().fetchSettings(), fetch("/api/tunnel/status", { cache: "no-store" }) ]); - if (settingsRes.ok) { - const data = await settingsRes.json(); - setRequireApiKey(data.requireApiKey || false); - setRequireLogin(data.requireLogin !== false); - setHasPassword(data.hasPassword || false); - setTunnelDashboardAccess(data.tunnelDashboardAccess || false); + if (settingsData) { + setRequireApiKey(settingsData.requireApiKey || false); + setRequireLogin(settingsData.requireLogin !== false); + setHasPassword(settingsData.hasPassword || false); + setTunnelDashboardAccess(settingsData.tunnelDashboardAccess || false); } if (statusRes.ok) { const data = await statusRes.json(); @@ -246,12 +246,8 @@ export default function APIPageClient({ machineId }) { const handleTunnelDashboardAccess = async (value) => { try { - const res = await fetch("/api/settings", { - method: "PATCH", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ tunnelDashboardAccess: value }), - }); - if (res.ok) setTunnelDashboardAccess(value); + const updated = await useSettingsStore.getState().patchSettings({ tunnelDashboardAccess: value }); + if (updated) setTunnelDashboardAccess(value); } catch (error) { console.log("Error updating tunnelDashboardAccess:", error); } @@ -259,12 +255,8 @@ export default function APIPageClient({ machineId }) { const handleRequireApiKey = async (value) => { try { - const res = await fetch("/api/settings", { - method: "PATCH", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ requireApiKey: value }), - }); - if (res.ok) setRequireApiKey(value); + const updated = await useSettingsStore.getState().patchSettings({ requireApiKey: value }); + if (updated) setRequireApiKey(value); } catch (error) { console.log("Error updating requireApiKey:", error); } diff --git a/src/app/(dashboard)/dashboard/providers/[id]/AddCustomModelModal.js b/src/app/(dashboard)/dashboard/providers/[id]/AddCustomModelModal.js index ef33c42b..4e49a5a4 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/AddCustomModelModal.js +++ b/src/app/(dashboard)/dashboard/providers/[id]/AddCustomModelModal.js @@ -2,8 +2,8 @@ import { useState, useEffect } from "react"; import PropTypes from "prop-types"; -import { Button, Modal, Toggle } from "@/shared/components"; -import { CAPACITY_META } from "@/shared/constants/models"; +import { Button, Modal, Select, Toggle } from "@/shared/components"; +import { CAPACITY_META, STT_TRANSPORT_META, STT_TRANSPORTS } from "@/shared/constants/models"; const emptyCaps = () => Object.fromEntries(Object.keys(CAPACITY_META).map((key) => [key, false])); @@ -33,6 +33,8 @@ export default function AddCustomModelModal({ const [testStatus, setTestStatus] = useState(null); // null | "testing" | "ok" | "error" const [testError, setTestError] = useState(""); const [saving, setSaving] = useState(false); + // Realtime dispatch marker for the transport select; "" = provider default REST. + const [transport, setTransport] = useState(""); // Re-seeded on open and when the target model changes while already open // (row-click editing reuses one mounted modal). `capsKey` is a primitive so a @@ -46,6 +48,7 @@ export default function AddCustomModelModal({ setModelId(initialModelId); setSeed(next); setCaps(next); + setTransport(""); setTestStatus(null); setTestError(""); }, [isOpen, initialModelId, capsKey]); @@ -85,7 +88,14 @@ export default function AddCustomModelModal({ for (const key of Object.keys(CAPACITY_META)) { if (!!caps[key] !== !!seed[key]) changed[key] = !!caps[key]; } - await onSave(cleanId, Object.keys(changed).length ? changed : undefined); + if (caps.stt) changed.stt = true; + // caps.stt is UI-only; the parent save flow derives the model type from + // it and forwards the pinned transport (null unless the caller picked one). + await onSave( + cleanId, + Object.keys(changed).length ? changed : undefined, + caps.stt ? transport : null + ); } finally { setSaving(false); } @@ -146,6 +156,32 @@ export default function AddCustomModelModal({ + {/* STT is a model TYPE, not a chat capability: the save flow turns this + flag into type "stt" (the API honours a transport only on stt + records). The select pins the realtime dispatch marker persisted + with the model; the whitelist is the shared STT_TRANSPORT_META. */} +
+ { setCaps((prev) => ({ ...prev, stt: v })); if (!v) setTransport(""); }} + label="Speech to text" + description="Transcribes audio via /v1/audio/transcriptions" + size="sm" + /> + {caps.stt && ( +
+ + +
+
+ +
+

Step 2: Paste the callback URL here

+

+ After authorization, copy the full URL from your browser (or the local callback page). +

+ setCallbackUrl(e.target.value)} + placeholder="http://127.0.0.1:.../?user_id=...&access_token=..." + className="font-mono text-xs" + /> +
+ + + {error && ( +
+

{error}

+
+ )} + +
+ + +
+ + )} + + + ); +} + +ZedAuthModal.propTypes = { + isOpen: PropTypes.bool.isRequired, + providerInfo: PropTypes.object, + onSuccess: PropTypes.func, + onClose: PropTypes.func.isRequired, +}; diff --git a/src/shared/components/index.js b/src/shared/components/index.js index 8e180c06..aa71d8e4 100644 --- a/src/shared/components/index.js +++ b/src/shared/components/index.js @@ -18,7 +18,6 @@ export { default as ModelSelectModal, ModelSelectSidePanel } from "./ModelSelect export { default as ManualConfigModal } from "./ManualConfigModal"; export { default as ComboFormModal } from "./ComboFormModal"; export { default as McpMarketplaceModal } from "./McpMarketplaceModal"; -export { default as UsageStats } from "./UsageStats"; export { default as LanguageSwitcher } from "./LanguageSwitcher"; export { default as NineRemoteButton } from "./NineRemoteButton"; export { default as HeaderMenu } from "./HeaderMenu"; @@ -28,6 +27,7 @@ export { default as KiroAuthModal } from "./KiroAuthModal"; export { default as KiroOAuthWrapper } from "./KiroOAuthWrapper"; export { default as KiroSocialOAuthModal } from "./KiroSocialOAuthModal"; export { default as CursorAuthModal } from "./CursorAuthModal"; +export { default as ZedAuthModal } from "./ZedAuthModal"; export { default as XiaomiMimoAuthModal } from "./XiaomiMimoAuthModal"; export { default as IFlowCookieModal } from "./IFlowCookieModal"; export { default as GitLabAuthModal } from "./GitLabAuthModal"; diff --git a/src/shared/components/layouts/DashboardLayout.js b/src/shared/components/layouts/DashboardLayout.js index aa555bd5..0f175f95 100644 --- a/src/shared/components/layouts/DashboardLayout.js +++ b/src/shared/components/layouts/DashboardLayout.js @@ -1,6 +1,6 @@ "use client"; -import { useState } from "react"; +import { useState, useEffect } from "react"; import { usePathname } from "next/navigation"; import { useNotificationStore } from "@/store/notificationStore"; import Sidebar from "../Sidebar"; @@ -37,6 +37,24 @@ export default function DashboardLayout({ children }) { const notifications = useNotificationStore((state) => state.notifications); const removeNotification = useNotificationStore((state) => state.removeNotification); + // Preload heavy usage charts in background when browser is idle + useEffect(() => { + const preload = () => { + import("@/shared/components/UsageStats").catch(() => {}); + import("@/app/(dashboard)/dashboard/usage/components/UsageChart").catch(() => {}); + import("@/app/(dashboard)/dashboard/usage/components/ProviderBarChart").catch(() => {}); + import("@/app/(dashboard)/dashboard/usage/components/TopModelsChart").catch(() => {}); + }; + if (typeof window !== "undefined") { + if ("requestIdleCallback" in window) { + const id = window.requestIdleCallback(preload, { timeout: 4000 }); + return () => window.cancelIdleCallback(id); + } + const timer = setTimeout(preload, 2500); + return () => clearTimeout(timer); + } + }, []); + return (
diff --git a/src/shared/constants/cliTools.js b/src/shared/constants/cliTools.js index 77c307fb..3f521231 100644 --- a/src/shared/constants/cliTools.js +++ b/src/shared/constants/cliTools.js @@ -165,6 +165,22 @@ export const CLI_TOOLS = { color: "#8B5CF6", description: "Nous Research self-improving AI agent", configType: "custom", + // Model slots Hermes supports besides the default ("model:" block). + // "default" is not listed — the card renders it as the main model picker. + roles: [ + { id: "delegation", label: "Delegation (subagents)" }, + { id: "vision", label: "Vision" }, + { id: "web_extract", label: "Web Extract" }, + { id: "compression", label: "Compression" }, + { id: "title_generation", label: "Title Generation" }, + { id: "approval", label: "Approval" }, + { id: "skills_hub", label: "Skills Hub" }, + { id: "mcp", label: "MCP" }, + { id: "memory_query_rewrite", label: "Memory Query Rewrite" }, + { id: "background_review", label: "Background Review" }, + { id: "curator", label: "Curator" }, + { id: "monitor", label: "Monitor" }, + ], }, droid: { id: "droid", diff --git a/src/shared/constants/models.js b/src/shared/constants/models.js index 297daae7..96667048 100644 --- a/src/shared/constants/models.js +++ b/src/shared/constants/models.js @@ -45,3 +45,23 @@ export const CAPACITY_META = { // search: temporarily hidden (feature not wired yet) reasoning: { icon: "neurology", label: "Reasoning", desc: "Supports reasoning / thinking", color: "text-amber-500" }, }; + +// Realtime STT transport markers accepted on custom models — single source of +// truth across layers: the API whitelist (src/app/api/models/custom/route.js +// sanitizeTransport) and the dashboard transport select +// (providers/[id]/AddCustomModelModal) both import this map, so one new row +// here makes a realtime engine dispatch case (open-sse/handlers/sttCore.js) +// selectable and validated end-to-end. Keys must mirror a sttCore case. +export const STT_TRANSPORT_META = { + "gemini-live": { + label: "Gemini Live (realtime WebSocket)", + desc: "Streams audio over bidiGenerateContent and returns incremental transcription segments", + }, +}; + +export const STT_TRANSPORTS = Object.freeze(Object.keys(STT_TRANSPORT_META)); + +export function isSttTransport(transport) { + if (typeof transport !== "string") return false; + return Object.prototype.hasOwnProperty.call(STT_TRANSPORT_META, transport.trim()); +} diff --git a/src/sse/handlers/chat.js b/src/sse/handlers/chat.js index 4d7e9a0c..26f2440a 100644 --- a/src/sse/handlers/chat.js +++ b/src/sse/handlers/chat.js @@ -18,6 +18,7 @@ import { DEFAULT_HEADROOM_URL } from "@/lib/headroom/detect"; import { getTransform as getPxpipeTransform } from "@/lib/pxpipe/loader.js"; import { appendPxpipeEvent } from "@/lib/pxpipe/events.js"; import { errorResponse, unavailableResponse } from "open-sse/utils/error.js"; +import { upstreamResponseHeaders } from "open-sse/utils/upstreamHeaders.js"; import { handleComboChat, handleFusionChat, detectRequiredCapabilities } from "open-sse/services/combo.js"; import { augmentModelsWithCapacityAdapter, withCapacityAdapterStripping, getActiveAdapterStrategy } from "open-sse/services/capacityAdapter.js"; import { handleBypassRequest } from "open-sse/utils/bypassHandler.js"; @@ -264,6 +265,7 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re const excludeConnectionIds = new Set(); let lastError = null; let lastStatus = null; + let lastHeaders = null; while (true) { const credentials = await getProviderCredentials(provider, excludeConnectionIds, model, { preferredConnectionId }); @@ -274,14 +276,14 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re const errorMsg = lastError || credentials.lastError || "Unavailable"; const status = HTTP_STATUS.SERVICE_UNAVAILABLE; log.warn("CHAT", `[${provider}/${model}] ${errorMsg} (${credentials.retryAfterHuman})`); - return unavailableResponse(status, `[${provider}/${model}] ${errorMsg}`, credentials.retryAfter, credentials.retryAfterHuman); + return unavailableResponse(status, `[${provider}/${model}] ${errorMsg}`, credentials.retryAfter, credentials.retryAfterHuman, lastHeaders); } if (excludeConnectionIds.size === 0) { log.warn("AUTH", `No active credentials for provider: ${provider}`); return errorResponse(HTTP_STATUS.NOT_FOUND, `No active credentials for provider: ${provider}`); } log.warn("CHAT", "No more accounts available", { provider }); - return errorResponse(lastStatus || HTTP_STATUS.SERVICE_UNAVAILABLE, lastError || "All accounts unavailable"); + return errorResponse(lastStatus || HTTP_STATUS.SERVICE_UNAVAILABLE, lastError || "All accounts unavailable", lastHeaders); } // Account selection shown in the unified "▶" line (acc:...) @@ -387,6 +389,7 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re excludeConnectionIds.add(credentials.connectionId); lastError = result.error; lastStatus = result.status; + lastHeaders = upstreamResponseHeaders(result.response?.headers); continue; } diff --git a/src/sse/handlers/stt.js b/src/sse/handlers/stt.js index 1840965e..3489678e 100644 --- a/src/sse/handlers/stt.js +++ b/src/sse/handlers/stt.js @@ -2,7 +2,7 @@ import { extractApiKey, isValidApiKey, getProviderCredentials, markAccountUnavailable, } from "../services/auth.js"; -import { getSettings } from "@/lib/localDb"; +import { getSettings, getCustomModels } from "@/lib/localDb"; import { getModelInfo } from "../services/model.js"; import { handleSttCore } from "open-sse/handlers/sttCore.js"; import { errorResponse, unavailableResponse } from "open-sse/utils/error.js"; @@ -17,6 +17,23 @@ const CREDENTIALED_PROVIDERS = new Set( .map(([id]) => id) ); +// Custom-model transport marker: models registered through +// /api/models/custom may pin a specialized STT transport (e.g. +// "gemini-live"). The engine dispatches on the marker itself, so the app +// layer only resolves it — same getModelInfo-style provider+model pairing, +// restricted to type "stt" records. +async function resolveCustomModelTransport(provider, model) { + try { + const customModels = await getCustomModels(); + const hit = customModels.find((c) => c && c.type === "stt" + && c.providerAlias === provider && c.id === model + && typeof c.transport === "string" && c.transport.trim()); + return hit ? hit.transport.trim() : null; + } catch { + return null; // DB unreadable → built-in registry marker still applies + } +} + export async function handleStt(request) { let formData; try { @@ -45,9 +62,11 @@ export async function handleStt(request) { const { provider, model } = modelInfo; log.info("ROUTING", `Provider: ${provider}, Model: ${model}`); + const modelTransport = await resolveCustomModelTransport(provider, model); + // noAuth providers if (!CREDENTIALED_PROVIDERS.has(provider)) { - const result = await handleSttCore({ provider, model, formData, sttConfig: AI_PROVIDERS[provider]?.sttConfig }); + const result = await handleSttCore({ provider, model, formData, sttConfig: AI_PROVIDERS[provider]?.sttConfig, transport: modelTransport }); if (result.success) return result.response; return errorResponse(result.status || HTTP_STATUS.BAD_GATEWAY, result.error || "STT failed"); } @@ -72,7 +91,7 @@ export async function handleStt(request) { log.info("AUTH", `\x1b[32mUsing ${provider} account: ${credentials.connectionName}\x1b[0m`); - const result = await handleSttCore({ provider, model, formData, credentials, sttConfig: AI_PROVIDERS[provider]?.sttConfig }); + const result = await handleSttCore({ provider, model, formData, credentials, sttConfig: AI_PROVIDERS[provider]?.sttConfig, transport: modelTransport }); if (result.success) return result.response; diff --git a/src/store/settingsStore.js b/src/store/settingsStore.js index e5f0a713..563d2e39 100644 --- a/src/store/settingsStore.js +++ b/src/store/settingsStore.js @@ -3,6 +3,8 @@ import { create } from "zustand"; import { CLIENT_STORE_TTL_MS } from "@/shared/constants/config"; +let inFlightSettingsPromise = null; + const useSettingsStore = create((set, get) => ({ settings: null, loading: false, @@ -11,23 +13,30 @@ const useSettingsStore = create((set, get) => ({ invalidate: () => set({ lastFetched: 0 }), - // Skips network when cache is fresh; pass {force:true} to override + // Skips network when cache is fresh; coalesce concurrent in-flight requests fetchSettings: async ({ force = false } = {}) => { const { lastFetched, settings } = get(); if (!force && settings && Date.now() - lastFetched < CLIENT_STORE_TTL_MS) return settings; + if (inFlightSettingsPromise) return inFlightSettingsPromise; + set({ loading: true, error: null }); - try { - const res = await fetch("/api/settings"); - const data = await res.json(); - if (res.ok) { - set({ settings: data, loading: false, lastFetched: Date.now() }); - return data; + inFlightSettingsPromise = (async () => { + try { + const res = await fetch("/api/settings"); + const data = await res.json(); + if (res.ok) { + set({ settings: data, loading: false, lastFetched: Date.now() }); + return data; + } + set({ error: data.error, loading: false }); + } catch (e) { + set({ error: "Failed to fetch settings", loading: false }); + } finally { + inFlightSettingsPromise = null; } - set({ error: data.error, loading: false }); - } catch (e) { - set({ error: "Failed to fetch settings", loading: false }); - } - return null; + return null; + })(); + return inFlightSettingsPromise; }, // PATCH server + merge into local cache (no extra fetch needed) @@ -40,7 +49,8 @@ const useSettingsStore = create((set, get) => ({ }); if (!res.ok) return null; const updated = await res.json(); - set({ settings: updated, lastFetched: Date.now() }); + // Merge, not replace: PATCH response omits GET-only fields (e.g. hasPassword) + set({ settings: { ...get().settings, ...updated }, lastFetched: Date.now() }); return updated; } catch { return null; diff --git a/tests/__baseline__/alias-baseline.json b/tests/__baseline__/alias-baseline.json index a387cb7f..80d3af8a 100644 --- a/tests/__baseline__/alias-baseline.json +++ b/tests/__baseline__/alias-baseline.json @@ -119,6 +119,7 @@ "morphllm": "morph" }, "idToAlias": { + "agnes": "agnes", "alicode": "alicode", "alicode-intl": "alicode-intl", "alims-intl": "alims-intl", @@ -127,7 +128,9 @@ "antigravity": "ag", "api-airforce": "af", "assemblyai": "assemblyai", + "atria": "atria", "azure": "azure", + "bai": "bai", "baidu": "qianfan", "bazaarlink": "bzl", "blackbox": "blackbox", @@ -145,6 +148,7 @@ "cohere": "cohere", "commandcode": "commandcode", "cursor": "cu", + "dahl": "dahl", "deepgram": "deepgram", "deepseek": "deepseek", "featherless": "featherless", @@ -192,6 +196,7 @@ "siliconflow": "siliconflow", "tencent": "hunyuan", "together": "together", + "tokenharbor": "tokenharbor", "tokenrouter": "tokenrouter", "venice": "venice", "vercel-ai-gateway": "vercel-ai-gateway", @@ -212,6 +217,7 @@ "alitp-intl", "anthropic", "assemblyai", + "atria", "black-forest-labs", "blackbox", "bm", @@ -229,6 +235,7 @@ "commandcode", "cu", "cx", + "dahl", "deepgram", "deepseek", "edge-tts", @@ -293,6 +300,7 @@ "siliconflow", "stability-ai", "together", + "tokenharbor", "tokenrouter", "venice", "vertex", diff --git a/tests/__baseline__/providers-baseline.json b/tests/__baseline__/providers-baseline.json index e28594df..8c12ef73 100644 --- a/tests/__baseline__/providers-baseline.json +++ b/tests/__baseline__/providers-baseline.json @@ -93,7 +93,7 @@ "Anthropic-Version": "2023-06-01", "Anthropic-Beta": "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20,effort-2025-11-24,structured-outputs-2025-12-15,fast-mode-2026-02-01,redact-thinking-2026-02-12,token-efficient-tools-2026-03-28", "Anthropic-Dangerous-Direct-Browser-Access": "true", - "User-Agent": "claude-cli/2.1.258 (external, sdk-cli)", + "User-Agent": "claude-cli/2.1.280 (external, sdk-cli)", "X-App": "cli", "X-Stainless-Helper-Method": "stream", "X-Stainless-Retry-Count": "0", @@ -1085,5 +1085,33 @@ "preserveCacheControl": true }, "format": "openai" + }, + "dahl": { + "baseUrl": "https://inference.dahl.global/v1/chat/completions", + "validateUrl": "https://inference.dahl.global/v1/models", + "format": "openai" + }, + "atria": { + "baseUrl": "https://api.atria-asi.ai/v1/chat/completions", + "validateUrl": "https://api.atria-asi.ai/v1/models", + "format": "openai" + }, + "agnes": { + "baseUrl": "https://apihub.agnes-ai.com/v1/chat/completions", + "validateUrl": "https://apihub.agnes-ai.com/v1/models", + "format": "openai" + }, + "bai": { + "baseUrl": "https://api.b.ai/v1/chat/completions", + "validateUrl": "https://api.b.ai/v1/models", + "format": "openai" + }, + "tokenharbor": { + "baseUrl": "https://tokenharbor.ai/v1/chat/completions", + "validateUrl": "https://tokenharbor.ai/v1/models", + "retry": { + "429": 2 + }, + "format": "openai" } } \ No newline at end of file diff --git a/tests/translator/__snapshots__/golden-request.test.js.snap b/tests/translator/__snapshots__/golden-request.test.js.snap index f5a5253b..db2c7378 100644 --- a/tests/translator/__snapshots__/golden-request.test.js.snap +++ b/tests/translator/__snapshots__/golden-request.test.js.snap @@ -119,6 +119,7 @@ exports[`GOLDEN request: OpenAI → Claude > reasoning_effort → adaptive outpu }, ], "thinking": { + "display": "summarized", "type": "adaptive", }, } diff --git a/tests/translator/__snapshots__/golden-response-stream.test.js.snap b/tests/translator/__snapshots__/golden-response-stream.test.js.snap index 190047fe..c7050947 100644 --- a/tests/translator/__snapshots__/golden-response-stream.test.js.snap +++ b/tests/translator/__snapshots__/golden-response-stream.test.js.snap @@ -17,21 +17,6 @@ exports[`GOLDEN response stream: Claude → OpenAI > text + thinking + tool_use "model": "claude-opus-4-6", "object": "chat.completion.chunk", }, - { - "choices": [ - { - "delta": { - "content": "", - }, - "finish_reason": null, - "index": 0, - }, - ], - "created": 0, - "id": "chatcmpl-msg_1", - "model": "claude-opus-4-6", - "object": "chat.completion.chunk", - }, { "choices": [ { @@ -47,21 +32,6 @@ exports[`GOLDEN response stream: Claude → OpenAI > text + thinking + tool_use "model": "claude-opus-4-6", "object": "chat.completion.chunk", }, - { - "choices": [ - { - "delta": { - "content": "", - }, - "finish_reason": null, - "index": 0, - }, - ], - "created": 0, - "id": "chatcmpl-msg_1", - "model": "claude-opus-4-6", - "object": "chat.completion.chunk", - }, { "choices": [ { diff --git a/tests/translator/claude-claude-stream-decloak.test.js b/tests/translator/claude-claude-stream-decloak.test.js index 35c0fcdf..d223129c 100644 --- a/tests/translator/claude-claude-stream-decloak.test.js +++ b/tests/translator/claude-claude-stream-decloak.test.js @@ -36,10 +36,10 @@ describe("Claude → Claude streaming passthrough (OAuth tool cloak)", () => { expect(outText).toBe(textChunk); }); - it("is a no-op when no cloak map is present", () => { + it("falls back to suffix-stripping when no cloak map is present", () => { const chunk = toolUseStart(CLOAKED); const [out] = translateResponse(FORMATS.CLAUDE, FORMATS.CLAUDE, chunk, {}); - expect(out).toBe(chunk); + expect(out.content_block.name).toBe("run_code"); }); it("tolerates the null flush chunk", () => { diff --git a/tests/translator/openai-thinking-display.test.js b/tests/translator/openai-thinking-display.test.js new file mode 100644 index 00000000..7cbef379 --- /dev/null +++ b/tests/translator/openai-thinking-display.test.js @@ -0,0 +1,83 @@ +// OpenAI-format clients asking for reasoning get Claude's thinking text back. +// +// Claude only returns thinking text when the request sets thinking.display to +// "summarized" — otherwise the redact-thinking beta is sent and every thinking +// block comes back signature-only. That field has no OpenAI equivalent, so an +// OpenAI-format client (opencode, DeepSeek Harness, Cherry Studio, ...) could never +// see its reasoning: reasoning_content stayed empty however high the effort was. +// +// The client's intent is read from the pre-translation body: +// - Chat Completions: setting reasoning_effort is the request for reasoning. +// - Responses API: reasoning.summary is OpenAI's explicit ask for summaries. +// Claude-format clients are untouched — they set display themselves. +import { describe, it, expect } from "vitest"; +import "./registerAll.js"; +import { translateRequest } from "../../open-sse/translator/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { selectAnthropicBeta } from "../../open-sse/providers/shared.js"; + +const REDACT = "redact-thinking-2026-02-12"; +const MODEL = "claude-opus-5"; + +function toClaude(sourceFormat, body) { + return translateRequest(sourceFormat, FORMATS.CLAUDE, MODEL, body, true, null, "claude"); +} + +const chat = (extra = {}) => ({ model: MODEL, messages: [{ role: "user", content: "Is 391 prime?" }], ...extra }); +const responses = (extra = {}) => ({ model: MODEL, input: [{ role: "user", content: "Is 391 prime?" }], ...extra }); + +describe("Chat Completions clients", () => { + it("reasoning_effort asks Claude for summarized thinking and drops redact-thinking", () => { + const out = toClaude(FORMATS.OPENAI, chat({ reasoning_effort: "high" })); + expect(out.thinking.display).toBe("summarized"); + expect(selectAnthropicBeta(MODEL, out)).not.toContain(REDACT); + }); + + it("reasoning_effort none leaves thinking off and keeps redact-thinking", () => { + const out = toClaude(FORMATS.OPENAI, chat({ reasoning_effort: "none" })); + expect(out.thinking?.display).toBeUndefined(); + expect(selectAnthropicBeta(MODEL, out)).toContain(REDACT); + }); + + it("no reasoning_effort changes nothing", () => { + const out = toClaude(FORMATS.OPENAI, chat()); + expect(out.thinking?.display).toBeUndefined(); + expect(selectAnthropicBeta(MODEL, out)).toContain(REDACT); + }); +}); + +describe("Responses API clients", () => { + it("reasoning.summary asks Claude for summarized thinking", () => { + const out = toClaude(FORMATS.OPENAI_RESPONSES, responses({ reasoning: { effort: "high", summary: "auto" } })); + expect(out.thinking.display).toBe("summarized"); + expect(selectAnthropicBeta(MODEL, out)).not.toContain(REDACT); + }); + + it("effort without summary keeps thinking text redacted, as OpenAI would", () => { + const out = toClaude(FORMATS.OPENAI_RESPONSES, responses({ reasoning: { effort: "high" } })); + expect(out.thinking?.display).toBeUndefined(); + expect(selectAnthropicBeta(MODEL, out)).toContain(REDACT); + }); +}); + +describe("Claude-format clients are untouched", () => { + it("thinking without display stays redacted", () => { + const out = toClaude(FORMATS.CLAUDE, { + model: MODEL, max_tokens: 4096, + thinking: { type: "enabled", budget_tokens: 2048 }, + messages: [{ role: "user", content: "Is 391 prime?" }], + }); + expect(out.thinking?.display).toBeUndefined(); + expect(selectAnthropicBeta(MODEL, out)).toContain(REDACT); + }); + + it("an explicit display is kept as sent", () => { + const out = toClaude(FORMATS.CLAUDE, { + model: MODEL, max_tokens: 4096, + thinking: { type: "enabled", budget_tokens: 2048, display: "omitted" }, + messages: [{ role: "user", content: "Is 391 prime?" }], + }); + expect(out.thinking.display).toBe("omitted"); + expect(selectAnthropicBeta(MODEL, out)).toContain(REDACT); + }); +}); diff --git a/tests/unit/aggregator-providers-batch.test.js b/tests/unit/aggregator-providers-batch.test.js new file mode 100644 index 00000000..c270e93b --- /dev/null +++ b/tests/unit/aggregator-providers-batch.test.js @@ -0,0 +1,137 @@ +import { describe, expect, it } from "vitest"; +import { existsSync, readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; + +import REGISTRY from "../../open-sse/providers/registry/index.js"; +import { PROVIDERS } from "../../open-sse/providers/index.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; +import { DefaultExecutor } from "../../open-sse/executors/default.js"; + +/** + * OpenAI-compatible aggregator providers. Each was verified by probing the + * live /v1/models endpoint: a 401 with a structured error body confirms a + * real API behind the host, and Dahl/Kira answer 200 with no credentials. + */ +const BATCH = [ + { + id: "dahl", + category: "apikey", + baseUrl: "https://inference.dahl.global/v1/chat/completions", + modelsUrl: "https://inference.dahl.global/v1/models", + aliases: ["dahl-inference"], + }, + { + id: "atria", + category: "apikey", + baseUrl: "https://api.atria-asi.ai/v1/chat/completions", + modelsUrl: "https://api.atria-asi.ai/v1/models", + aliases: ["atria-asi"], + }, + { + id: "agnes", + category: "freeTier", + baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions", + modelsUrl: "https://apihub.agnes-ai.com/v1/models", + aliases: ["agnes-ai"], + }, + { + id: "bai", + category: "apikey", + baseUrl: "https://api.b.ai/v1/chat/completions", + modelsUrl: "https://api.b.ai/v1/models", + aliases: ["b-ai"], + }, +]; + +describe.each(BATCH)("$id provider", (p) => { + const entry = REGISTRY.find((e) => e.id === p.id); + + it("is registered with the expected category and base URL", () => { + expect(entry).toBeDefined(); + expect(entry.category).toBe(p.category); + expect(PROVIDERS[p.id].baseUrl).toBe(p.baseUrl); + expect(PROVIDERS[p.id].format).toBe("openai"); + }); + + it("exposes its aliases and a UI display name", () => { + for (const a of p.aliases) expect(entry.aliases).toContain(a); + expect(entry.display?.name).toBeTruthy(); + expect(entry.display?.textIcon).toBeTruthy(); + }); + + it("routes through the shared DefaultExecutor", () => { + expect(getExecutor(p.id)).toBeInstanceOf(DefaultExecutor); + }); + + it("accepts arbitrary model ids via passthrough", () => { + expect(entry.passthroughModels).toBe(true); + }); +}); + +describe("Atria Dawn specifics", () => { + const entry = REGISTRY.find((e) => e.id === "atria"); + + it("is named after the service, not just the host", () => { + expect(entry.display.name).toBe("Atria Dawn"); + }); + + it("pins the single documented preview model", () => { + expect(entry.models.map((m) => m.id)).toEqual(["Atria-Dawn-Preview"]); + }); +}); + +describe("provider icons", () => { + // This file lives in tests/unit/, so resolve icons against the repo root. + const REPO_ROOT = join(dirname(fileURLToPath(import.meta.url)), "..", ".."); + + it("ships a /public/providers/{id}.png for every provider in the batch", () => { + // getProviderIconSrc() resolves /providers/{id}.png and falls back to the + // textIcon tile when the file 404s, so a missing icon is silent in the UI. + for (const p of BATCH.map((x) => x.id)) { + expect(existsSync(join(REPO_ROOT, "public", "providers", `${p}.png`))).toBe(true); + } + }); +}); + +describe("authenticated model discovery", () => { + const ROUTE = join( + dirname(fileURLToPath(import.meta.url)), "..", "..", + "src", "app", "api", "providers", "[id]", "models", "route.js" + ); + const source = readFileSync(ROUTE, "utf8"); + + it("registers every batch provider in the /models resolver", () => { + // Without an entry here the route answers + // 400 "Provider X does not support models listing" and live discovery + // silently fails once a key is saved. + for (const p of BATCH.map((x) => x.id)) { + expect(source).toContain(`${p}: createOpenAIModelsConfig(`); + } + }); + + it("points each entry at that provider's own /models URL", () => { + for (const p of BATCH) { + const line = source.split("\n").find((l) => l.trim().startsWith(`${p.id}: createOpenAIModelsConfig(`)); + expect(line, `no models entry for ${p.id}`).toBeTruthy(); + expect(line).toContain(p.modelsUrl); + } + }); +}); + +describe("batch invariants", () => { + it("keeps every registry id unique", () => { + const ids = REGISTRY.map((e) => e.id); + expect(new Set(ids).size).toBe(ids.length); + }); + + it("does not claim vision for any provider in the batch", () => { + // These aggregators relay third-party models; none of them has documented + // image input, so no registry entry may assert it. + for (const p of BATCH) { + const entry = REGISTRY.find((e) => e.id === p.id); + expect(entry.serviceKinds ?? ["llm"]).toContain("llm"); + expect(entry.imageToTextConfig).toBeUndefined(); + } + }); +}); diff --git a/tests/unit/anthropic-gateway-passthrough.test.js b/tests/unit/anthropic-gateway-passthrough.test.js new file mode 100644 index 00000000..308f6636 --- /dev/null +++ b/tests/unit/anthropic-gateway-passthrough.test.js @@ -0,0 +1,80 @@ +import { describe, it, expect, beforeEach, vi } from "vitest"; +import { mergeAnthropicBeta } from "open-sse/providers/shared.js"; +import { upstreamResponseHeaders } from "open-sse/utils/upstreamHeaders.js"; +import { createErrorResult, unavailableResponse } from "open-sse/utils/error.js"; + +const betaFlags = (headers) => (headers["Anthropic-Beta"] || "").split(",").map((s) => s.trim()).filter(Boolean); + +describe("mergeAnthropicBeta", () => { + it("unions and dedupes comma lists, ignoring blanks", () => { + expect(mergeAnthropicBeta("a,b", " b , c ,", undefined, "")).toBe("a,b,c"); + }); +}); + +describe("DefaultExecutor.buildHeaders() forwards client anthropic-beta", () => { + let DefaultExecutor; + + beforeEach(async () => { + vi.resetModules(); + ({ DefaultExecutor } = await import("open-sse/executors/default.js")); + }); + + it("keeps unknown client flags alongside the pinned set on claude", () => { + const executor = new DefaultExecutor("claude"); + const rawHeaders = { "anthropic-beta": "safeguards-2026-09-01,context-1m-2025-08-07" }; + const flags = betaFlags(executor.buildHeaders({ apiKey: "k", rawHeaders }, true, undefined, "claude-opus-5")); + expect(flags).toContain("safeguards-2026-09-01"); + expect(flags).toContain("context-1m-2025-08-07"); + expect(flags).toContain("context-management-2025-06-27"); + expect(new Set(flags).size).toBe(flags.length); + }); + + it("forwards client flags on anthropic-compatible Claude models", () => { + const executor = new DefaultExecutor("anthropic-compatible-custom"); + const creds = { apiKey: "k", rawHeaders: { "anthropic-beta": "safeguards-2026-09-01" }, providerSpecificData: { baseUrl: "https://gw.example.com/v1" } }; + const flags = betaFlags(executor.buildHeaders(creds, true, undefined, "claude-sonnet-5")); + expect(flags).toContain("safeguards-2026-09-01"); + expect(flags).not.toContain("claude-code-20250219"); + }); + + it("forwards client flags on the anthropic provider", () => { + const executor = new DefaultExecutor("anthropic"); + const flags = betaFlags(executor.buildHeaders({ apiKey: "k", rawHeaders: { "anthropic-beta": "safeguards-2026-09-01" } }, true, undefined, "claude-sonnet-5")); + expect(flags).toContain("safeguards-2026-09-01"); + }); +}); + +describe("upstream response header forwarding", () => { + const upstream = new Headers({ + "retry-after": "12", + "x-should-retry": "false", + "anthropic-ratelimit-unified-status": "rejected", + "anthropic-ratelimit-unified-reset": "1790000000", + "set-cookie": "secret=1", + "content-length": "99", + }); + + it("picks only retry and ratelimit headers", () => { + expect(upstreamResponseHeaders(upstream)).toEqual({ + "retry-after": "12", + "x-should-retry": "false", + "anthropic-ratelimit-unified-status": "rejected", + "anthropic-ratelimit-unified-reset": "1790000000", + }); + expect(upstreamResponseHeaders(undefined)).toEqual({}); + }); + + it("attaches them to error results", () => { + const { response } = createErrorResult(429, "limited", undefined, upstreamResponseHeaders(upstream)); + expect(response.headers.get("x-should-retry")).toBe("false"); + expect(response.headers.get("anthropic-ratelimit-unified-status")).toBe("rejected"); + expect(response.headers.get("set-cookie")).toBeNull(); + }); + + it("keeps the gateway retry-after on all-accounts-limited responses", () => { + const retryAt = new Date(Date.now() + 30000).toISOString(); + const res = unavailableResponse(503, "busy", retryAt, "30s", upstreamResponseHeaders(upstream)); + expect(Number(res.headers.get("retry-after"))).toBeGreaterThan(20); + expect(res.headers.get("anthropic-ratelimit-unified-reset")).toBe("1790000000"); + }); +}); diff --git a/tests/unit/claude-cloaking.test.js b/tests/unit/claude-cloaking.test.js index 64994dc5..bc4cf541 100644 --- a/tests/unit/claude-cloaking.test.js +++ b/tests/unit/claude-cloaking.test.js @@ -12,7 +12,7 @@ import { CLAUDE_TOOL_SUFFIX } from "../../open-sse/config/appConstants.js"; it("advertises a Claude Code version accepted by Fable 5.1", () => { const body = applyCloaking({ messages: [] }, "sk-ant-oat-test", "session-id"); - expect(body.system[0].text).toMatch(/^x-anthropic-billing-header: cc_version=2.1.258\./); + expect(body.system[0].text).toMatch(/^x-anthropic-billing-header: cc_version=2.1.280\./); }); describe("cloakClaudeTools", () => { @@ -117,7 +117,8 @@ describe("decloakStreamChunk", () => { it("tolerates null chunks and missing maps (stream flush path)", () => { expect(decloakStreamChunk(null, toolNameMap)).toBeNull(); - expect(decloakStreamChunk(toolUseStart("run_code" + CLAUDE_TOOL_SUFFIX), null).content_block.name).toBe("run_code" + CLAUDE_TOOL_SUFFIX); - expect(decloakStreamChunk(toolUseStart("run_code" + CLAUDE_TOOL_SUFFIX), new Map()).content_block.name).toBe("run_code" + CLAUDE_TOOL_SUFFIX); + expect(decloakStreamChunk(toolUseStart("run_code" + CLAUDE_TOOL_SUFFIX), null).content_block.name).toBe("run_code"); + expect(decloakStreamChunk(toolUseStart("run_code" + CLAUDE_TOOL_SUFFIX), new Map()).content_block.name).toBe("run_code"); + expect(decloakStreamChunk(toolUseStart("uncloaked_tool"), null).content_block.name).toBe("uncloaked_tool"); }); }); diff --git a/tests/unit/claude-header-forwarding.test.js b/tests/unit/claude-header-forwarding.test.js index d813f356..653fa462 100644 --- a/tests/unit/claude-header-forwarding.test.js +++ b/tests/unit/claude-header-forwarding.test.js @@ -29,7 +29,7 @@ describe("DefaultExecutor.buildHeaders() — claude provider", () => { headers["Anthropic-Version"] === "2023-06-01" || headers["anthropic-version"] === "2023-06-01"; expect(hasVersion).toBe(true); - expect(headers["User-Agent"]).toBe("claude-cli/2.1.258 (external, sdk-cli)"); + expect(headers["User-Agent"]).toBe("claude-cli/2.1.280 (external, sdk-cli)"); }); it("includes heavy-agent beta flags for claude-opus-5", () => { @@ -95,6 +95,38 @@ describe("DefaultExecutor.buildHeaders() — claude provider", () => { const executor = new DefaultExecutor("claude"); expect(() => executor.buildHeaders({ apiKey: "sk" }, false)).not.toThrow(); }); + + it("sets x-claude-code-session-id from metadata.user_id on Claude OAuth", () => { + const executor = new DefaultExecutor("claude"); + const headers = executor.buildHeaders( + { accessToken: "sk-ant-oat-test-token" }, + true, + undefined, + "claude-opus-5", + { + metadata: { + user_id: '{"device_id":"d","account_uuid":"a","session_id":"sess-abc"}', + }, + } + ); + expect(headers["x-claude-code-session-id"]).toBe("sess-abc"); + }); + + it("omits x-claude-code-session-id for non-OAuth API keys", () => { + const executor = new DefaultExecutor("claude"); + const headers = executor.buildHeaders( + { apiKey: "sk-ant-api03-xxx" }, + true, + undefined, + "claude-opus-5", + { + metadata: { + user_id: '{"device_id":"d","account_uuid":"a","session_id":"sess-abc"}', + }, + } + ); + expect(headers["x-claude-code-session-id"]).toBeUndefined(); + }); }); // ─── anthropic-compatible header stripping ──────────────────────────────────── diff --git a/tests/unit/claude-reset-grants.test.js b/tests/unit/claude-reset-grants.test.js new file mode 100644 index 00000000..ec7cb6ad --- /dev/null +++ b/tests/unit/claude-reset-grants.test.js @@ -0,0 +1,23 @@ +import { describe, it, expect } from "vitest"; +import { parseClaudeResetGrants } from "../../open-sse/services/usage/claude.js"; + +describe("parseClaudeResetGrants", () => { + it("sums usable grants and picks next_grant_id", () => { + const r = parseClaudeResetGrants({ + eligible: true, + next_grant_id: "g2", + grants: [ + { id: "g1", resets_left: 1, ends_at: "2026-10-01T00:00:00Z" }, + { id: "g2", resets_left: 2, ends_at: "2026-10-22T00:00:00Z", clears: ["five_hour", "seven_day"] }, + { id: "g3", resets_left: 5, paused: true }, + ], + }); + expect(r).toMatchObject({ availableCount: 3, nextGrantId: "g2", expiresAt: "2026-10-22T00:00:00Z" }); + expect(r.grants.map((g) => g.id)).toEqual(["g1", "g2", "g3"]); // modal lists paused too + expect(r.grants[1].clears).toEqual(["five_hour", "seven_day"]); + }); + it("returns null when ineligible or missing", () => { + expect(parseClaudeResetGrants(undefined)).toBeNull(); + expect(parseClaudeResetGrants({ eligible: false, grants: [] })).toBeNull(); + }); +}); diff --git a/tests/unit/claude-thinking-stream-boundaries.test.js b/tests/unit/claude-thinking-stream-boundaries.test.js new file mode 100644 index 00000000..99ab7925 --- /dev/null +++ b/tests/unit/claude-thinking-stream-boundaries.test.js @@ -0,0 +1,175 @@ +// Thinking/answer boundaries across the OpenAI pivot. +// +// claude-to-openai used to mark a Claude thinking block with literal "" / +// "" chunks in delta.content while the thinking text itself went out in +// reasoning_content. The pair always arrived empty and adjacent, so OpenAI-format +// clients (opencode, DeepSeek Harness, ...) rendered a bare "" above +// every answer (#3399, #4199). +// +// The Responses translators leaned on that "" marker as their only signal +// to close the reasoning item before the answer. Dropping the marker therefore +// requires closing reasoning when the first message text or tool call arrives — +// which also fixes item ordering for every reasoning_content provider (DeepSeek, +// GLM, Qwen, Kimi), not just Claude. +import { describe, it, expect } from "vitest"; +import { claudeToOpenAIResponse } from "../../open-sse/translator/response/claude-to-openai.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { createSSETransformStreamWithLogger } from "../../open-sse/utils/stream.js"; +import { createResponsesApiTransformStream } from "../../open-sse/transformer/responsesTransformer.js"; + +const THINKING = "391 factors as 17 times 23, so it's not prime."; +const ANSWER = "No — 391 = 17 × 23."; + +function claudeThinkingStream({ thinkingText = THINKING, answer = ANSWER } = {}) { + const thinkingDeltas = thinkingText + ? [{ type: "content_block_delta", index: 0, delta: { type: "thinking_delta", thinking: thinkingText } }] + : []; + return [ + { type: "message_start", message: { id: "msg_1", model: "claude-opus-5", role: "assistant", content: [], usage: { input_tokens: 10, output_tokens: 0 } } }, + { type: "content_block_start", index: 0, content_block: { type: "thinking", thinking: "", signature: "" } }, + ...thinkingDeltas, + { type: "content_block_delta", index: 0, delta: { type: "signature_delta", signature: "sig_abc" } }, + { type: "content_block_stop", index: 0 }, + { type: "content_block_start", index: 1, content_block: { type: "text", text: "" } }, + { type: "content_block_delta", index: 1, delta: { type: "text_delta", text: answer } }, + { type: "content_block_stop", index: 1 }, + { type: "message_delta", delta: { stop_reason: "end_turn", stop_sequence: null }, usage: { output_tokens: 20 } }, + { type: "message_stop" }, + ]; +} + +function runClaudeToOpenAI(events) { + const state = {}; + const out = []; + for (const ev of events) { + const r = claudeToOpenAIResponse(ev, state); + if (Array.isArray(r)) out.push(...r); + else if (r) out.push(r); + } + const deltas = out.map((c) => c.choices?.[0]?.delta || {}); + return { + content: deltas.map((d) => d.content || "").join(""), + reasoning: deltas.map((d) => d.reasoning_content || "").join(""), + contentChunks: deltas.map((d) => d.content).filter((c) => c != null), + }; +} + +async function drain(stream) { + const reader = stream.getReader(); + const decoder = new TextDecoder(); + let text = ""; + for (;;) { + const { value, done } = await reader.read(); + if (done) break; + text += typeof value === "string" ? value : decoder.decode(value, { stream: true }); + } + return text + decoder.decode(); +} + +function sseStream(chunks) { + const encoder = new TextEncoder(); + const body = chunks.map((c) => `data: ${JSON.stringify(c)}\n\n`).join("") + "data: [DONE]\n\n"; + return new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(body)); + controller.close(); + }, + }); +} + +// Upstream speaks `upstream`, client speaks the Responses API. +async function viaResponsesTranslator(chunks, upstream, provider, model) { + const out = sseStream(chunks).pipeThrough( + createSSETransformStreamWithLogger(upstream, FORMATS.OPENAI_RESPONSES, provider, null, null, model), + ); + return parseEvents(await drain(out)); +} + +async function viaResponsesTransformer(chunks) { + return parseEvents(await drain(sseStream(chunks).pipeThrough(createResponsesApiTransformStream(null)))); +} + +function parseEvents(text) { + return text + .split("\n") + .filter((l) => l.startsWith("data: ") && l.slice(6).trim() !== "[DONE]") + .map((l) => { + try { return JSON.parse(l.slice(6)); } catch { return null; } + }) + .filter(Boolean); +} + +// Index of the event announcing/finishing an output item of the given type. +function itemEventIndex(events, eventType, itemType) { + return events.findIndex((e) => e.type === eventType && e.item?.type === itemType); +} + +function expectReasoningClosedBefore(events, nextItemType) { + const reasoningDone = itemEventIndex(events, "response.output_item.done", "reasoning"); + const nextAdded = itemEventIndex(events, "response.output_item.added", nextItemType); + expect(reasoningDone).toBeGreaterThanOrEqual(0); + expect(nextAdded).toBeGreaterThanOrEqual(0); + expect(reasoningDone).toBeLessThan(nextAdded); +} + +describe("claude-to-openai: thinking never leaks markers into content", () => { + it("summarized thinking goes to reasoning_content, answer to content, no text", () => { + const { content, reasoning, contentChunks } = runClaudeToOpenAI(claudeThinkingStream()); + expect(reasoning).toBe(THINKING); + expect(content).toBe(ANSWER); + expect(contentChunks.some((c) => c.includes("") || c.includes(""))).toBe(false); + }); + + it("signature-only (redacted) thinking yields no stray markers", () => { + const { content, reasoning } = runClaudeToOpenAI(claudeThinkingStream({ thinkingText: "" })); + expect(reasoning).toBe(""); + expect(content).toBe(ANSWER); + }); +}); + +describe("Responses translator: reasoning closes before the answer", () => { + it("Claude upstream: reasoning item is done before the message item opens", async () => { + const events = await viaResponsesTranslator(claudeThinkingStream(), FORMATS.CLAUDE, "claude", "claude-opus-5"); + expectReasoningClosedBefore(events, "message"); + const summary = events.find((e) => e.type === "response.reasoning_summary_text.done"); + expect(summary?.text).toBe(THINKING); + }); + + it("reasoning_content upstream: reasoning item is done before the message item opens", async () => { + const events = await viaResponsesTranslator([ + { id: "c1", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] }, + { id: "c1", choices: [{ index: 0, delta: { content: ANSWER } }] }, + { id: "c1", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }, + ], FORMATS.OPENAI, "deepseek", "deepseek-flash"); + expectReasoningClosedBefore(events, "message"); + }); + + it("reasoning_content upstream: reasoning item is done before a tool call opens", async () => { + const events = await viaResponsesTranslator([ + { id: "c2", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] }, + { id: "c2", choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_1", type: "function", function: { name: "lookup", arguments: "{}" } }] } }] }, + { id: "c2", choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }] }, + ], FORMATS.OPENAI, "deepseek", "deepseek-flash"); + expectReasoningClosedBefore(events, "function_call"); + }); +}); + +describe("responsesTransformer (/v1/responses handler): reasoning closes before the answer", () => { + it("reasoning item is done before the message item opens", async () => { + const events = await viaResponsesTransformer([ + { id: "c3", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] }, + { id: "c3", choices: [{ index: 0, delta: { content: ANSWER } }] }, + { id: "c3", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }, + ]); + expectReasoningClosedBefore(events, "message"); + }); + + it("reasoning item is done before a tool call opens", async () => { + const events = await viaResponsesTransformer([ + { id: "c4", choices: [{ index: 0, delta: { role: "assistant", reasoning_content: THINKING } }] }, + { id: "c4", choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_2", type: "function", function: { name: "lookup", arguments: "{}" } }] } }] }, + { id: "c4", choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }] }, + ]); + expectReasoningClosedBefore(events, "function_call"); + }); +}); diff --git a/tests/unit/cline-free-tier-models.test.js b/tests/unit/cline-free-tier-models.test.js new file mode 100644 index 00000000..7dac1858 --- /dev/null +++ b/tests/unit/cline-free-tier-models.test.js @@ -0,0 +1,125 @@ +// Cline's free tier lives in the `cline-free/` namespace and is published only +// by the recommended-models feed, not by /api/v1/models. These tests pin that +// resolveClineModels() merges the feed's `free[]` into its catalog so the free +// models reach /v1/models and the dashboard picker. + +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +const MODELS_URL = "https://api.cline.bot/api/v1/models"; +const FEED_URL = "https://api.cline.bot/api/v1/ai/cline/recommended-models"; + +const MODELS_RESPONSE = [ + { id: "meta/muse-spark-1.3-contributor" }, + { id: "deepseek/deepseek-v4.1-flash" }, + { id: "stealth/space-bunny-alpha" }, +]; + +const FEED_RESPONSE = { + recommended: [{ id: "anthropic/claude-opus-5", name: "Claude Opus 5", description: "", tags: ["NEW"] }], + free: [ + { id: "stealth/space-bunny-alpha", name: "Space Bunny Alpha", description: "", tags: [] }, + { id: "cline-free/muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", description: "", tags: [] }, + { id: "cline-free/deepseek-v4.1-flash", name: "Deepseek V4.1 Flash", description: "", tags: [] }, + { id: "cline-free/gemini-3.8-flash", name: "Gemini 3.8 Flash", description: "", tags: [] }, + { id: "cline-free/mimo-v2.6-flash", name: "Mimo V2.6 Flash", description: "", tags: [] }, + ], + clinePass: [{ id: "cline-pass/glm-5.3", name: "GLM-5.3", description: "", tags: [] }], +}; + +let fetchMock; + +function jsonResponse(obj) { + return { ok: true, status: 200, json: async () => obj, text: async () => JSON.stringify(obj) }; +} + +beforeEach(() => { + fetchMock = vi.fn(async (url) => { + if (String(url) === MODELS_URL) return jsonResponse(MODELS_RESPONSE); + if (String(url) === FEED_URL) return jsonResponse(FEED_RESPONSE); + throw new Error("unexpected fetch: " + url); + }); + vi.stubGlobal("fetch", fetchMock); +}); + +afterEach(() => vi.unstubAllGlobals()); + +describe("resolveClineModels free-tier merge", () => { + it("includes the cline-free/* models that /api/v1/models omits", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({ accessToken: "test-token" }); + const ids = result.models.map((m) => m.id); + expect(ids).toContain("cline-free/muse-spark-1.3-contributor"); + expect(ids).toContain("cline-free/deepseek-v4.1-flash"); + expect(ids).toContain("cline-free/gemini-3.8-flash"); + expect(ids).toContain("cline-free/mimo-v2.6-flash"); + }); + + it("keeps every /api/v1/models entry (feed is additive)", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({ accessToken: "test-token" }); + const ids = result.models.map((m) => m.id); + expect(ids).toContain("meta/muse-spark-1.3-contributor"); + expect(ids).toContain("deepseek/deepseek-v4.1-flash"); + }); + + it("deduplicates ids present in both sources", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({ accessToken: "test-token" }); + const ids = result.models.map((m) => m.id); + expect(ids.filter((id) => id === "stealth/space-bunny-alpha")).toHaveLength(1); + }); + + it("returns {id, name} for feed entries", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({ accessToken: "test-token" }); + const entry = result.models.find((m) => m.id === "cline-free/muse-spark-1.3-contributor"); + expect(entry.name).toBe("Muse Spark 1.3 Contributor"); + }); + + it("survives a failing feed and still returns the /models catalog", async () => { + fetchMock.mockImplementation(async (url) => { + if (String(url) === MODELS_URL) return jsonResponse(MODELS_RESPONSE); + return { ok: false, status: 503, json: async () => ({}), text: async () => "" }; + }); + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result.models.map((m) => m.id)).toEqual(MODELS_RESPONSE.map((m) => m.id)); + }); + + it("does not leak the cline-pass/ subscription tier into the cline list", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result.models.map((m) => m.id)).not.toContain("cline-pass/glm-5.3"); + }); +}); + +describe("cline-free namespace pricing", () => { + it("bills cline-free/* at zero", async () => { + const { getPricingForModel } = await import("../../open-sse/providers/pricing.js"); + const pricing = getPricingForModel("cline", "cline-free/deepseek-v4.1-flash"); + expect(pricing).toMatchObject({ + input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0, + }); + }); + + it("bills cline-free/* muse-spark at zero", async () => { + const { getPricingForModel } = await import("../../open-sse/providers/pricing.js"); + expect(getPricingForModel("cline", "cline-free/muse-spark-1.3-contributor").input).toBe(0); + }); + + it("still bills the paid twin at its published rate", async () => { + const { getPricingForModel } = await import("../../open-sse/providers/pricing.js"); + expect(getPricingForModel("cline", "deepseek/deepseek-v4.1-flash").input).toBe(0.14); + expect(getPricingForModel("cline", "meta/muse-spark-1.3-contributor")).toBeNull(); + }); + + it("zero price survives cost calculation over a large usage", async () => { + const { getPricingForModel, calculateCostFromTokens } = await import("../../open-sse/providers/pricing.js"); + const pricing = getPricingForModel("cline", "cline-free/deepseek-v4.1-flash"); + const cost = calculateCostFromTokens( + { prompt_tokens: 1_000_000, completion_tokens: 1_000_000, reasoning_tokens: 500_000 }, + pricing + ); + expect(cost).toBe(0); + }); +}); diff --git a/tests/unit/codex-current-provider-base-url.test.js b/tests/unit/codex-current-provider-base-url.test.js new file mode 100644 index 00000000..8c3ef06a --- /dev/null +++ b/tests/unit/codex-current-provider-base-url.test.js @@ -0,0 +1,44 @@ +import { describe, expect, it } from "vitest"; +import { getCurrentCodexProviderBaseUrl, getCurrentCodexProviderSettings } from "../../src/app/(dashboard)/dashboard/cli-tools/components/codexConfig.js"; + +describe("Codex current provider base URL", () => { + it("uses the base URL from the configured model provider, not an earlier provider", () => { + const config = `model = "gpt-5" +model_provider = "9router" + +[model_providers.omniroute] +base_url = "https://omniroute.example/v1" + +[model_providers.9router] +base_url = "http://127.0.0.1:20128/v1" +`; + + expect(getCurrentCodexProviderBaseUrl(config)).toBe("http://127.0.0.1:20128/v1"); + }); + + it("reads the active provider URL and bearer key when another provider appears first", () => { + const config = `model_provider = "9router" + +[model_providers.omniroute] +base_url = "https://omniroute.example/v1" + +[model_providers.omniroute.http_headers] +Authorization = "Bearer placeholder-omniroute-key" + +[model_providers.9router] +base_url = "https://9router.example/v1/" + +[model_providers.9router.http_headers] +Authorization = "Bearer placeholder-9router-key" +`; + + expect(getCurrentCodexProviderSettings(config)).toEqual({ + baseUrl: "https://9router.example/v1/", + apiKey: "placeholder-9router-key", + }); + }); + + it("returns empty settings when no active provider is configured", () => { + expect(getCurrentCodexProviderSettings("model = \"gpt-5\"\n")).toEqual({ baseUrl: "", apiKey: "" }); + }); +}); diff --git a/tests/unit/codex-gpt6-lite.test.js b/tests/unit/codex-gpt6-lite.test.js new file mode 100644 index 00000000..035c8329 --- /dev/null +++ b/tests/unit/codex-gpt6-lite.test.js @@ -0,0 +1,105 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; + +import { CodexExecutor } from "../../open-sse/executors/codex.js"; +import { getModelsByProviderId } from "../../open-sse/config/providerModels.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { getThinkingLevels } from "../../open-sse/providers/thinkingLevels.js"; +import * as proxyFetchModule from "../../open-sse/utils/proxyFetch.js"; + +const credentials = { connectionId: "fixture", accessToken: "fixture-token" }; +afterEach(() => vi.restoreAllMocks()); + +describe("Codex GPT-6 Sol/Luna transport", () => { + it.each(["gpt-6-sol", "gpt-6-luna"])("lists %s with Codex capabilities", (model) => { + const entry = getModelsByProviderId("codex").find((item) => item.id === model); + expect(entry?.responsesLite).toBe(true); + expect(entry?.thinkingLevels).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(getCapabilitiesForModel("codex", model)).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "openai", + }); + expect(getThinkingLevels("codex", model)).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(getThinkingLevels("codex", `${model}(high)`)).toEqual(entry.thinkingLevels); + }); + + it("keeps a native Responses Lite request intact", () => { + const executor = new CodexExecutor(); + const input = [ + { type: "additional_tools", role: "developer", tools: [{ type: "function", name: "run", parameters: { type: "object", properties: {} } }] }, + { type: "message", id: "msg_native", role: "developer", content: [{ type: "input_text", text: "Native instructions" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }, + ]; + const body = executor.transformRequest("gpt-6-luna", { + model: "gpt-6-luna", input: structuredClone(input), instructions: "", tools: null, parallel_tool_calls: false, + reasoning: { effort: "high", context: "all_turns" }, + }, true, credentials); + const headers = executor.buildHeaders(credentials, true, null, "gpt-6-luna"); + + expect(headers["x-openai-internal-codex-responses-lite"]).toBe("true"); + expect(body.instructions).toBe(""); + expect(body.tools).toBeNull(); + expect(body.parallel_tool_calls).toBe(false); + expect(body.input).toEqual(input); + expect(body.reasoning).toEqual({ effort: "high", context: "all_turns" }); + }); + + it("converts an ordinary Responses request to the Lite shape", () => { + const executor = new CodexExecutor(); + const tool = { type: "function", name: "run", parameters: { type: "object", properties: {} } }; + const body = executor.transformRequest("gpt-6-sol", { + model: "gpt-6-sol", input: "hello", instructions: "Do the task", tools: [tool], + }, true, credentials); + + expect(body.instructions).toBe(""); + expect(body.tools).toBeNull(); + expect(body.parallel_tool_calls).toBe(false); + expect(body.reasoning).toEqual({ effort: "medium", context: "all_turns" }); + expect(body.input[0]).toEqual({ type: "additional_tools", role: "developer", tools: [tool] }); + expect(body.input[1]).toEqual({ type: "message", role: "developer", content: [{ type: "input_text", text: "Do the task" }] }); + expect(executor.buildHeaders(credentials, true, null, "gpt-6-sol")["x-openai-internal-codex-responses-lite"]).toBe("true"); + }); + + it("clamps unsupported GPT-6 reasoning values to Codex's lowest supported level", () => { + const body = new CodexExecutor().transformRequest("gpt-6-luna", { + model: "gpt-6-luna", input: "hello", reasoning: { effort: "none" }, + }, true, credentials); + + expect(body.reasoning.effort).toBe("low"); + expect(body.reasoning.context).toBe("all_turns"); + }); + + it("sends the Lite shape and header in the actual outbound request", async () => { + const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ + ok: true, status: 200, headers: new Map(), + }); + await new CodexExecutor().execute({ + model: "gpt-6-luna", + body: { model: "gpt-6-luna", input: "hello", instructions: "Do the task" }, + stream: true, + credentials, + }); + + const [url, options] = fetchMock.mock.calls[0]; + const body = JSON.parse(options.body); + expect(url).toBe("https://chatgpt.com/backend-api/codex/responses"); + expect(options.headers["x-openai-internal-codex-responses-lite"]).toBe("true"); + expect(options.headers.version).toBe("0.155.0"); + expect(body.model).toBe("gpt-6-luna"); + expect(body.instructions).toBe(""); + expect(body.input[0].type).toBe("additional_tools"); + expect(body.reasoning.context).toBe("all_turns"); + }); + + it("keeps the legacy transport for other models", () => { + const executor = new CodexExecutor(); + const body = executor.transformRequest("gpt-5.5", { model: "gpt-5.5", input: "hello" }, true, credentials); + + expect(body.instructions).toBeTruthy(); + expect(body.input[0].type).not.toBe("additional_tools"); + expect(body.reasoning.context).toBeUndefined(); + expect(executor.buildHeaders(credentials, true, null, "gpt-5.5")["x-openai-internal-codex-responses-lite"]).toBeUndefined(); + expect(getThinkingLevels("codex", "gpt-6-astra")).toContain("none"); + expect(executor.buildHeaders(credentials, true, null, "gpt-6-astra")["x-openai-internal-codex-responses-lite"]).toBeUndefined(); + }); +}); diff --git a/tests/unit/codex-profiles.test.js b/tests/unit/codex-profiles.test.js new file mode 100644 index 00000000..6439f615 --- /dev/null +++ b/tests/unit/codex-profiles.test.js @@ -0,0 +1,28 @@ +import { describe, expect, it } from "vitest"; +import { + deriveProfileNameFromModel, + buildCodexProfileToml, + parseCodexProfileModel, +} from "../../src/app/(dashboard)/dashboard/cli-tools/components/codexConfig.js"; + +describe("Codex profiles configuration", () => { + it("derives provider name as profile name and avoids conflicts", () => { + expect(deriveProfileNameFromModel("anthropic/claude-3-7-sonnet")).toBe("anthropic"); + expect(deriveProfileNameFromModel("anthropic/claude-3-5-haiku", ["anthropic"])).toBe("anthropic-2"); + expect(deriveProfileNameFromModel("anthropic/claude-3-5-haiku", ["anthropic", "anthropic-2"])).toBe("anthropic-3"); + expect(deriveProfileNameFromModel("deepseek/deepseek-chat")).toBe("deepseek"); + expect(deriveProfileNameFromModel("google/gemini-2.5-pro")).toBe("google"); + }); + + it("derives model name when model has no slash", () => { + expect(deriveProfileNameFromModel("claude-3-7-sonnet")).toBe("claude-3-7-sonnet"); + expect(deriveProfileNameFromModel("gpt-4o")).toBe("gpt-4o"); + }); + + it("builds and parses profile TOML", () => { + const toml = buildCodexProfileToml({ name: "claude", model: "anthropic/claude-3-7-sonnet" }); + expect(toml).toContain('model = "anthropic/claude-3-7-sonnet"'); + expect(toml).toContain('model_provider = "9router"'); + expect(parseCodexProfileModel(toml)).toBe("anthropic/claude-3-7-sonnet"); + }); +}); diff --git a/tests/unit/codex-settings-refresh.test.js b/tests/unit/codex-settings-refresh.test.js new file mode 100644 index 00000000..2e4af6f1 --- /dev/null +++ b/tests/unit/codex-settings-refresh.test.js @@ -0,0 +1,30 @@ +import { readFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import { describe, expect, it } from "vitest"; + +const readSource = (relativePath) => + readFile(fileURLToPath(new URL(relativePath, import.meta.url)), "utf8"); + +describe("Codex settings refresh", () => { + it("bypasses cached status after applying a selected endpoint", async () => { + const [routeSource, cardSource] = await Promise.all([ + readSource("../../src/app/api/cli-tools/codex-settings/route.js"), + readSource("../../src/app/(dashboard)/dashboard/cli-tools/components/CodexToolCard.js"), + ]); + + // Route Handlers already run on the server; a Server Action directive would reject this export. + expect(routeSource).not.toContain('"use server";'); + expect(routeSource).toContain('export const dynamic = "force-dynamic";'); + expect(cardSource).toContain('fetch("/api/cli-tools/codex-settings", { cache: "no-store" })'); + expect(cardSource).toContain("setSelectedApiKey(apiKey);"); + expect(cardSource).toContain("setCustomBaseUrl(baseUrl);"); + }); + + it("keeps an unmatched active URL in the custom endpoint slot", async () => { + const selectorSource = await readSource("../../src/app/(dashboard)/dashboard/cli-tools/components/BaseUrlSelect.js"); + + expect(selectorSource).toContain("if (current) {"); + expect(selectorSource).toContain("setCustomInput(current);"); + expect(selectorSource).toContain("onChange(current);"); + }); +}); diff --git a/tests/unit/combo-caps-resolver.test.js b/tests/unit/combo-caps-resolver.test.js new file mode 100644 index 00000000..24d0d388 --- /dev/null +++ b/tests/unit/combo-caps-resolver.test.js @@ -0,0 +1,67 @@ +import { describe, expect, it } from "vitest"; + +import { aggregateComboCapabilities, getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; + +// A combo's limits are the conservative aggregate of its members: ctx = min, +// maxOutput = max. Resolving those members needs the synced model catalog, which +// is server-only (it reads a file), so the browser bundle falls back to the +// generic patterns. The dashboard computed its badges there and under-reported: +// /v1/models and pi-settings (both server-side) said 1M while the badge said 200k. +// +// resolveCaps lets a caller hand in the server's answer. It must only override +// what it carries — the local tables still own tools/pdf/audio/video/thinking*. +const GLM53_FED = { vision: true, search: false, reasoning: true, contextWindow: 1_000_000, maxOutput: 131_072 }; + +describe("aggregateComboCapabilities: resolveCaps override", () => { + const models = ["glm-cn/glm-5.3", "deepseek-v4.1-flash"]; + + it("falls back to the pattern default without a resolver", () => { + const caps = aggregateComboCapabilities(models); + // glm-5.3 has no exact entry, so the *glm-5.3* pattern gives 200k and caps the combo. + expect(caps.contextWindow).toBe(200_000); + }); + + it("uses the fed limits when a resolver supplies them", () => { + const resolver = (fullId) => (fullId === "glm-cn/glm-5.3" ? GLM53_FED : null); + const caps = aggregateComboCapabilities(models, null, resolver); + expect(caps.contextWindow).toBe(1_000_000); + }); + + it("keeps the fields the override does not carry", () => { + const plain = aggregateComboCapabilities(models); + const fed = aggregateComboCapabilities(models, null, (id) => (id === "glm-cn/glm-5.3" ? GLM53_FED : null)); + // The override carries no tools/pdf/thinking fields, so those must be unchanged. + for (const field of ["tools", "pdf", "audioInput", "videoInput", "imageOutput", "audioOutput", "thinkingFormat"]) { + expect(fed[field]).toEqual(plain[field]); + } + }); + + it("still applies the conservative rule across members", () => { + const resolver = (fullId) => (fullId === "glm-cn/glm-5.3" ? GLM53_FED : null); + const caps = aggregateComboCapabilities(models, null, resolver); + // Only glm-5.3 was fed 1M; deepseek-v4.1-flash resolves locally to 1M, so min stays 1M. + // Feeding a *smaller* value for one member must pull the aggregate down. + const smaller = aggregateComboCapabilities(models, null, (id) => (id === "glm-cn/glm-5.3" ? { ...GLM53_FED, contextWindow: 64_000 } : null)); + expect(smaller.contextWindow).toBe(64_000); + expect(caps.maxOutput).toBe(384_000); // max across members, from deepseek + }); + + it("passes the resolver into nested combos", () => { + const lookup = { + zap: ["deepseek-v4.1-flash", "glm-cn/glm-5.3-flash"], + "deepseek-v4.1-flash": ["cmc/deepseek/deepseek-v4.1-flash", "ocg/deepseek-v4.1-flash"], + }; + const seen = []; + const resolver = (fullId) => { seen.push(fullId); return fullId === "glm-cn/glm-5.3-flash" ? { contextWindow: 1_000_000 } : null; }; + aggregateComboCapabilities(lookup.zap, lookup, resolver); + // The nested combo's own members were resolved with the same resolver. + expect(seen).toContain("cmc/deepseek/deepseek-v4.1-flash"); + expect(seen).toContain("ocg/deepseek-v4.1-flash"); + }); + + it("leaves the plain two-argument call unchanged", () => { + const caps = aggregateComboCapabilities(["kimi/kimi-k3"], null); + expect(caps).toEqual(aggregateComboCapabilities(["kimi/kimi-k3"])); + expect(caps.contextWindow).toBe(getCapabilitiesForModel("kimi", "kimi-k3").contextWindow); + }); +}); diff --git a/tests/unit/commandcode-executor.test.js b/tests/unit/commandcode-executor.test.js index f498b39c..3c97e27a 100644 --- a/tests/unit/commandcode-executor.test.js +++ b/tests/unit/commandcode-executor.test.js @@ -133,6 +133,30 @@ describe("inspectAndWrapCommandCodeResponse", () => { expect(text).toContain("data: [DONE]"); }); + it("preserves all lines in a multi-line packet when inspecting tool-input-start", async () => { + const packet = [ + JSON.stringify({ type: "start" }), + JSON.stringify({ type: "start-step" }), + JSON.stringify({ type: "tool-input-start", id: "call_1", toolName: "terminal" }), + JSON.stringify({ type: "tool-input-delta", id: "call_1", delta: '{"command": "ls"}' }), + JSON.stringify({ type: "finish-step", finishReason: "tool-calls" }), + JSON.stringify({ type: "finish", finishReason: "tool-calls" }), + ].join("\n") + "\n"; + + const ndjsonBody = createNdjsonStream([packet]); + + const fakeResponse = new Response(ndjsonBody, { + status: 200, + headers: { "Content-Type": "text/event-stream" }, + }); + + const result = await inspectAndWrapCommandCodeResponse(fakeResponse, "cmc/deepseek/deepseek-v4.1-flash"); + expect(result.ok).toBe(true); + const text = await result.text(); + expect(text).toContain('"name":"terminal"'); + expect(text).toContain('"arguments":"{\\"command\\": \\"ls\\"}"'); + }); + it("retries when initial stream yields an error and succeeds on second attempt", async () => { let callCount = 0; const executor = new CommandCodeExecutor(); diff --git a/tests/unit/gemini-contents-normalization.test.js b/tests/unit/gemini-contents-normalization.test.js new file mode 100644 index 00000000..4a4b46b0 --- /dev/null +++ b/tests/unit/gemini-contents-normalization.test.js @@ -0,0 +1,141 @@ +import { describe, it, expect } from "vitest"; +import { normalizeGeminiContents } from "../../open-sse/translator/formats/gemini.js"; + +describe("normalizeGeminiContents terminal turn guards", () => { + it("appends user Continue turn when ending with model text turn", () => { + const contents = [ + { role: "user", parts: [{ text: "hi" }] }, + { role: "model", parts: [{ text: "hello" }] } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[2]).toEqual({ role: "user", parts: [{ text: "Continue." }] }); + }); + + it("appends functionResponse user turn when ending with functionCall", () => { + const contents = [ + { role: "user", parts: [{ text: "run" }] }, + { + role: "model", + parts: [ + { functionCall: { id: "call_1", name: "search", args: { q: "test" } } } + ] + } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[2]).toEqual({ + role: "user", + parts: [ + { + functionResponse: { + id: "call_1", + name: "search", + response: { result: "Continue." } + } + } + ] + }); + }); + + it("handles multiple functionCalls in terminal model turn", () => { + const contents = [ + { role: "user", parts: [{ text: "run" }] }, + { + role: "model", + parts: [ + { functionCall: { id: "call_1", name: "fn_1" } }, + { functionCall: { id: "call_2", name: "fn_2" } } + ] + } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[2].parts).toHaveLength(2); + expect(out[2].parts[0].functionResponse.id).toBe("call_1"); + expect(out[2].parts[1].functionResponse.id).toBe("call_2"); + }); + + it("handles terminal model turn with both text and functionCall", () => { + const contents = [ + { role: "user", parts: [{ text: "run" }] }, + { + role: "model", + parts: [ + { text: "Executing..." }, + { functionCall: { id: "call_3", name: "exec" } } + ] + } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[2].parts[0].functionResponse.id).toBe("call_3"); + }); + + it("handles single model turn by prepending user prompt and appending terminal user", () => { + const contents = [{ role: "model", parts: [{ text: "prefill" }] }]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[0]).toEqual({ role: "user", parts: [{ text: "..." }] }); + expect(out[1]).toEqual({ role: "model", parts: [{ text: "prefill" }] }); + expect(out[2]).toEqual({ role: "user", parts: [{ text: "Continue." }] }); + }); + + it("does not mutate payloads already ending with a user turn", () => { + const contents = [{ role: "user", parts: [{ text: "question" }] }]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(1); + expect(out[0].role).toBe("user"); + }); + + it("handles functionCall without name or id with fallback defaults", () => { + const contents = [ + { role: "user", parts: [{ text: "Go" }] }, + { role: "model", parts: [{ functionCall: {} }] } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[2].parts[0]).toEqual({ + functionResponse: { + name: "tool", + response: { result: "Continue." } + } + }); + expect(out[2].parts[0].functionResponse.id).toBeUndefined(); + }); + + it("merges adjacent model turns before appending terminal user turn", () => { + const contents = [ + { role: "user", parts: [{ text: "Prompt" }] }, + { role: "model", parts: [{ text: "Part A" }] }, + { role: "model", parts: [{ text: "Part B" }] } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[1].role).toBe("model"); + expect(out[1].parts).toHaveLength(2); + expect(out[2]).toEqual({ role: "user", parts: [{ text: "Continue." }] }); + }); + + it("appends user Continue turn when terminal model turn has thought parts", () => { + const contents = [ + { role: "user", parts: [{ text: "Solve math" }] }, + { + role: "model", + parts: [ + { thought: true, text: "Let 2x = 4..." }, + { thoughtSignature: "sig123", text: "" } + ] + } + ]; + const out = normalizeGeminiContents(contents); + expect(out).toHaveLength(3); + expect(out[2]).toEqual({ role: "user", parts: [{ text: "Continue." }] }); + }); + + it("handles empty, null, and undefined inputs gracefully", () => { + expect(normalizeGeminiContents([])).toEqual([]); + expect(normalizeGeminiContents(null)).toEqual([]); + expect(normalizeGeminiContents(undefined)).toEqual([]); + }); +}); diff --git a/tests/unit/gemini-live-stt.test.js b/tests/unit/gemini-live-stt.test.js new file mode 100644 index 00000000..3699f27a --- /dev/null +++ b/tests/unit/gemini-live-stt.test.js @@ -0,0 +1,538 @@ +// Gemini Live (realtime bidi) STT transport contract. +// +// Black-box tests against open-sse/handlers/sttCore.js. Wire observables only: +// - transport marker drives dispatch (caller param / registry entry), never a +// hardcoded model id; +// - session opens with a setup frame declaring the model; audio rides +// realtimeInput frames only AFTER the server's setup-complete ack; +// - inputTranscription deltas accumulate into {text}; verbose_json adds +// segments {id,text} with NO timing keys (protocol carries none); +// - error frame → gateway error envelope (any 4xx/5xx, shape only); +// - system_instruction / prompt override setup instruction (substring); +// - client-supplied setup_timeout_ms (tiny) bounds the wait (error occurs); +// - response_format never reaches the session setup; +// - custom-model transport persists via POST /api/models/custom (whitelist) +// with unknown values silently dropped; +// - the persisted custom transport is resolved by the app layer (stt.js) +// and reaches engine dispatch end-to-end (handleStt → WS, not REST). +// +// NOT pinned (unstated or implementation-only): goAway/reconnect semantics, +// timeout clamp ceilings, specific status codes, byte-exact WS URLs (only +// wss:// + bidiGenerateContent + model-id substrings), exact frame JSON paths. +import { describe, it, expect, afterEach, vi, beforeEach } from "vitest"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +import { handleSttCore } from "open-sse/handlers/sttCore.js"; +import { PROVIDER_MODELS } from "open-sse/config/providerModels.js"; + +// ── fixtures ────────────────────────────────────────────────────────────── + +const STTCFG = { + baseUrl: "https://generativelanguage.googleapis.com/v1beta/models", + authType: "apikey", + authHeader: "key", + format: "gemini-stt", +}; +const CRED = { apiKey: "AIza-TEST" }; + +const LIVE_ID = "probe-live-capability-1"; + +function mkFile() { + return new File([new Uint8Array([1, 2, 3, 4])], "a.wav", { type: "audio/wav" }); +} + +function mkFormData(extra = {}) { + const fd = new FormData(); + fd.set("file", mkFile()); + for (const [k, v] of Object.entries(extra)) fd.set(k, v); + return fd; +} + +// ── fake WebSocket ──────────────────────────────────────────────────────── + +class FakeWS { + static instances = []; + static CONNECTING = 0; + static OPEN = 1; + static CLOSING = 2; + static CLOSED = 3; + + constructor(url) { + this.url = url; + this.sent = []; // JSON.parsed frames, in send order + this.readyState = FakeWS.CONNECTING; + this.closed = false; + this.closeCalls = []; // {code, reason} recordings + this._listeners = {}; + FakeWS.instances.push(this); + queueMicrotask(() => { + if (this.closed) return; + this.readyState = FakeWS.OPEN; + if (typeof this.onopen === "function") this.onopen({}); + (this._listeners.open || []).forEach((f) => f({})); + }); + } + + addEventListener(type, fn) { + (this._listeners[type] = this._listeners[type] || []).push(fn); + } + + send(data) { + this.sent.push(JSON.parse(data)); + } + + // Fire a server frame through both supported binding styles. + emit(obj) { + const ev = { data: JSON.stringify(obj) }; + if (typeof this.onmessage === "function") this.onmessage(ev); + (this._listeners.message || []).forEach((f) => f(ev)); + } + + // Fire a server-initiated close through both supported binding styles. + // Distinct from close(), which only records the client-side shutdown. + emitClose(code = 1000) { + this.closed = true; + this.readyState = FakeWS.CLOSED; + const ev = { code, reason: "" }; + if (typeof this.onclose === "function") this.onclose(ev); + (this._listeners.close || []).forEach((f) => f(ev)); + } + + close(code, reason) { + this.closed = true; + this.readyState = FakeWS.CLOSED; + this.closeCalls.push({ code: code ?? 1000, reason: reason ?? "" }); + } +} + +function stubWs() { + vi.stubGlobal("WebSocket", FakeWS); +} + +// Fetch spy that records calls; handler defaults to "REST must not happen". +function stubFetch(handler = () => { throw new Error("REST must not be used for live transport"); }) { + const calls = []; + vi.stubGlobal("fetch", async (url, opts) => { + calls.push(String(url && url.url ? url.url : url)); + return handler(url, opts); + }); + return calls; +} + +// Server drives a completed session: setup ack → transcription deltas → turn done. +function serverScript(instance, texts) { + instance.emit({ serverContent: { setupComplete: true } }); + for (const t of texts) instance.emit({ serverContent: { inputTranscription: { text: t } } }); + instance.emit({ serverContent: { turnComplete: true } }); +} + +async function liveSession({ model = LIVE_ID, formData = mkFormData(), transport = "gemini-live" } = {}) { + stubWs(); + const fetchCalls = stubFetch(); + const pending = handleSttCore({ provider: "gemini", model, formData, credentials: CRED, sttConfig: STTCFG, transport }); + await vi.waitFor(() => expect(FakeWS.instances.length).toBe(1)); + const ws = FakeWS.instances[0]; + await vi.waitFor(() => expect(ws.sent.length).toBeGreaterThanOrEqual(1)); // setup frame sent + return { pending, ws, fetchCalls }; +} + +afterEach(() => { + vi.unstubAllGlobals(); + FakeWS.instances.length = 0; +}); + +// ── S1/S2/S3: transport marker dispatch, text envelope, no REST ─────────── + +describe("Live transport dispatch via caller marker", () => { + it("T3: transport 'gemini-live' opens a WebSocket, never REST; text = accumulated deltas", async () => { + // REST-fallback contrast (folded from T1): a live-capability id with no + // transport marker falls to REST and fails cleanly — the live path is opt-in. + stubFetch(() => ({ + ok: false, + status: 400, + text: async () => JSON.stringify({ error: { message: "live models require the streaming endpoint" } }), + })); + const restResult = await handleSttCore({ + provider: "gemini", + model: LIVE_ID, + formData: mkFormData(), + credentials: CRED, + sttConfig: STTCFG, + }); + expect(restResult.success).toBe(false); + + // Caller-marker dispatch: explicit transport "gemini-live" opens the WS, + // never REST; text = accumulated inputTranscription deltas. + const { pending, ws, fetchCalls } = await liveSession(); + serverScript(ws, ["hello ", "world"]); + const result = await pending; + expect(result.success).toBe(true); + await expect(result.response.json()).resolves.toEqual({ text: "hello world" }); + expect(fetchCalls).toHaveLength(0); + + // Registry-marker dispatch: the live entry's transport field alone — no + // caller transport param — routes to the WS path; id derived from the + // registry, never a literal. + const regId = (PROVIDER_MODELS.gemini || []).find( + (m) => m && m.kind === "stt" && m.transport === "gemini-live", + )?.id; + FakeWS.instances.length = 0; + stubWs(); + const regFetchCalls = stubFetch(); + const regPending = handleSttCore({ + provider: "gemini", + model: regId, + formData: mkFormData(), + credentials: CRED, + sttConfig: STTCFG, + }); + await vi.waitFor(() => expect(FakeWS.instances.length).toBe(1)); + const regWs = FakeWS.instances[0]; + await vi.waitFor(() => expect(regWs.sent.length).toBeGreaterThanOrEqual(1)); + serverScript(regWs, ["reg ", "live"]); + const regResult = await regPending; + expect(regResult.success).toBe(true); + await expect(regResult.response.json()).resolves.toEqual({ text: "reg live" }); + expect(regFetchCalls).toHaveLength(0); + }); + + it("T4: setup frame first (carries model id); audio only in realtimeInput at index >=1; WS URL is the bidi endpoint", async () => { + const { pending, ws, fetchCalls } = await liveSession(); + const setup = ws.sent[0]; + expect(setup.setup).toBeTruthy(); + expect(JSON.stringify(setup.setup)).toContain(LIVE_ID); + expect(setup.realtimeInput).toBeUndefined(); + + serverScript(ws, ["x"]); + const result = await pending; + expect(result.success).toBe(true); + expect(fetchCalls).toHaveLength(0); + + const audioIdx = ws.sent.findIndex((f) => f.realtimeInput); + expect(audioIdx).toBeGreaterThanOrEqual(1); + const media = ws.sent[audioIdx].realtimeInput.mediaChunks; + expect(Array.isArray(media)).toBe(true); + expect(typeof media[0].data).toBe("string"); + expect(media[0].data.length).toBeGreaterThan(0); + expect(Buffer.from(media[0].data, "base64").length).toBeGreaterThan(0); + + expect(ws.url).toContain("wss://"); + expect(ws.url).toContain("bidiGenerateContent"); + expect(ws.url).toContain(LIVE_ID); + + // REST contrast (folded from T2): an ordinary gemini model still transcribes + // over REST generateContent — the live path is opt-in, never the default. + stubFetch(() => ({ + ok: true, + status: 200, + json: async () => ({ candidates: [{ content: { parts: [{ text: "hello rest" }] } }] }), + text: async () => "", + })); + const restResult = await handleSttCore({ + provider: "gemini", + model: "gemini-2.0-flash", + formData: mkFormData(), + credentials: CRED, + sttConfig: STTCFG, + }); + expect(restResult.success).toBe(true); + await expect(restResult.response.json()).resolves.toEqual({ text: "hello rest" }); + }); + + it("T5: server error frame yields the gateway error envelope (any 4xx/5xx, no text pin)", async () => { + const { pending, ws } = await liveSession(); + ws.emit({ error: { code: "X", message: "Y" } }); + const result = await pending; + expect(result.success).toBe(false); + expect(typeof result.status).toBe("number"); + expect(result.status).toBeGreaterThanOrEqual(400); + expect(result.status).toBeLessThanOrEqual(599); + }); +}); + +// ── S7: client knobs (instruction overrides ride the setup frame) ───────── + +describe("Setup frame knobs", () => { + it("T6: client prompt overrides the setup instruction", async () => { + const { pending, ws } = await liveSession({ formData: mkFormData({ prompt: "Say it back" }) }); + const setupJson = JSON.stringify(ws.sent[0]); + expect(setupJson).toContain("Say it back"); + serverScript(ws, ["ok"]); + const result = await pending; + expect(result.success).toBe(true); + }); + + it("T7: system_instruction override appears in the setup frame", async () => { + const { pending, ws } = await liveSession({ formData: mkFormData({ system_instruction: "TRANSCRIBE-VERBATIM-OVERRIDE-42" }) }); + const setupJson = JSON.stringify(ws.sent[0]); + expect(setupJson).toContain("TRANSCRIBE-VERBATIM-OVERRIDE-42"); + serverScript(ws, ["ok"]); + const result = await pending; + expect(result.success).toBe(true); + }); + + it("T8: response_format is client-only and never reaches the session setup", async () => { + const { pending, ws } = await liveSession({ formData: mkFormData({ response_format: "verbose_json" }) }); + expect(JSON.stringify(ws.sent[0])).not.toContain("response_format"); + serverScript(ws, ["a", "b"]); + const result = await pending; + expect(result.success).toBe(true); + }); + + it("T9: tiny setup_timeout_ms with no server ack errors out within the bound", async () => { + const { pending } = await liveSession({ formData: mkFormData({ setup_timeout_ms: "5" }) }); + // deliberately emit nothing — the client knob must end the wait + const result = await pending; + expect(result.success).toBe(false); + expect(typeof result.status).toBe("number"); + expect(result.status).toBeGreaterThanOrEqual(400); + expect(result.status).toBeLessThanOrEqual(599); + }); +}); + +// ── S8: verbose_json shaping ────────────────────────────────────────────── + +describe("Response shaping", () => { + it("T10: verbose_json adds {id,text} segments in arrival order with NO timing fields", async () => { + const { pending, ws } = await liveSession({ formData: mkFormData({ response_format: "verbose_json" }) }); + serverScript(ws, ["hello ", "world"]); + const result = await pending; + const body = await result.response.json(); + expect(body.text).toBe("hello world"); + expect(Array.isArray(body.segments)).toBe(true); + expect(body.segments).toHaveLength(2); + const [s0, s1] = body.segments; + expect(s0.id).toBe(0); + expect(s1.id).toBe(1); + for (const seg of body.segments) { + expect(Object.keys(seg)).toContain("id"); + expect(Object.keys(seg)).toContain("text"); + expect("start" in seg).toBe(false); + expect("end" in seg).toBe(false); + expect("duration" in seg).toBe(false); + } + expect(s0.text).toBe("hello "); + expect(s1.text).toBe("world"); + expect("duration" in body).toBe(false); + }); + + it("T11: default format carries text only, no segments", async () => { + const { pending, ws } = await liveSession(); + serverScript(ws, ["one", "two"]); + const result = await pending; + const body = await result.response.json(); + expect(Object.keys(body)).toContain("text"); + expect("segments" in body).toBe(false); + }); +}); + +// ── S2a: registry marks the live family (data, not code) ───────────────── + +describe("Registry family marking", () => { + it("T12: gemini stt catalog includes a live-transport entry advertising the lifecycle params", () => { + const live = (PROVIDER_MODELS.gemini || []).find( + (m) => m && m.kind === "stt" && m.transport === "gemini-live", + ); + expect(live).toBeTruthy(); + expect(live.id).toBeTruthy(); + expect(Array.isArray(live.params)).toBe(true); + for (const p of ["language", "prompt", "system_instruction", "setup_timeout_ms", "turn_timeout_ms"]) { + expect(live.params).toContain(p); + } + }); +}); + +// ── S2b: custom-model transport persistence (route level) ───────────────── + +describe("Custom-model transport persistence via POST /api/models/custom", () => { + let tempDir; + const originalDataDir = process.env.DATA_DIR; + + beforeEach(() => { + // paths.js freezes DATA_DIR at module load — re-evaluate the db chain per test + vi.resetModules(); + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "9router-gemini-live-")); + process.env.DATA_DIR = tempDir; + delete global._dbAdapter; + }); + + afterEach(() => { + try { global._dbAdapter?.instance?.close?.(); } catch { /* already closed */ } + delete global._dbAdapter; + if (tempDir) fs.rmSync(tempDir, { recursive: true, force: true }); + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; + }); + + async function postCustom(payload) { + const { POST } = await import("@/app/api/models/custom/route.js"); + const res = await POST({ json: async () => payload }); + return res.json(); + } + + async function customRows() { + const { getCustomModels } = await import("@/lib/db/repos/aliasRepo.js"); + return getCustomModels(); + } + + it("T13: whitelisted transport persists on the saved row", { timeout: 30000 }, async () => { + const body = await postCustom({ + providerAlias: "gemini", + id: "probe-custom-capability-9", + type: "stt", + transport: "gemini-live", + }); + expect(body.success).toBe(true); + const row = (await customRows()).find((m) => m && m.providerAlias === "gemini" && m.id === "probe-custom-capability-9"); + expect(row).toBeTruthy(); + expect(row.type).toBe("stt"); + expect(row.transport).toBe("gemini-live"); + }); + + it("T14: unknown transport is silently dropped — prior whitelisted transport survives re-save", { timeout: 30000 }, async () => { + const first = await postCustom({ + providerAlias: "gemini", + id: "probe-custom-capability-10", + type: "stt", + transport: "gemini-live", + }); + expect(first.success).toBe(true); + // Re-saving the same model with an unknown transport must not clobber + // the persisted marker: silent-drop keeps the stored value (merge keeps + // omitted fields, per addCustomModel). + const second = await postCustom({ + providerAlias: "gemini", + id: "probe-custom-capability-10", + type: "stt", + transport: "nope", + }); + expect(second.success).toBe(true); + const row = (await customRows()).find((m) => m && m.providerAlias === "gemini" && m.id === "probe-custom-capability-10"); + expect(row).toBeTruthy(); + expect(row.transport).toBe("gemini-live"); + }); +}); + +// ── S2c (T15): app-layer custom-transport resolution, end-to-end ───────── +// +// Gate C M1 closure. Only handleStt (src/sse/handlers/stt.js) maps a +// persisted custom-model transport onto the handleSttCore dispatch; T13/T14 +// stop at repo persistence. Fake model id is absent from the registry, so a +// WebSocket opening here is observable proof the caller-supplied transport +// marker was resolved and passed — deleting that resolution fails T15. + +describe("App-layer custom transport resolution (stt.js)", () => { + it("T15: persisted custom gemini-live transport reaches WS dispatch through real handleStt", async () => { + const LOCALDB = "@/lib/localDb"; + const AUTH = "../../src/sse/services/auth.js"; + try { + vi.resetModules(); + vi.doMock(LOCALDB, () => ({ + getSettings: async () => ({ requireApiKey: false }), + getCustomModels: async () => ([{ + providerAlias: "gemini", id: "probe-sttjs-1", type: "stt", transport: "gemini-live", + }]), + getProviderConnections: async () => [{ id: "c1", provider: "gemini", isActive: true }], + getProviderNodes: async () => [], + getModelAliases: async () => ({}), + getComboByName: async () => null, + })); + vi.doMock(AUTH, () => ({ + extractApiKey: () => null, + isValidApiKey: async () => true, + getProviderCredentials: async () => ({ + apiKey: "AIza-TEST", connectionId: "c1", connectionName: "t", providerSpecificData: {}, + }), + markAccountUnavailable: async () => ({ shouldFallback: false }), + })); + // getModelInfo stays REAL: "gemini/..." is a reserved-prefix passthrough, + // so the parse→route hop in the chain is exercised, not stubbed. + const { handleStt } = await import("../../src/sse/handlers/stt.js"); + + stubWs(); + const fetchCalls = stubFetch(); // default handler throws: REST must not happen + const fd = mkFormData(); + fd.set("model", "gemini/probe-sttjs-1"); + + const pending = handleStt({ formData: async () => fd }); + await vi.waitFor(() => expect(FakeWS.instances.length).toBe(1)); + const ws = FakeWS.instances[0]; + await vi.waitFor(() => expect(ws.sent.length).toBeGreaterThanOrEqual(1)); // setup first + serverScript(ws, ["hello ", "world"]); + const res = await pending; + expect(res.status).toBe(200); + const body = await res.json(); + expect(body.text).toBe("hello world"); + expect(fetchCalls).toHaveLength(0); + } finally { + vi.doUnmock(LOCALDB); + vi.doUnmock(AUTH); + } + }); +}); + +// ── S5/S6/S9: lifecycle gates (setup ack, turn timeout, turn completion) ── + +// Setup/turn gating, graceful close, and transcript accumulation hygiene: +// padding-only frames are dropped, and a whitespace-only run is a failure +// rather than a blank success. +describe("Lifecycle gates", () => { + it("T16: audio streaming waits for the server setup-complete reply", async () => { + const { pending, ws } = await liveSession(); + // deliberately do NOT emit setupComplete + await new Promise((r) => setTimeout(r, 80)); + const audioFrames = ws.sent.filter((f) => f.realtimeInput); + expect(audioFrames).toHaveLength(0); + // clean up: let the pending promise settle so afterEach unstub works cleanly + ws.emit({ serverContent: { setupComplete: true } }); + ws.emit({ serverContent: { turnComplete: true } }); + await pending; + }); + + it("T17: tiny turn_timeout_ms with setup-complete but no turn-complete errors out", async () => { + const { pending, ws } = await liveSession({ formData: mkFormData({ turn_timeout_ms: "5" }) }); + ws.emit({ serverContent: { setupComplete: true } }); + // deliberately do NOT emit turnComplete + const result = await pending; + expect(result.success).toBe(false); + expect(typeof result.status).toBe("number"); + expect(result.status).toBeGreaterThanOrEqual(400); + expect(result.status).toBeLessThanOrEqual(599); + }); + + it("T18: server turnComplete closes the WebSocket gracefully", async () => { + const { pending, ws } = await liveSession(); + serverScript(ws, ["done"]); + await pending; + expect(ws.closeCalls.length).toBeGreaterThanOrEqual(1); + expect(ws.closeCalls[0].code).toBe(1000); + }); + + it("T19: whitespace-only transcription frames are not appended to the transcript", async () => { + const { pending, ws } = await liveSession(); + ws.emit({ serverContent: { setupComplete: true } }); + // a padding-only frame must contribute nothing to the transcript + ws.emit({ serverContent: { inputTranscription: { text: " " } } }); + ws.emit({ serverContent: { inputTranscription: { text: "done" } } }); + ws.emit({ serverContent: { turnComplete: true } }); + const result = await pending; + const body = await result.response.json(); + expect(body.text).toBe("done"); + }); + + it("T20: a run that receives only whitespace frames errors instead of returning a blank transcript", async () => { + const { pending, ws } = await liveSession(); + ws.emit({ serverContent: { setupComplete: true } }); + ws.emit({ serverContent: { inputTranscription: { text: " " } } }); + // server closes before any real transcript arrived: a partial success would + // hand the client a whitespace-only transcript + ws.emitClose(1000); + const result = await pending; + expect(result.success).toBe(false); + expect(typeof result.status).toBe("number"); + expect(result.status).toBeGreaterThanOrEqual(400); + expect(result.status).toBeLessThanOrEqual(599); + }); +}); diff --git a/tests/unit/model-catalog-scope.test.js b/tests/unit/model-catalog-scope.test.js index c582d28f..5177901f 100644 --- a/tests/unit/model-catalog-scope.test.js +++ b/tests/unit/model-catalog-scope.test.js @@ -122,6 +122,22 @@ describe("model catalog", () => { } expect(globalThis.__9rCatalogSource).toBeNull(); }); + + it("detaches the source from a copy that already resolved through it", async () => { + capabilities.setCatalogSource({ + getModalities: (provider) => (provider === "gateway-a" ? { vision: true } : null), + getLimits: () => null, + }); + const other = await import("../../open-sse/providers/capabilities.js?copy=3"); + try { + expect(other.getCapabilitiesForModel("gateway-a", "laguna-9-preview").vision).toBe(true); + } finally { + capabilities.setCatalogSource(null); + } + // the sync resets the source before rebuilding; a copy that has read the + // slot once must not keep serving the uninstalled reader + expect(other.getCapabilitiesForModel("gateway-a", "laguna-9-preview").vision).toBe(false); + }); }); describe("catalog schema", () => { diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 92e7a059..fcdf3888 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -4,13 +4,14 @@ import { PROVIDERS } from "../../open-sse/config/providers.js"; import { resolveTransport } from "../../open-sse/services/provider.js"; // Chat-only models (no /messages, no /responses support on opencode-go) -const CHAT_ONLY = ["glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", - "deepseek-flash", "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", "hy4-preview", "hy3"]; +const CHAT_ONLY = ["glm-5.3", "glm-5.2", "glm-5.1", "glm-5", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "kimi-k3", + "deepseek-flash", "longcat-2.0", "mimo-v2.6-flash", "mimo-v2.6-pro", + "mimo-v2.5", "mimo-v2.5-pro", "mimo-v2-pro", "mimo-v2-omni", "hy4-preview", "hy3", "hy3-preview", "omen-alpha"]; // Models that also expose the Anthropic /messages endpoint -const CLAUDE_CAPABLE = ["minimax-m3", "minimax-m2.7", "minimax-m2.5", - "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus"]; +const CLAUDE_CAPABLE = ["minimax-m3", "minimax-m2.7", "minimax-m2.5", "space-bunny-free", + "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.5-plus"]; // Models that also expose the OpenAI /responses endpoint -const RESPONSES_CAPABLE = ["deepseek-v4-pro", "deepseek-v4-flash"]; +const RESPONSES_CAPABLE = ["deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4.1-flash"]; // Mirror of chatCore's per-model transport guard: use the sourceFormat-matched // transport only when the model declares support for that sourceFormat. @@ -25,18 +26,42 @@ describe("OpenCode Go model catalog", () => { const ids = (PROVIDER_MODELS["opencode-go"] || []).map((m) => m.id); expect(ids).toEqual([ "deepseek-flash", - "glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", - "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", - "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", - "minimax-m3", "minimax-m2.7", "minimax-m2.5", - "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", - "hy4-preview", "hy3", - "grok-4.6", "gpt-5.6-luna", + "glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "glm-5", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "kimi-k3", + "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", "deepseek-v4.1-flash", + "longcat-2.0", "mimo-v2.6-flash", "mimo-v2.6-pro", "mimo-v2.5", "mimo-v2.5-pro", "mimo-v2-pro", "mimo-v2-omni", + "minimax-m3", "minimax-m2.7", "minimax-m2.5", "space-bunny-free", + "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.5-plus", + "hy4-preview", "hy3", "hy3-preview", "omen-alpha", + "grok-4.7", "grok-4.6", "grok-4.5", "gpt-5.6-luna", "gpt-6-luna", "muse-spark-1.2-contributor", "muse-spark-1.3-contributor", ]); }); }); +describe("OpenCode Go family fallback (unknown/passthrough ids)", () => { + it("routes unknown grok/gpt ids to the responses lane", () => { + expect(getModelSupportedFormats("opencode-go", "grok-4.8")).toEqual(["openai-responses"]); + expect(getModelTargetFormat("opencode-go", "gpt-6-foo")).toBe("openai-responses"); + }); + + it("gives unknown chat-family ids the chat-only lane, never /messages", () => { + for (const m of ["kimi-k4", "glm-6", "mimo-v3", "omen-beta"]) { + expect(getModelSupportedFormats("opencode-go", m)).toEqual(["openai"]); + } + }); + + it("keeps unknown minimax/qwen ids on the /messages lane too", () => { + for (const m of ["minimax-m9", "qwen4-max"]) { + expect(getModelSupportedFormats("opencode-go", m)).toEqual(["openai", "claude"]); + } + }); + + it("curated entries win over the family regex", () => { + expect(getModelSupportedFormats("opencode-go", "deepseek-flash")).toEqual(["openai"]); + expect(getModelSupportedFormats("opencode-go", "deepseek-v4-pro")).toEqual(["openai", "claude", "openai-responses"]); + }); +}); + describe("OpenCode Go thinking-suffix model lookup", () => { it("preserves Responses routing for gpt-5.6-luna thinking variants", () => { expect(getModelSupportedFormats("opencode-go", "gpt-5.6-luna(high)")).toEqual(["openai-responses"]); @@ -108,7 +133,7 @@ describe("OpenCode Go per-model transport guard (chatCore logic)", () => { }); it("routes Muse Spark (responses-only) to /responses, never to /messages", () => { - for (const m of ["muse-spark-1.2-contributor", "muse-spark-1.3-contributor", "grok-4.6", "gpt-5.6-luna"]) { + for (const m of ["muse-spark-1.2-contributor", "muse-spark-1.3-contributor", "grok-4.7", "grok-4.6", "grok-4.5", "gpt-5.6-luna", "gpt-6-luna"]) { expect(getModelSupportedFormats("opencode-go", m)).toEqual(["openai-responses"]); expect(pickTransport("opencode-go", "openai-responses", "opencode-go", m)?.baseUrl).toBe("https://opencode.ai/zen/go/v1/responses"); expect(pickTransport("opencode-go", "claude", "opencode-go", m)).toBeNull(); diff --git a/tests/unit/provider-priority-insert-cost.test.js b/tests/unit/provider-priority-insert-cost.test.js new file mode 100644 index 00000000..bd2f2ef6 --- /dev/null +++ b/tests/unit/provider-priority-insert-cost.test.js @@ -0,0 +1,138 @@ +import { describe, expect, it } from "vitest"; + +import { + createProviderConnection, + getProviderConnections, + deleteProviderConnection, + updateProviderConnection, +} from "../../src/lib/db/index.js"; + +// #4311: POST /api/providers was O(pool) per insert. Inside one transaction it +// read the whole pool AND renumbered every row's priority, so a 5k-key import +// was O(n*m) — ~25M statements at a 5k pool — and every parallel writer +// serialized on the same transaction. On top of that, an apikey name collision +// silently overwrote the stored key with no 409. +// +// The test DB persists across tests in a file, so each case uses its own +// provider alias; priorities are per-provider. + +async function seed(provider, n) { + for (let i = 0; i < n; i++) { + await createProviderConnection({ + provider, + authType: "apikey", + name: `seed-${i}`, + apiKey: `k${i}`, + }); + } +} + +describe("provider insert is O(1) in pool size (#4311)", () => { + it("assigns sequential priorities without a renumber pass", async () => { + const P = `openai-compatible-seq-${Date.now()}`; + await seed(P, 3); + const list = await getProviderConnections({ provider: P }); + expect(list.map((c) => c.name)).toEqual(["seed-0", "seed-1", "seed-2"]); + expect(list.map((c) => c.priority)).toEqual([1, 2, 3]); + }); + + it("keeps a large pool in insertion order", async () => { + const P = `openai-compatible-ord-${Date.now()}`; + await seed(P, 60); + const list = await getProviderConnections({ provider: P }); + expect(list).toHaveLength(60); + // The bug showed up as reordering once the pool grew past a few rows. + expect(list[0].name).toBe("seed-0"); + expect(list[59].name).toBe("seed-59"); + for (let i = 1; i < list.length; i++) { + expect(list[i].priority).toBeGreaterThan(list[i - 1].priority); + } + }); + + it("still renumbers on delete, so gaps do not accumulate", async () => { + const P = `openai-compatible-del-${Date.now()}`; + await seed(P, 4); + const before = await getProviderConnections({ provider: P }); + await deleteProviderConnection(before[0].id); + const after = await getProviderConnections({ provider: P }); + expect(after.map((c) => c.priority)).toEqual([1, 2, 3]); + }); + + it("still renumbers on an explicit priority update", async () => { + // Unique alias per run: the DB persists across runs, so a fixed alias + // would accumulate rows and make this assertion depend on test order. + const P = `openai-compatible-upd-${Date.now()}`; + await seed(P, 4); + await new Promise((r) => setTimeout(r, 10)); + const list = await getProviderConnections({ provider: P }); + // Move the last one to the front. + await updateProviderConnection(list[3].id, { priority: 1 }); + const after = await getProviderConnections({ provider: P }); + expect(after[0].name).toBe("seed-3"); + }); +}); + +describe("name collision no longer destroys a key silently (#4311)", () => { + // Seeded once: these cases each mutate the SAME row, so a per-test seed + // would make the later assertions depend on earlier ones. + const P = `openai-compatible-clash-${Date.now()}`; + const original = (async () => { + await seed(P, 1); + return (await getProviderConnections({ provider: P }))[0]; + })(); + + it("throws a typed conflict instead of overwriting, when overwrite is refused", async () => { + const orig = await original; + await expect( + createProviderConnection({ + provider: P, + authType: "apikey", + name: orig.name, + apiKey: "REPLACEMENT-KEY", + allowOverwrite: false, + }) + ).rejects.toMatchObject({ code: "PROVIDER_NAME_CONFLICT", existingId: orig.id }); + + // The stored key must be untouched. + const after = (await getProviderConnections({ provider: P }))[0]; + expect(after.apiKey).toBe(orig.apiKey); + }); + + it("still overwrites when the caller opts in", async () => { + const orig = await original; + const updated = await createProviderConnection({ + provider: P, + authType: "apikey", + name: orig.name, + apiKey: "REPLACEMENT-KEY", + allowOverwrite: true, + }); + expect(updated.id).toBe(orig.id); + const after = (await getProviderConnections({ provider: P }))[0]; + expect(after.apiKey).toBe("REPLACEMENT-KEY"); + }); + + it("defaults to the previous overwrite behaviour for existing callers", async () => { + // Every other call site in the repo (oauth routes, bulk import) omits the + // flag, so they must keep working exactly as before. + const orig = await original; + const updated = await createProviderConnection({ + provider: P, + authType: "apikey", + name: orig.name, + apiKey: "LEGACY-PATH-KEY", + }); + expect(updated.id).toBe(orig.id); + }); + + it("does not collide across different providers", async () => { + const orig = await original; + const other = await createProviderConnection({ + provider: "openai-compatible-other", + authType: "apikey", + name: orig.name, + apiKey: "other-key", + }); + expect(other.id).not.toBe(orig.id); + }); +}); diff --git a/tests/unit/responses-completed-output.test.js b/tests/unit/responses-completed-output.test.js new file mode 100644 index 00000000..f72c4603 --- /dev/null +++ b/tests/unit/responses-completed-output.test.js @@ -0,0 +1,151 @@ +import { describe, expect, it } from "vitest"; + +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { initState } from "../../open-sse/translator/index.js"; +import { openaiToOpenAIResponsesResponse } from "../../open-sse/translator/response/openai-responses.js"; + +// targetFormat === OPENAI is the direct openai -> openai-responses route, which is +// the only one where flush() reaches this translator (see the flushReachesUs note +// above the finish_reason branch). +function newState() { + return { ...initState(FORMATS.OPENAI_RESPONSES), targetFormat: FORMATS.OPENAI }; +} + +function textChunk(text, index = 0) { + return { id: "chatcmpl-1", choices: [{ index, delta: { content: text } }] }; +} + +function reasoningChunk(text, index = 0) { + return { id: "chatcmpl-1", choices: [{ index, delta: { reasoning_content: text } }] }; +} + +function finishChunk(usage) { + return { id: "chatcmpl-1", choices: [{ index: 0, delta: {}, finish_reason: "stop" }], usage }; +} + +function runChunks(chunks) { + const state = newState(); + const events = []; + for (const chunk of chunks) { + for (const event of openaiToOpenAIResponsesResponse(chunk, state)) events.push(event); + } + return { state, events }; +} + +function completedResponse(events) { + const completed = events.find((event) => event.event === "response.completed"); + expect(completed, "expected a response.completed event").toBeTruthy(); + return completed.data.response; +} + +function doneItems(events) { + return events + .filter((event) => event.event === "response.output_item.done") + .map((event) => event.data.item); +} + +describe("response.completed output (issue #4307)", () => { + // The regression: sendCompleted() built the response object without an `output` + // key at all, so response.completed arrived with no output even though the + // message had already been streamed. Clients that build the final result from + // the terminal event (GitHub Copilot CLI 1.0.89 with a BYOK provider) printed + // the text and then failed with "No response was returned". + it("repeats the streamed message in response.completed", () => { + const state = newState(); + openaiToOpenAIResponsesResponse(textChunk("O"), state); + openaiToOpenAIResponsesResponse(textChunk("K"), state); + const response = completedResponse(openaiToOpenAIResponsesResponse(null, state)); + + expect(response.status).toBe("completed"); + expect(Array.isArray(response.output)).toBe(true); + expect(response.output).toHaveLength(1); + expect(response.output[0]).toMatchObject({ type: "message", role: "assistant" }); + expect(response.output[0].content[0]).toMatchObject({ type: "output_text", text: "OK" }); + }); + + it("matches exactly the items already delivered in response.output_item.done", () => { + const { events } = runChunks([ + textChunk("hello"), + finishChunk({ prompt_tokens: 7, completion_tokens: 2, total_tokens: 9 }), + ]); + const response = completedResponse(events); + const streamed = doneItems(events); + + expect(streamed).toHaveLength(1); + expect(response.output).toEqual(streamed); + }); + + it("includes a function_call item", () => { + const { events } = runChunks([ + { + id: "chatcmpl-1", + choices: [ + { + index: 0, + delta: { + tool_calls: [ + { index: 0, id: "call_1", function: { name: "get_weather", arguments: '{"city":"Paris"}' } }, + ], + }, + }, + ], + }, + finishChunk({ prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }), + ]); + const response = completedResponse(events); + + expect(response.output).toHaveLength(1); + expect(response.output[0]).toMatchObject({ + type: "function_call", + name: "get_weather", + arguments: '{"city":"Paris"}', + call_id: "call_1", + }); + }); + + it("orders output by output_index", () => { + const { events } = runChunks([ + reasoningChunk("thinking", 0), + textChunk("answer", 1), + finishChunk({ prompt_tokens: 4, completion_tokens: 3, total_tokens: 7 }), + ]); + const response = completedResponse(events); + + expect(response.output.map((item) => item.type)).toEqual(["reasoning", "message"]); + expect(response.output[1].content[0]).toMatchObject({ type: "output_text", text: "answer" }); + }); + + it("reports an empty output array when nothing was produced", () => { + const state = newState(); + const response = completedResponse(openaiToOpenAIResponsesResponse(null, state)); + expect(response.output).toEqual([]); + }); + + it("keeps the usage block alongside output", () => { + const { events } = runChunks([ + textChunk("OK"), + finishChunk({ prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 }), + ]); + const response = completedResponse(events); + + expect(response.usage).toMatchObject({ input_tokens: 3, output_tokens: 1, total_tokens: 4 }); + expect(response.output).toHaveLength(1); + }); + + it("leaves the in-progress response.created output empty", () => { + const { events } = runChunks([textChunk("hi")]); + const created = events.find((event) => event.event === "response.created"); + expect(created.data.response.status).toBe("in_progress"); + expect(created.data.response.output).toEqual([]); + }); + + it("does not duplicate items when flush runs more than once", () => { + const state = newState(); + openaiToOpenAIResponsesResponse(textChunk("once"), state); + openaiToOpenAIResponsesResponse(null, state); + const second = openaiToOpenAIResponsesResponse(null, state); + + expect(second).toEqual([]); + expect(state.completedOutputItems.size).toBe(1); + }); +}); \ No newline at end of file diff --git a/tests/unit/tokenharbor-provider.test.js b/tests/unit/tokenharbor-provider.test.js new file mode 100644 index 00000000..2f4af4fa --- /dev/null +++ b/tests/unit/tokenharbor-provider.test.js @@ -0,0 +1,79 @@ +import { describe, expect, it } from "vitest"; + +import REGISTRY from "../../open-sse/providers/registry/index.js"; +import { PROVIDERS, PROVIDER_MODELS } from "../../open-sse/providers/index.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; +import { DefaultExecutor } from "../../open-sse/executors/default.js"; + +describe("Token Harbor provider", () => { + const entry = REGISTRY.find((e) => e.id === "tokenharbor"); + + it("is registered as an OpenAI-compatible apikey provider", () => { + expect(entry).toBeDefined(); + expect(entry.category).toBe("apikey"); + expect(entry.authType).toBe("apikey"); + expect(entry.alias).toBe("tokenharbor"); + expect(entry.aliases).toContain("th"); + }); + + it("points at the verified OpenAI-compatible base URL", () => { + expect(PROVIDERS.tokenharbor.baseUrl).toBe("https://tokenharbor.ai/v1/chat/completions"); + expect(PROVIDERS.tokenharbor.validateUrl).toBe("https://tokenharbor.ai/v1/models"); + // transport.format defaults to "openai" via the shared provider default + expect(PROVIDERS.tokenharbor.format).toBe("openai"); + }); + + it("declares no provider-wide thinkingFormat so each model resolves its own", () => { + // Token Harbor forwards bodies verbatim. A provider-wide thinkingFormat + // would override capabilities.js and force one wire format (e.g. + // claude-adaptive) onto every model, which an OpenAI endpoint rejects. + expect(PROVIDERS.tokenharbor.thinkingFormat).toBeUndefined(); + }); + + it("enables dynamic model discovery and passthrough", () => { + expect(entry.passthroughModels).toBe(true); + expect(entry.modelsFetcher).toMatchObject({ + url: "https://tokenharbor.ai/v1/models", + type: "openai", + }); + }); + + it("exposes a small seed of bare (unprefixed) model ids", () => { + const ids = (PROVIDER_MODELS.tokenharbor || []).map((m) => m.id); + expect(ids.length).toBeGreaterThan(0); + expect(ids).toContain("claude-opus-5.5"); + // Token Harbor does not prefix ids by upstream vendor + expect(ids.every((id) => !id.includes("/"))).toBe(true); + }); + + it("routes through the shared DefaultExecutor (no custom adapter)", () => { + expect(getExecutor("tokenharbor")).toBeInstanceOf(DefaultExecutor); + }); + + it("resolves per-model capabilities from the shared tables", () => { + // Bare ids must still reach the canonical family patterns. + expect(getCapabilitiesForModel("tokenharbor", "claude-opus-5.5")).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "claude-adaptive", + }); + expect(getCapabilitiesForModel("tokenharbor", "gpt-6-astra")).toMatchObject({ + reasoning: true, + thinkingFormat: "openai", + }); + }); + + it("does not invent capabilities for an uncatalogued model", () => { + // Vision/reasoning must not be blanket-granted across the provider. + const caps = getCapabilitiesForModel("tokenharbor", "some-unknown-model-x"); + expect(caps.vision).toBe(false); + expect(caps.reasoning).toBe(false); + expect(caps.thinkingFormat).toBeNull(); + }); + + it("keeps every registry id unique after adding tokenharbor", () => { + const ids = REGISTRY.map((e) => e.id); + expect(new Set(ids).size).toBe(ids.length); + }); +}); diff --git a/tests/unit/usage-api-key-attribution.test.js b/tests/unit/usage-api-key-attribution.test.js new file mode 100644 index 00000000..049c2898 --- /dev/null +++ b/tests/unit/usage-api-key-attribution.test.js @@ -0,0 +1,57 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; + +let tempDir; +let db; + +beforeEach(async () => { + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "9router-api-key-")); + process.env.DATA_DIR = tempDir; + vi.resetModules(); + db = await import("@/lib/db/index.js"); + await db.initDb(); +}); + +afterEach(() => { + delete process.env.DATA_DIR; +}); + +describe("Usage stats API key attribution", () => { + it("keeps API keys with the same masked prefix in separate buckets", async () => { + const apiKeyA = "sk-machine-aaaaaa-11111111"; + const apiKeyB = "sk-machine-bbbbbb-22222222"; + + await db.saveRequestUsage({ + provider: "openai", + model: "gpt-4", + connectionId: "c1", + apiKey: apiKeyA, + tokens: { prompt_tokens: 10, completion_tokens: 5 }, + endpoint: "/v1/chat", + status: "ok", + }); + + await db.saveRequestUsage({ + provider: "openai", + model: "gpt-4", + connectionId: "c1", + apiKey: apiKeyB, + tokens: { prompt_tokens: 20, completion_tokens: 10 }, + endpoint: "/v1/chat", + status: "ok", + }); + + const stats = await db.getUsageStats("24h"); + const apiKeyEntries = Object.values(stats.byApiKey); + + expect(apiKeyEntries).toHaveLength(2); + + expect( + apiKeyEntries + .map((entry) => entry.promptTokens) + .sort((a, b) => a - b) + ).toEqual([10, 20]); + }); +}); diff --git a/tests/unit/zed-live-models.test.js b/tests/unit/zed-live-models.test.js index 1b999044..eaaea2fe 100644 --- a/tests/unit/zed-live-models.test.js +++ b/tests/unit/zed-live-models.test.js @@ -1,9 +1,10 @@ // Route-level acceptance for the Zed live-model wiring: // GET /api/providers/[connectionId]/models → resolveZedModels → UI rows -// RUN WITH AN ISOLATED DB: DATA_DIR=$(mktemp -d) npx vitest run ... -import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; -import { GET } from "@/app/api/providers/[id]/models/route.js"; -import { createProviderConnection } from "@/models/index.js"; +// Self-isolating: DATA_DIR points at a temp dir so seeding never touches ~/.9router. +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { describe, it, expect, beforeAll, afterAll, beforeEach, afterEach, vi } from "vitest"; // Transport stub BELOW resolveZedModels: proxyAwareFetch captures the native // fetch at import time, so stubbing globalThis.fetch cannot intercept it. @@ -72,6 +73,24 @@ afterEach(() => { vi.restoreAllMocks(); }); +// Imports must be dynamic so DATA_DIR is set before the DB layer loads. +const originalDataDir = process.env.DATA_DIR; +let GET; +let createProviderConnection; + +beforeAll(async () => { + process.env.DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "9router-zed-live-")); + vi.resetModules(); + ({ GET } = await import("@/app/api/providers/[id]/models/route.js")); + ({ createProviderConnection } = await import("@/models/index.js")); +}); + +afterAll(() => { + fs.rmSync(process.env.DATA_DIR, { recursive: true, force: true }); + if (originalDataDir === undefined) delete process.env.DATA_DIR; + else process.env.DATA_DIR = originalDataDir; +}); + async function seedZed(n) { return createProviderConnection({ provider: "zed",