diff --git a/CHANGELOG.md b/CHANGELOG.md index 6cb89240..b6d8cd61 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,7 @@ # v0.5.91 (2026-09-26) ## Features +- **Web Search & Fetch**: add TinyFish Search and Fetch with one API-key connection, normalized results, and official provider icon - **Providers**: add Token Harbor provider and four OpenAI-compatible aggregator providers (dahl, atria, agnes, bai) - **Claude**: forward `x-claude-code-session-id` on OAuth requests; merge client `anthropic-beta` flags and forward rate-limit headers; return thinking text to OpenAI-format clients - **Codex**: add GPT-6 Sol and Luna support diff --git a/open-sse/handlers/fetch/index.js b/open-sse/handlers/fetch/index.js index 9bad7e18..364caa5d 100644 --- a/open-sse/handlers/fetch/index.js +++ b/open-sse/handlers/fetch/index.js @@ -1,4 +1,4 @@ -// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama +// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama, tinyfish // Returns normalized shape across all providers const DEFAULT_TIMEOUT_MS = 15000; @@ -129,6 +129,9 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr baseUrl: providerConfig?.baseUrl, }); } + if (provider === "tinyfish") { + return await runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl: providerConfig?.baseUrl }); + } return { success: false, status: 400, error: `Unsupported provider: ${provider}` }; } catch (err) { log?.("fetch handler error:", err?.message || err); @@ -136,6 +139,33 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr } } +async function runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl }) { + if (!["markdown", "html"].includes(fmt)) return { success: false, status: 400, error: `Unsupported TinyFish format: ${fmt}` }; + const upstreamStart = Date.now(); + const r = await tryFetch(baseUrl, { + method: "POST", + headers: { "content-type": "application/json", "X-API-Key": apiKey }, + body: JSON.stringify({ urls: [url], format: fmt }), + }, timeoutMs); + if (!r.ok) return { success: false, status: r.timeout ? 504 : 502, error: r.error }; + const upstreamMs = Date.now() - upstreamStart; + const { json } = await readJsonOrText(r.res); + if (!r.res.ok) return { success: false, status: r.res.status, error: json?.error?.message || `TinyFish error: ${r.res.status}` }; + const failure = json?.errors?.[0]; + if (failure) return { success: false, status: failure.status || 502, error: `TinyFish fetch failed: ${failure.error || "unknown error"}` }; + const page = json?.results?.[0]; + if (!page || typeof page.text !== "string") return { success: false, status: 502, error: "TinyFish returned no extractable content" }; + const text = truncate(page.text, maxCharacters); + return { + success: true, + data: { + ...buildData({ provider: "tinyfish", url, title: page.title, format: fmt, text, links: page.links, + costUsd: costPerQuery, responseMs: Date.now() - startedAt, upstreamMs }), + metadata: { author: page.author || null, published_at: page.published_date || null, language: page.language || null }, + }, + }; +} + async function runFirecrawl({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt }) { const upstreamStart = Date.now(); const r = await tryFetch("https://api.firecrawl.dev/v1/scrape", { diff --git a/open-sse/handlers/search/callers.js b/open-sse/handlers/search/callers.js index 5e3b3c09..56d08e4f 100644 --- a/open-sse/handlers/search/callers.js +++ b/open-sse/handlers/search/callers.js @@ -374,6 +374,29 @@ function buildXquikRequest(config, params) { }; } +function buildTinyfishRequest(config, params) { + if (params.searchType && !["web", "news", "research_paper"].includes(params.searchType)) { + throw new Error("Unsupported TinyFish search type"); + } + const qp = new URLSearchParams({ query: params.query }); + if (params.searchType && params.searchType !== "web") qp.set("domain_type", params.searchType); + if (params.country) qp.set("location", params.country); + if (params.language) qp.set("language", params.language); + const { includes, excludes } = parseDomainFilter(params.domainFilter); + if (includes.length) qp.set("include_domains", includes.join(",")); + if (excludes.length) qp.set("exclude_domains", excludes.join(",")); + if (Number.isInteger(params.offset) && params.offset > 0) { + if (params.offset >= 110) throw new Error("TinyFish search offset exceeds available pages"); + if (params.offset % 10 + params.maxResults > 10) throw new Error("TinyFish search offset and max_results must fit within one page"); + qp.set("page", String(Math.floor(params.offset / 10))); + } + return { + // Keep API-key endpoint fixed; client baseUrl overrides must not receive the key. + url: `${config.baseUrl}?${qp}`, + init: { method: "GET", headers: { Accept: "application/json", "X-API-Key": params.token } }, + }; +} + // ── Ollama Cloud web_search ────────────────────────────────────────────── // POST https://ollama.com/api/web_search { query, max_results } // Response: { results: [{ title, url, content, published_at? }] } @@ -436,6 +459,7 @@ const BUILDERS = { "youcom": buildYouComRequest, "searxng": buildSearxngRequest, "xquik": buildXquikRequest, + "tinyfish": buildTinyfishRequest, "ollama-search": buildOllamaSearchRequest, "glm": buildGlmSearchRequest, }; diff --git a/open-sse/handlers/search/index.js b/open-sse/handlers/search/index.js index 70c3c749..fa6d146d 100644 --- a/open-sse/handlers/search/index.js +++ b/open-sse/handlers/search/index.js @@ -110,7 +110,10 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential } const data = await resp.json(); const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType); - const results = normalized.results.slice(0, params.maxResults); + // TinyFish uses fixed 10-result pages; offset within a page is applied locally. + const pageOffset = provider.id === "tinyfish" && Number.isInteger(params.offset) && params.offset > 0 + ? params.offset % 10 : 0; + const results = normalized.results.slice(pageOffset, pageOffset + params.maxResults); const duration = Date.now() - startTime; const usage = { queries_used: 1, diff --git a/open-sse/handlers/search/normalizers.js b/open-sse/handlers/search/normalizers.js index 3b415ae5..5ef7f36f 100644 --- a/open-sse/handlers/search/normalizers.js +++ b/open-sse/handlers/search/normalizers.js @@ -107,6 +107,19 @@ function normalizeTavily(data, _query, _searchType) { return { results, totalResults: results.length }; } +function normalizeTinyfish(data) { + const now = new Date().toISOString(); + const items = Array.isArray(data?.results) ? data.results : []; + return { + results: items.map((item, idx) => makeResult("tinyfish", { + title: item.title, url: item.url, snippet: item.snippet, + published_at: item.date, source_type: item.publisher || null, + author: Array.isArray(item.authors) ? item.authors.join(", ") : null, + }, idx, now)), + totalResults: data?.total_results ?? null, + }; +} + function normalizeGooglePse(data, _query, _searchType) { const now = new Date().toISOString(); const items = Array.isArray(data.items) ? data.items : []; @@ -294,6 +307,7 @@ const NORMALIZERS = { "youcom": normalizeYouCom, "searxng": normalizeSearxng, "xquik": normalizeXquik, + "tinyfish": normalizeTinyfish, "ollama-search": normalizeOllamaSearch, "glm": normalizeGlmSearch, }; diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 07b71d2c..6c9d1f41 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -130,6 +130,7 @@ import p126 from "./dahl.js"; import p127 from "./atria.js"; import p129 from "./agnes.js"; import p130 from "./bai.js"; +import p131 from "./tinyfish.js"; export default [ p0, p1, @@ -260,4 +261,5 @@ export default [ p127, p129, p130, + p131, ]; diff --git a/open-sse/providers/registry/tinyfish.js b/open-sse/providers/registry/tinyfish.js new file mode 100644 index 00000000..c6875251 --- /dev/null +++ b/open-sse/providers/registry/tinyfish.js @@ -0,0 +1,36 @@ +export default { + id: "tinyfish", + alias: "tinyfish", + display: { + name: "TinyFish", + color: "#FF6700", + textIcon: "TF", + website: "https://www.tinyfish.ai/", + notice: { apiKeyUrl: "https://agent.tinyfish.ai/api-keys" }, + }, + category: "apikey", + authType: "apikey", + serviceKinds: ["webSearch", "webFetch"], + searchConfig: { + baseUrl: "https://api.search.tinyfish.ai", + validateUrl: "https://api.search.tinyfish.ai/usage?limit=1", + method: "GET", + authType: "apikey", + authHeader: "x-api-key", + costPerQuery: 0, + searchTypes: ["web", "news", "research_paper"], + defaultMaxResults: 5, + maxMaxResults: 10, + timeoutMs: 10000, + }, + fetchConfig: { + baseUrl: "https://api.fetch.tinyfish.ai", + method: "POST", + authType: "apikey", + authHeader: "x-api-key", + costPerQuery: 0, + formats: ["markdown", "html"], + maxCharacters: 100000, + timeoutMs: 150000, + }, +}; diff --git a/public/providers/tinyfish.png b/public/providers/tinyfish.png new file mode 100644 index 00000000..9768f27a Binary files /dev/null and b/public/providers/tinyfish.png differ diff --git a/tests/unit/tinyfish-web-provider.test.js b/tests/unit/tinyfish-web-provider.test.js new file mode 100644 index 00000000..394c1f1f --- /dev/null +++ b/tests/unit/tinyfish-web-provider.test.js @@ -0,0 +1,91 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import REGISTRY from "../../open-sse/providers/registry/index.js"; +import { buildSearchRequest } from "../../open-sse/handlers/search/callers.js"; +import { normalizeSearchResponse } from "../../open-sse/handlers/search/normalizers.js"; +import { handleSearchCore } from "../../open-sse/handlers/search/index.js"; +import { handleFetchCore } from "../../open-sse/handlers/fetch/index.js"; +import { getProvidersByKind } from "@/shared/constants/providers.js"; + +const provider = REGISTRY.find((entry) => entry.id === "tinyfish"); +const url = "https://example.com/article"; + +afterEach(() => vi.unstubAllGlobals()); + +describe("TinyFish Search and Fetch", () => { + it("registers both kinds on one API-key provider", () => { + expect(provider.serviceKinds).toEqual(["webSearch", "webFetch"]); + expect(getProvidersByKind("webSearch").some((p) => p.id === "tinyfish")).toBe(true); + expect(getProvidersByKind("webFetch").some((p) => p.id === "tinyfish")).toBe(true); + expect(provider.searchConfig.validateUrl).toContain("/usage?limit=1"); + }); + + it("builds GET search with header auth and maps results", () => { + const request = buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, { + query: "latest release", searchType: "news", token: "secret", maxResults: 5, + country: "TR", language: "tr", domainFilter: ["example.com", "-other.com"], offset: 5, + }); + const parsed = new URL(request.url); + expect(Object.fromEntries(parsed.searchParams)).toEqual({ + query: "latest release", domain_type: "news", location: "TR", language: "tr", + include_domains: "example.com", exclude_domains: "other.com", page: "0", + }); + expect(request.url).not.toContain("secret"); + expect(request.init.headers["X-API-Key"]).toBe("secret"); + expect(buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, { + query: "release", searchType: "web", token: "secret", maxResults: 5, + providerOptions: { baseUrl: "https://example.org/collect" }, + }).url.startsWith(provider.searchConfig.baseUrl)).toBe(true); + const normalized = normalizeSearchResponse("tinyfish", { + total_results: 1, results: [{ title: "Article", url, snippet: "Preview", date: "2026-09-01" }], + }, "latest release", "news"); + expect(normalized.totalResults).toBe(1); + expect(normalized.results[0]).toMatchObject({ title: "Article", url, snippet: "Preview", published_at: "2026-09-01" }); + }); + + it("applies offset within TinyFish fixed-size result pages", async () => { + const entries = Array.from({ length: 10 }, (_, i) => ({ title: `Article ${i}`, url: `${url}/${i}`, snippet: "Preview" })); + vi.stubGlobal("fetch", vi.fn(async () => new Response(JSON.stringify({ results: entries, total_results: 10 }), { + headers: { "content-type": "application/json" }, + }))); + const result = await handleSearchCore({ + body: { query: "release", max_results: 2, offset: 3 }, + provider: { id: "tinyfish" }, providerConfig: provider.searchConfig, + credentials: { apiKey: "secret" }, + }); + expect(result.success).toBe(true); + expect(result.data.results.map((entry) => entry.title)).toEqual(["Article 3", "Article 4"]); + expect(new URL(vi.mocked(fetch).mock.calls[0][0]).searchParams.get("page")).toBe("0"); + }); + + it("rejects offsets that would silently omit requested results", () => { + const params = { query: "release", searchType: "web", token: "secret", maxResults: 5, offset: 8 }; + expect(() => buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, params)) + .toThrow("TinyFish search offset and max_results must fit within one page"); + }); + + it("maps fetched content and treats HTTP 200 per-URL errors as failures", async () => { + const fetchMock = vi.fn(async () => new Response(JSON.stringify({ + results: [{ url, title: "Article", text: "# Article", language: "en" }], errors: [], + }), { headers: { "content-type": "application/json" } })); + vi.stubGlobal("fetch", fetchMock); + const input = { url, provider: "tinyfish", providerConfig: provider.fetchConfig, credentials: { apiKey: "secret" } }; + const result = await handleFetchCore(input); + expect(result.data).toMatchObject({ provider: "tinyfish", title: "Article", content: { text: "# Article" }, metadata: { language: "en" } }); + expect(fetchMock.mock.calls[0][0]).toBe(provider.fetchConfig.baseUrl); + expect(fetchMock.mock.calls[0][1].headers["X-API-Key"]).toBe("secret"); + expect(JSON.parse(fetchMock.mock.calls[0][1].body)).toEqual({ urls: [url], format: "markdown" }); + + fetchMock.mockResolvedValueOnce(new Response(JSON.stringify({ results: [], errors: [{ url, error: "bot_blocked" }] }), { + headers: { "content-type": "application/json" }, + })); + expect(await handleFetchCore(input)).toMatchObject({ success: false, status: 502, error: "TinyFish fetch failed: bot_blocked" }); + + expect(await handleFetchCore({ ...input, format: "text" })).toMatchObject({ success: false, status: 400 }); + expect(fetchMock).toHaveBeenCalledTimes(2); + + fetchMock.mockResolvedValueOnce(new Response(JSON.stringify({ error: { message: "Invalid API key" } }), { + status: 401, headers: { "content-type": "application/json" }, + })); + expect(await handleFetchCore(input)).toMatchObject({ success: false, status: 401 }); + }); +});