feat(web): add TinyFish search and fetch provider

This commit is contained in:
Emirhan committed 2026-09-28 12:41:05 +07:00
1 parent dcc6a2055e
commit 60e890d7f1
9 files changed
+203 -2

No files matched your search

+1
View File
@@ -1,6 +1,7 @@
# v0.5.91 (2026-09-26)
## Features
- **Web Search & Fetch**: add TinyFish Search and Fetch with one API-key connection, normalized results, and official provider icon
- **Providers**: add Token Harbor provider and four OpenAI-compatible aggregator providers (dahl, atria, agnes, bai)
- **Claude**: forward `x-claude-code-session-id` on OAuth requests; merge client `anthropic-beta` flags and forward rate-limit headers; return thinking text to OpenAI-format clients
- **Codex**: add GPT-6 Sol and Luna support
+31 -1
View File
@@ -1,4 +1,4 @@
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama, tinyfish
// Returns normalized shape across all providers
const DEFAULT_TIMEOUT_MS = 15000;
@@ -129,6 +129,9 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
baseUrl: providerConfig?.baseUrl,
});
}
if (provider === "tinyfish") {
return await runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl: providerConfig?.baseUrl });
}
return { success: false, status: 400, error: `Unsupported provider: ${provider}` };
} catch (err) {
log?.("fetch handler error:", err?.message || err);
@@ -136,6 +139,33 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
}
}
async function runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl }) {
if (!["markdown", "html"].includes(fmt)) return { success: false, status: 400, error: `Unsupported TinyFish format: ${fmt}` };
const upstreamStart = Date.now();
const r = await tryFetch(baseUrl, {
method: "POST",
headers: { "content-type": "application/json", "X-API-Key": apiKey },
body: JSON.stringify({ urls: [url], format: fmt }),
}, timeoutMs);
if (!r.ok) return { success: false, status: r.timeout ? 504 : 502, error: r.error };
const upstreamMs = Date.now() - upstreamStart;
const { json } = await readJsonOrText(r.res);
if (!r.res.ok) return { success: false, status: r.res.status, error: json?.error?.message || `TinyFish error: ${r.res.status}` };
const failure = json?.errors?.[0];
if (failure) return { success: false, status: failure.status || 502, error: `TinyFish fetch failed: ${failure.error || "unknown error"}` };
const page = json?.results?.[0];
if (!page || typeof page.text !== "string") return { success: false, status: 502, error: "TinyFish returned no extractable content" };
const text = truncate(page.text, maxCharacters);
return {
success: true,
data: {
...buildData({ provider: "tinyfish", url, title: page.title, format: fmt, text, links: page.links,
costUsd: costPerQuery, responseMs: Date.now() - startedAt, upstreamMs }),
metadata: { author: page.author || null, published_at: page.published_date || null, language: page.language || null },
},
};
}
async function runFirecrawl({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt }) {
const upstreamStart = Date.now();
const r = await tryFetch("https://api.firecrawl.dev/v1/scrape", {
+24
View File
@@ -374,6 +374,29 @@ function buildXquikRequest(config, params) {
};
}
function buildTinyfishRequest(config, params) {
if (params.searchType && !["web", "news", "research_paper"].includes(params.searchType)) {
throw new Error("Unsupported TinyFish search type");
}
const qp = new URLSearchParams({ query: params.query });
if (params.searchType && params.searchType !== "web") qp.set("domain_type", params.searchType);
if (params.country) qp.set("location", params.country);
if (params.language) qp.set("language", params.language);
const { includes, excludes } = parseDomainFilter(params.domainFilter);
if (includes.length) qp.set("include_domains", includes.join(","));
if (excludes.length) qp.set("exclude_domains", excludes.join(","));
if (Number.isInteger(params.offset) && params.offset > 0) {
if (params.offset >= 110) throw new Error("TinyFish search offset exceeds available pages");
if (params.offset % 10 + params.maxResults > 10) throw new Error("TinyFish search offset and max_results must fit within one page");
qp.set("page", String(Math.floor(params.offset / 10)));
}
return {
// Keep API-key endpoint fixed; client baseUrl overrides must not receive the key.
url: `${config.baseUrl}?${qp}`,
init: { method: "GET", headers: { Accept: "application/json", "X-API-Key": params.token } },
};
}
// ── Ollama Cloud web_search ──────────────────────────────────────────────
// POST https://ollama.com/api/web_search { query, max_results }
// Response: { results: [{ title, url, content, published_at? }] }
@@ -436,6 +459,7 @@ const BUILDERS = {
"youcom": buildYouComRequest,
"searxng": buildSearxngRequest,
"xquik": buildXquikRequest,
"tinyfish": buildTinyfishRequest,
"ollama-search": buildOllamaSearchRequest,
"glm": buildGlmSearchRequest,
};
+4 -1
View File
@@ -110,7 +110,10 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
}
const data = await resp.json();
const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType);
const results = normalized.results.slice(0, params.maxResults);
// TinyFish uses fixed 10-result pages; offset within a page is applied locally.
const pageOffset = provider.id === "tinyfish" && Number.isInteger(params.offset) && params.offset > 0
? params.offset % 10 : 0;
const results = normalized.results.slice(pageOffset, pageOffset + params.maxResults);
const duration = Date.now() - startTime;
const usage = {
queries_used: 1,
+14
View File
@@ -107,6 +107,19 @@ function normalizeTavily(data, _query, _searchType) {
return { results, totalResults: results.length };
}
function normalizeTinyfish(data) {
const now = new Date().toISOString();
const items = Array.isArray(data?.results) ? data.results : [];
return {
results: items.map((item, idx) => makeResult("tinyfish", {
title: item.title, url: item.url, snippet: item.snippet,
published_at: item.date, source_type: item.publisher || null,
author: Array.isArray(item.authors) ? item.authors.join(", ") : null,
}, idx, now)),
totalResults: data?.total_results ?? null,
};
}
function normalizeGooglePse(data, _query, _searchType) {
const now = new Date().toISOString();
const items = Array.isArray(data.items) ? data.items : [];
@@ -294,6 +307,7 @@ const NORMALIZERS = {
"youcom": normalizeYouCom,
"searxng": normalizeSearxng,
"xquik": normalizeXquik,
"tinyfish": normalizeTinyfish,
"ollama-search": normalizeOllamaSearch,
"glm": normalizeGlmSearch,
};
+2
View File
@@ -130,6 +130,7 @@ import p126 from "./dahl.js";
import p127 from "./atria.js";
import p129 from "./agnes.js";
import p130 from "./bai.js";
import p131 from "./tinyfish.js";
export default [
p0,
p1,
@@ -260,4 +261,5 @@ export default [
p127,
p129,
p130,
p131,
];
+36
View File
@@ -0,0 +1,36 @@
export default {
id: "tinyfish",
alias: "tinyfish",
display: {
name: "TinyFish",
color: "#FF6700",
textIcon: "TF",
website: "https://www.tinyfish.ai/",
notice: { apiKeyUrl: "https://agent.tinyfish.ai/api-keys" },
},
category: "apikey",
authType: "apikey",
serviceKinds: ["webSearch", "webFetch"],
searchConfig: {
baseUrl: "https://api.search.tinyfish.ai",
validateUrl: "https://api.search.tinyfish.ai/usage?limit=1",
method: "GET",
authType: "apikey",
authHeader: "x-api-key",
costPerQuery: 0,
searchTypes: ["web", "news", "research_paper"],
defaultMaxResults: 5,
maxMaxResults: 10,
timeoutMs: 10000,
},
fetchConfig: {
baseUrl: "https://api.fetch.tinyfish.ai",
method: "POST",
authType: "apikey",
authHeader: "x-api-key",
costPerQuery: 0,
formats: ["markdown", "html"],
maxCharacters: 100000,
timeoutMs: 150000,
},
};
Binary file not shown.

After

Width:  |  Height:  |  Size: 7.3 KiB

+91
View File
@@ -0,0 +1,91 @@
import { afterEach, describe, expect, it, vi } from "vitest";
import REGISTRY from "../../open-sse/providers/registry/index.js";
import { buildSearchRequest } from "../../open-sse/handlers/search/callers.js";
import { normalizeSearchResponse } from "../../open-sse/handlers/search/normalizers.js";
import { handleSearchCore } from "../../open-sse/handlers/search/index.js";
import { handleFetchCore } from "../../open-sse/handlers/fetch/index.js";
import { getProvidersByKind } from "@/shared/constants/providers.js";
const provider = REGISTRY.find((entry) => entry.id === "tinyfish");
const url = "https://example.com/article";
afterEach(() => vi.unstubAllGlobals());
describe("TinyFish Search and Fetch", () => {
it("registers both kinds on one API-key provider", () => {
expect(provider.serviceKinds).toEqual(["webSearch", "webFetch"]);
expect(getProvidersByKind("webSearch").some((p) => p.id === "tinyfish")).toBe(true);
expect(getProvidersByKind("webFetch").some((p) => p.id === "tinyfish")).toBe(true);
expect(provider.searchConfig.validateUrl).toContain("/usage?limit=1");
});
it("builds GET search with header auth and maps results", () => {
const request = buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, {
query: "latest release", searchType: "news", token: "secret", maxResults: 5,
country: "TR", language: "tr", domainFilter: ["example.com", "-other.com"], offset: 5,
});
const parsed = new URL(request.url);
expect(Object.fromEntries(parsed.searchParams)).toEqual({
query: "latest release", domain_type: "news", location: "TR", language: "tr",
include_domains: "example.com", exclude_domains: "other.com", page: "0",
});
expect(request.url).not.toContain("secret");
expect(request.init.headers["X-API-Key"]).toBe("secret");
expect(buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, {
query: "release", searchType: "web", token: "secret", maxResults: 5,
providerOptions: { baseUrl: "https://example.org/collect" },
}).url.startsWith(provider.searchConfig.baseUrl)).toBe(true);
const normalized = normalizeSearchResponse("tinyfish", {
total_results: 1, results: [{ title: "Article", url, snippet: "Preview", date: "2026-09-01" }],
}, "latest release", "news");
expect(normalized.totalResults).toBe(1);
expect(normalized.results[0]).toMatchObject({ title: "Article", url, snippet: "Preview", published_at: "2026-09-01" });
});
it("applies offset within TinyFish fixed-size result pages", async () => {
const entries = Array.from({ length: 10 }, (_, i) => ({ title: `Article ${i}`, url: `${url}/${i}`, snippet: "Preview" }));
vi.stubGlobal("fetch", vi.fn(async () => new Response(JSON.stringify({ results: entries, total_results: 10 }), {
headers: { "content-type": "application/json" },
})));
const result = await handleSearchCore({
body: { query: "release", max_results: 2, offset: 3 },
provider: { id: "tinyfish" }, providerConfig: provider.searchConfig,
credentials: { apiKey: "secret" },
});
expect(result.success).toBe(true);
expect(result.data.results.map((entry) => entry.title)).toEqual(["Article 3", "Article 4"]);
expect(new URL(vi.mocked(fetch).mock.calls[0][0]).searchParams.get("page")).toBe("0");
});
it("rejects offsets that would silently omit requested results", () => {
const params = { query: "release", searchType: "web", token: "secret", maxResults: 5, offset: 8 };
expect(() => buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, params))
.toThrow("TinyFish search offset and max_results must fit within one page");
});
it("maps fetched content and treats HTTP 200 per-URL errors as failures", async () => {
const fetchMock = vi.fn(async () => new Response(JSON.stringify({
results: [{ url, title: "Article", text: "# Article", language: "en" }], errors: [],
}), { headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const input = { url, provider: "tinyfish", providerConfig: provider.fetchConfig, credentials: { apiKey: "secret" } };
const result = await handleFetchCore(input);
expect(result.data).toMatchObject({ provider: "tinyfish", title: "Article", content: { text: "# Article" }, metadata: { language: "en" } });
expect(fetchMock.mock.calls[0][0]).toBe(provider.fetchConfig.baseUrl);
expect(fetchMock.mock.calls[0][1].headers["X-API-Key"]).toBe("secret");
expect(JSON.parse(fetchMock.mock.calls[0][1].body)).toEqual({ urls: [url], format: "markdown" });
fetchMock.mockResolvedValueOnce(new Response(JSON.stringify({ results: [], errors: [{ url, error: "bot_blocked" }] }), {
headers: { "content-type": "application/json" },
}));
expect(await handleFetchCore(input)).toMatchObject({ success: false, status: 502, error: "TinyFish fetch failed: bot_blocked" });
expect(await handleFetchCore({ ...input, format: "text" })).toMatchObject({ success: false, status: 400 });
expect(fetchMock).toHaveBeenCalledTimes(2);
fetchMock.mockResolvedValueOnce(new Response(JSON.stringify({ error: { message: "Invalid API key" } }), {
status: 401, headers: { "content-type": "application/json" },
}));
expect(await handleFetchCore(input)).toMatchObject({ success: false, status: 401 });
});
});