pi-web-kit 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/http.ts ADDED
@@ -0,0 +1,56 @@
1
+ import { DEFAULT_TIMEOUT_MS } from "./limits.js";
2
+ import { normalizeUrlInput } from "./urls.js";
3
+
4
+ const MAX_CHARS = 50_000;
5
+ const MAX_LINES = 2_000;
6
+
7
+ export function truncateText(text: string, maxChars = MAX_CHARS): string {
8
+ const lines = text.split(/\r?\n/);
9
+ const lineLimited = lines.length > MAX_LINES ? lines.slice(0, MAX_LINES).join("\n") + "\n[truncated: line limit]" : text;
10
+ return lineLimited.length > maxChars ? lineLimited.slice(0, maxChars) + "\n[truncated: size limit]" : lineLimited;
11
+ }
12
+
13
+ export async function fetchWithTimeout(url: string, init: RequestInit & { timeoutMs?: number } = {}): Promise<Response> {
14
+ const controller = new AbortController();
15
+ const timeoutMs = init.timeoutMs ?? DEFAULT_TIMEOUT_MS;
16
+ const timer = setTimeout(() => controller.abort(new Error(`Request timed out after ${timeoutMs}ms`)), timeoutMs);
17
+ const signal = init.signal ? AbortSignal.any([init.signal, controller.signal]) : controller.signal;
18
+ const { timeoutMs: _timeoutMs, ...rest } = init;
19
+ try {
20
+ return await fetch(url, { ...rest, signal });
21
+ } finally {
22
+ clearTimeout(timer);
23
+ }
24
+ }
25
+
26
+ export async function requestJson<T>(url: string, init: RequestInit & { timeoutMs?: number } = {}): Promise<T> {
27
+ const res = await fetchWithTimeout(url, init);
28
+ const text = await res.text();
29
+ if (!res.ok) throw new Error(`${res.status} ${res.statusText}: ${text.slice(0, 1000)}`);
30
+ if (!text) return undefined as T;
31
+ return JSON.parse(text) as T;
32
+ }
33
+
34
+ export function asSnippet(value: unknown): string | undefined {
35
+ if (value == null) return undefined;
36
+ if (Array.isArray(value)) return value.filter(Boolean).join("\n").slice(0, 1000) || undefined;
37
+ if (typeof value === "string") return value.slice(0, 1000) || undefined;
38
+ return JSON.stringify(value).slice(0, 1000);
39
+ }
40
+
41
+ export function normalizeUrls(input: { url?: string; urls?: string[] }, maxCount?: number): string[] {
42
+ return normalizeUrlInput(input, maxCount);
43
+ }
44
+
45
+ export async function mapConcurrent<T, R>(items: T[], concurrency: number, mapper: (item: T, index: number) => Promise<R>): Promise<R[]> {
46
+ const results = new Array<R>(items.length);
47
+ let next = 0;
48
+ const workers = Array.from({ length: Math.min(concurrency, items.length) }, async () => {
49
+ while (next < items.length) {
50
+ const index = next++;
51
+ results[index] = await mapper(items[index], index);
52
+ }
53
+ });
54
+ await Promise.all(workers);
55
+ return results;
56
+ }
package/src/index.ts ADDED
@@ -0,0 +1,3 @@
1
+ export * from "./types.js";
2
+ export * from "./config.js";
3
+ export * from "./providers/index.js";
package/src/limits.ts ADDED
@@ -0,0 +1,14 @@
1
+ export const MAX_QUERY_COUNT = 5;
2
+ export const MAX_URL_COUNT = 10;
3
+ export const MAX_URL_LENGTH = 2_048;
4
+ export const MAX_NUM_RESULTS = 20;
5
+ export const DEFAULT_NUM_RESULTS = 10;
6
+ export const MAX_OFFSET = 10_000_000;
7
+ export const MAX_LIMIT = 100_000;
8
+ export const DEFAULT_FETCH_LIMIT = 30_000;
9
+ export const MULTI_FETCH_LIMIT = 8_000;
10
+ export const DEFAULT_TIMEOUT_MS = 30_000;
11
+ export const FETCH_CACHE_MAX_ENTRIES = 100;
12
+ export const FETCH_CACHE_MAX_BYTES = 20 * 1024 * 1024;
13
+ export const FETCH_CACHE_TTL_MS = 30 * 60 * 1000;
14
+ export const FETCH_CONCURRENCY = 3;
@@ -0,0 +1,30 @@
1
+ import { asSnippet, requestJson } from "../http.js";
2
+ import type { SearchInput, SearchProvider, WebKitConfig } from "../types.js";
3
+ import { requireKey } from "../config.js";
4
+
5
+ export class BraveProvider implements SearchProvider {
6
+ private key: string;
7
+ constructor(config: WebKitConfig) { this.key = requireKey(config, "brave"); }
8
+
9
+ async search(input: SearchInput, signal?: AbortSignal) {
10
+ const url = new URL("https://api.search.brave.com/res/v1/llm/context");
11
+ url.searchParams.set("q", input.query);
12
+ if (input.numResults) {
13
+ const count = String(Math.max(1, Math.min(input.numResults, 20)));
14
+ url.searchParams.set("count", count);
15
+ url.searchParams.set("maximum_number_of_urls", String(input.maxUrls ?? count));
16
+ }
17
+ if (typeof input.country === "string") url.searchParams.set("country", input.country);
18
+ if (typeof input.searchLang === "string") url.searchParams.set("search_lang", input.searchLang);
19
+ if (typeof input.uiLang === "string") url.searchParams.set("ui_lang", input.uiLang);
20
+ if (typeof input.safesearch === "string") url.searchParams.set("safesearch", input.safesearch);
21
+ if (typeof input.freshness === "string") url.searchParams.set("freshness", input.freshness);
22
+ const data = await requestJson<any>(url.toString(), { headers: { "X-Subscription-Token": this.key, accept: "application/json" }, signal, timeoutMs: 30000 });
23
+ const sources = data.sources ?? data.web?.results ?? [];
24
+ const snippets = data.grounding?.generic ?? [];
25
+ const rows = sources.length ? sources : snippets;
26
+ return { provider: "brave" as const, query: input.query, results: rows.map((r: any, i: number) => ({
27
+ title: r.title ?? r.name, url: r.url, snippet: asSnippet(r.snippets ?? r.description ?? snippets[i]?.snippets), siteName: r.site_name ?? r.source, position: i + 1,
28
+ })).filter((r: any) => r.url) };
29
+ }
30
+ }
@@ -0,0 +1,91 @@
1
+ import { asSnippet, fetchWithTimeout, normalizeUrls } from "../http.js";
2
+ import { urlsMatch } from "../urls.js";
3
+ import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
4
+ import { applyExaFetchFallbacks } from "./fallback.js";
5
+
6
+ let nextId = 1;
7
+
8
+ export class ExaMcpProvider implements SearchProvider, FetchProvider {
9
+ private sessionId?: string;
10
+ constructor(private config: WebKitConfig) {}
11
+
12
+ async search(input: SearchInput, signal?: AbortSignal) {
13
+ const result = await this.callTool("web_search_exa", { query: input.query, numResults: input.numResults ?? 10 }, signal);
14
+ return { provider: "exa_mcp" as const, query: input.query, results: normalizeSearch(result, input.query) };
15
+ }
16
+
17
+ async fetch(input: FetchInput, signal?: AbortSignal) {
18
+ const urls = normalizeUrls(input);
19
+ const result = await this.callTool("web_fetch_exa", { urls }, signal);
20
+ const pages = normalizeFetch(result, urls);
21
+ return applyExaFetchFallbacks(this.config, input, urls, { provider: "exa_mcp" as const, results: pages }, signal);
22
+ }
23
+
24
+ private async callTool(name: string, args: Record<string, unknown>, signal?: AbortSignal): Promise<any> {
25
+ await this.initialize(signal);
26
+ return this.rpc("tools/call", { name, arguments: args }, signal);
27
+ }
28
+
29
+ private async initialize(signal?: AbortSignal) {
30
+ if (this.sessionId) return;
31
+ const data = await this.rpcRaw("initialize", { protocolVersion: "2025-06-18", capabilities: {}, clientInfo: { name: "pi-web-kit", version: "0.1.0" } }, signal);
32
+ this.sessionId = data.sessionId ?? data.payload?.sessionId;
33
+ if (!this.sessionId) throw new Error("Exa MCP initialize did not return an mcp-session-id response header.");
34
+ await this.rpcRaw("notifications/initialized", undefined, signal, true);
35
+ }
36
+
37
+ private async rpc(method: string, params: unknown, signal?: AbortSignal) {
38
+ const data = await this.rpcRaw(method, params, signal);
39
+ if (data.payload?.error) throw new Error(data.payload.error.message ?? JSON.stringify(data.payload.error));
40
+ return data.payload?.result ?? data.payload;
41
+ }
42
+
43
+ private async rpcRaw(method: string, params: unknown, signal?: AbortSignal, notification = false): Promise<{ payload?: any; sessionId?: string }> {
44
+ const headers: Record<string, string> = { "content-type": "application/json", accept: "application/json, text/event-stream" };
45
+ if (this.sessionId) headers["mcp-session-id"] = this.sessionId;
46
+ if (this.config.apiKeys.exa) headers["x-api-key"] = this.config.apiKeys.exa;
47
+ const body = notification ? { jsonrpc: "2.0", method, params } : { jsonrpc: "2.0", id: nextId++, method, params };
48
+ const res = await fetchWithTimeout("https://mcp.exa.ai/mcp", { method: "POST", headers, body: JSON.stringify(body), signal, timeoutMs: 45_000 });
49
+ const text = await res.text();
50
+ if (!res.ok) throw new Error(`Exa MCP ${method} failed: ${res.status} ${res.statusText}: ${text.slice(0, 1000)}`);
51
+ if (notification) return { sessionId: this.sessionId };
52
+ const sessionId = res.headers.get("mcp-session-id") ?? undefined;
53
+ const payload = parseMcpResponse(text, res.headers.get("content-type") ?? "");
54
+ return { payload, sessionId };
55
+ }
56
+ }
57
+
58
+ function parseMcpResponse(text: string, contentType: string): any {
59
+ if (!text.trim()) return undefined;
60
+ if (contentType.includes("text/event-stream") || text.startsWith("event:") || text.startsWith("data:")) {
61
+ const data = text.split(/\r?\n/).filter((line) => line.startsWith("data:")).map((line) => line.slice(5).trim()).join("\n");
62
+ return data ? JSON.parse(data) : undefined;
63
+ }
64
+ return JSON.parse(text);
65
+ }
66
+
67
+ function textFromContent(result: any): string {
68
+ const content = result?.content ?? result;
69
+ if (Array.isArray(content)) return content.map((c) => typeof c === "string" ? c : c.text ?? JSON.stringify(c)).join("\n");
70
+ return typeof content === "string" ? content : JSON.stringify(content);
71
+ }
72
+
73
+ function normalizeSearch(result: any, _query: string) {
74
+ const structured = result?.structuredContent ?? result?.result ?? result;
75
+ const list = structured.results ?? structured.data ?? structured.items;
76
+ if (Array.isArray(list)) return list.map((r: any, i: number) => ({ title: r.title, url: r.url, snippet: asSnippet(r.snippet ?? r.text ?? r.summary ?? r.highlights), siteName: r.siteName, position: i + 1 })).filter((r: any) => r.url);
77
+ const text = textFromContent(result);
78
+ const urls = [...text.matchAll(/https?:\/\/[^\s)\]}>"']+/g)].map((m) => m[0]);
79
+ return [...new Set(urls)].map((url, i) => ({ url, snippet: i === 0 ? text.slice(0, 1000) : undefined, position: i + 1 }));
80
+ }
81
+
82
+ function normalizeFetch(result: any, urls: string[]) {
83
+ const structured = result?.structuredContent ?? result?.result ?? result;
84
+ const list = structured.results ?? structured.data ?? structured.pages;
85
+ if (Array.isArray(list)) return urls.map((url, i) => {
86
+ const r = list.find((x: any) => urlsMatch(x.url, url)) ?? list[i];
87
+ return r ? { url, title: r.title, content: r.markdown ?? r.text ?? r.content ?? r.html, format: "markdown" as const, metadata: r, error: r.error } : { url, error: "No content returned by Exa MCP." };
88
+ });
89
+ const text = textFromContent(result);
90
+ return urls.length <= 1 ? [{ url: urls[0] ?? "", content: text, format: "markdown" as const }] : urls.map((url) => ({ url, content: text, format: "markdown" as const }));
91
+ }
@@ -0,0 +1,63 @@
1
+ import { asSnippet, normalizeUrls, requestJson } from "../http.js";
2
+ import { urlsMatch } from "../urls.js";
3
+ import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebFetchResult, WebKitConfig } from "../types.js";
4
+ import { requireKey } from "../config.js";
5
+ import { applyExaFetchFallbacks } from "./fallback.js";
6
+
7
+ export class ExaProvider implements SearchProvider, FetchProvider {
8
+ private key: string;
9
+ constructor(private config: WebKitConfig) { this.key = requireKey(config, "exa"); }
10
+
11
+ async search(input: SearchInput, signal?: AbortSignal) {
12
+ const body = {
13
+ query: input.query,
14
+ numResults: input.numResults ?? 10,
15
+ contents: input.contents ?? { highlights: true },
16
+ includeDomains: input.includeDomains,
17
+ excludeDomains: input.excludeDomains,
18
+ startPublishedDate: input.startPublishedDate,
19
+ endPublishedDate: input.endPublishedDate,
20
+ startCrawlDate: input.startCrawlDate,
21
+ endCrawlDate: input.endCrawlDate,
22
+ type: input.type,
23
+ category: input.category,
24
+ };
25
+ const data = await requestJson<any>("https://api.exa.ai/search", {
26
+ method: "POST",
27
+ headers: { "content-type": "application/json", "x-api-key": this.key },
28
+ body: JSON.stringify(body),
29
+ signal,
30
+ timeoutMs: 30000,
31
+ });
32
+ return {
33
+ provider: "exa" as const,
34
+ query: input.query,
35
+ results: (data.results ?? []).map((r: any, i: number) => ({
36
+ title: r.title,
37
+ url: r.url,
38
+ snippet: asSnippet(r.highlights ?? r.summary ?? r.text),
39
+ siteName: r.author ?? r.publishedDate,
40
+ position: i + 1,
41
+ })).filter((r: any) => r.url),
42
+ };
43
+ }
44
+
45
+ async fetch(input: FetchInput, signal?: AbortSignal): Promise<WebFetchResult> {
46
+ const urls = normalizeUrls(input);
47
+ if (urls.length === 0) return { provider: "exa", results: [] };
48
+ const data = await requestJson<any>("https://api.exa.ai/contents", {
49
+ method: "POST",
50
+ headers: { "content-type": "application/json", "x-api-key": this.key },
51
+ body: JSON.stringify({ urls, text: true, highlights: false }),
52
+ signal,
53
+ timeoutMs: 45000,
54
+ });
55
+ const list = data.results ?? [];
56
+ const primary: WebFetchResult = { provider: "exa", results: urls.map((url, i) => {
57
+ const r: any = list.find((item: any) => urlsMatch(item.url, url)) ?? list[i];
58
+ if (!r) return { url, error: "No content returned by Exa contents endpoint." };
59
+ return { url: r.url ?? url, title: r.title, content: r.text ?? r.summary ?? "", format: "markdown" as const, metadata: r };
60
+ }) };
61
+ return applyExaFetchFallbacks(this.config, input, urls, primary, signal);
62
+ }
63
+ }
@@ -0,0 +1,82 @@
1
+ import type { FetchInput, FetchProvider, FetchProviderName, WebFetchResult, WebKitConfig } from "../types.js";
2
+ import { urlsMatch } from "../urls.js";
3
+ import { FirecrawlProvider } from "./firecrawl.js";
4
+ import { MarkdownNewProvider } from "./markdown-new.js";
5
+ import { TinyFishProvider } from "./tinyfish.js";
6
+
7
+ type FetchPage = WebFetchResult["results"][number];
8
+
9
+ export function isMissingExaCrawlResult(page: FetchPage | undefined): boolean {
10
+ if (!page) return true;
11
+ const message = [page.error, page.content].filter((v): v is string => typeof v === "string").join("\n");
12
+ if (!message.trim()) return true;
13
+ return /\b(no crawl results?|not crawled|crawl(?:ed)? content (?:is )?(?:not )?(?:available|found)|no content returned by exa|no results? found)\b/i.test(message);
14
+ }
15
+
16
+ export async function applyExaFetchFallbacks(
17
+ config: WebKitConfig,
18
+ input: FetchInput,
19
+ urls: string[],
20
+ primary: WebFetchResult,
21
+ signal?: AbortSignal,
22
+ ): Promise<WebFetchResult> {
23
+ const results = urls.map((url, i) => primary.results.find((item) => urlsMatch(item.url, url)) ?? primary.results[i] ?? { url, error: "No content returned by Exa." });
24
+ let missing = urls.filter((url, i) => isMissingExaCrawlResult(results[i]));
25
+ if (missing.length === 0) return { ...primary, results };
26
+
27
+ for (const { name, provider } of fallbackProviders(config)) {
28
+ if (missing.length === 0) break;
29
+ let fetched: WebFetchResult;
30
+ try {
31
+ fetched = await provider.fetch({ ...input, url: undefined, urls: missing }, signal);
32
+ } catch {
33
+ continue;
34
+ }
35
+ const mapped = mapFetchResults(missing, fetched);
36
+ missing = missing.filter((url) => {
37
+ const item = mapped.get(url);
38
+ if (!hasUsableContent(item)) return true;
39
+ const index = urls.findIndex((candidate) => candidate === url);
40
+ results[index] = {
41
+ ...item,
42
+ url,
43
+ metadata: {
44
+ ...(isRecord(item?.metadata) ? item.metadata : {}),
45
+ fallbackProvider: name,
46
+ fallbackFrom: primary.provider,
47
+ fetchedUrl: item?.url,
48
+ },
49
+ };
50
+ return false;
51
+ });
52
+ }
53
+
54
+ return { ...primary, results };
55
+ }
56
+
57
+ function fallbackProviders(config: WebKitConfig): Array<{ name: FetchProviderName; provider: FetchProvider }> {
58
+ const providers: Array<{ name: FetchProviderName; provider: FetchProvider }> = [];
59
+ if (config.apiKeys.tinyfish) providers.push({ name: "tinyfish", provider: new TinyFishProvider(config) });
60
+ if (config.apiKeys.firecrawl) providers.push({ name: "firecrawl", provider: new FirecrawlProvider(config) });
61
+ providers.push({ name: "markdown_new", provider: new MarkdownNewProvider(config) });
62
+ return providers;
63
+ }
64
+
65
+ function mapFetchResults(requested: string[], fetched: WebFetchResult): Map<string, FetchPage> {
66
+ const out = new Map<string, FetchPage>();
67
+ const remaining = [...(fetched.results ?? [])];
68
+ for (const url of requested) {
69
+ const index = remaining.findIndex((item) => urlsMatch(item.url, url));
70
+ if (index >= 0) out.set(url, remaining.splice(index, 1)[0]);
71
+ }
72
+ requested.forEach((url, index) => { if (!out.has(url) && fetched.results?.[index]) out.set(url, fetched.results[index]); });
73
+ return out;
74
+ }
75
+
76
+ function hasUsableContent(item: FetchPage | undefined): item is FetchPage {
77
+ return !!item && !item.error && typeof item.content === "string" && item.content.trim().length > 0;
78
+ }
79
+
80
+ function isRecord(value: unknown): value is Record<string, unknown> {
81
+ return !!value && typeof value === "object" && !Array.isArray(value);
82
+ }
@@ -0,0 +1,60 @@
1
+ import { asSnippet, mapConcurrent, normalizeUrls, requestJson } from "../http.js";
2
+ import { FETCH_CONCURRENCY } from "../limits.js";
3
+ import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
4
+ import { requireKey } from "../config.js";
5
+
6
+ export class FirecrawlProvider implements SearchProvider, FetchProvider {
7
+ private key: string;
8
+ constructor(config: WebKitConfig) { this.key = requireKey(config, "firecrawl"); }
9
+ private headers() { return { "content-type": "application/json", authorization: `Bearer ${this.key}` }; }
10
+
11
+ async search(input: SearchInput, signal?: AbortSignal) {
12
+ const data = await requestJson<any>("https://api.firecrawl.dev/v2/search", {
13
+ method: "POST", headers: this.headers(), signal, timeoutMs: 45000,
14
+ body: JSON.stringify({
15
+ query: input.query,
16
+ limit: input.numResults ?? 10,
17
+ location: input.location,
18
+ country: input.country,
19
+ includeDomains: input.includeDomains,
20
+ excludeDomains: input.excludeDomains,
21
+ categories: input.categories,
22
+ tbs: input.tbs,
23
+ scrapeOptions: input.scrapeOptions ?? (input.scrape === true ? { formats: ["markdown"] } : undefined),
24
+ }),
25
+ });
26
+ const list = data.data?.web ?? data.web ?? data.data ?? [];
27
+ return { provider: "firecrawl" as const, query: input.query, results: list.map((r: any, i: number) => ({
28
+ title: r.title, url: r.url, snippet: asSnippet(r.description ?? r.markdown ?? r.content), siteName: r.siteName, position: i + 1,
29
+ })).filter((r: any) => r.url) };
30
+ }
31
+
32
+ async fetch(input: FetchInput, signal?: AbortSignal) {
33
+ const urls = normalizeUrls(input);
34
+ const results = await mapConcurrent(urls, FETCH_CONCURRENCY, async (url) => {
35
+ try {
36
+ const format = input.format ?? "markdown";
37
+ const formats = [format];
38
+ const data = await requestJson<any>("https://api.firecrawl.dev/v2/scrape", {
39
+ method: "POST", headers: this.headers(), signal, timeoutMs: 90000,
40
+ body: JSON.stringify({
41
+ url,
42
+ formats,
43
+ onlyMainContent: input.onlyMainContent ?? true,
44
+ waitFor: input.waitFor,
45
+ mobile: input.mobile,
46
+ location: input.location,
47
+ maxAge: input.maxAge,
48
+ }),
49
+ });
50
+ const d = data.data ?? data;
51
+ const selected = format === "html" ? d.html : format === "json" ? d.json : d.markdown;
52
+ const content = typeof selected === "string" ? selected : selected == null ? undefined : JSON.stringify(selected, null, 2);
53
+ return { url: d.url ?? url, content, format, title: d.metadata?.title, metadata: d.metadata ?? d };
54
+ } catch (e) {
55
+ return { url, error: e instanceof Error ? e.message : String(e) };
56
+ }
57
+ });
58
+ return { provider: "firecrawl" as const, results };
59
+ }
60
+ }
@@ -0,0 +1,30 @@
1
+ import { validateFetchProvider, validateSearchProvider } from "../config.js";
2
+ import type { FetchProvider, SearchProvider, WebKitConfig } from "../types.js";
3
+ import { BraveProvider } from "./brave.js";
4
+ import { ExaMcpProvider } from "./exa-mcp.js";
5
+ import { ExaProvider } from "./exa.js";
6
+ import { FirecrawlProvider } from "./firecrawl.js";
7
+ import { MarkdownNewProvider } from "./markdown-new.js";
8
+ import { TinyFishProvider } from "./tinyfish.js";
9
+
10
+ export function createSearchProvider(config: WebKitConfig): SearchProvider {
11
+ validateSearchProvider(config.provider_search);
12
+ switch (config.provider_search) {
13
+ case "exa_mcp": return new ExaMcpProvider(config);
14
+ case "exa": return new ExaProvider(config);
15
+ case "tinyfish": return new TinyFishProvider(config);
16
+ case "brave": return new BraveProvider(config);
17
+ case "firecrawl": return new FirecrawlProvider(config);
18
+ }
19
+ }
20
+
21
+ export function createFetchProvider(config: WebKitConfig): FetchProvider {
22
+ validateFetchProvider(config.provider_fetch);
23
+ switch (config.provider_fetch) {
24
+ case "exa_mcp": return new ExaMcpProvider(config);
25
+ case "exa": return new ExaProvider(config);
26
+ case "tinyfish": return new TinyFishProvider(config);
27
+ case "markdown_new": return new MarkdownNewProvider(config);
28
+ case "firecrawl": return new FirecrawlProvider(config);
29
+ }
30
+ }
@@ -0,0 +1,32 @@
1
+ import { FETCH_CONCURRENCY } from "../limits.js";
2
+ import { fetchWithTimeout, mapConcurrent, normalizeUrls } from "../http.js";
3
+ import type { FetchInput, FetchProvider, WebKitConfig } from "../types.js";
4
+
5
+ export class MarkdownNewProvider implements FetchProvider {
6
+ constructor(private config: WebKitConfig) {}
7
+
8
+ async fetch(input: FetchInput, signal?: AbortSignal) {
9
+ const urls = normalizeUrls(input);
10
+ const results = await mapConcurrent(urls, FETCH_CONCURRENCY, async (url) => {
11
+ try {
12
+ const res = await fetchWithTimeout("https://markdown.new/", {
13
+ method: "POST",
14
+ headers: { "content-type": "application/json", accept: "text/markdown, text/plain, */*" },
15
+ body: JSON.stringify({
16
+ url,
17
+ method: input.method ?? this.config.markdownNew.method,
18
+ retainImages: input.retainImages ?? this.config.markdownNew.retainImages,
19
+ }),
20
+ signal,
21
+ timeoutMs: 45_000,
22
+ });
23
+ const text = await res.text();
24
+ if (!res.ok) throw new Error(`${res.status} ${res.statusText}: ${text.slice(0, 1000)}`);
25
+ return { url, content: text, format: "markdown" as const };
26
+ } catch (e) {
27
+ return { url, error: e instanceof Error ? e.message : String(e) };
28
+ }
29
+ });
30
+ return { provider: "markdown_new" as const, results };
31
+ }
32
+ }
@@ -0,0 +1,39 @@
1
+ import { asSnippet, normalizeUrls, requestJson } from "../http.js";
2
+ import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
3
+ import { requireKey } from "../config.js";
4
+
5
+ export class TinyFishProvider implements SearchProvider, FetchProvider {
6
+ private key: string;
7
+ constructor(config: WebKitConfig) { this.key = requireKey(config, "tinyfish"); }
8
+
9
+ async search(input: SearchInput, signal?: AbortSignal) {
10
+ const url = new URL("https://api.search.tinyfish.ai/");
11
+ url.searchParams.set("query", input.query);
12
+ if (input.numResults) url.searchParams.set("limit", String(input.numResults));
13
+ if (typeof input.page === "number") url.searchParams.set("page", String(Math.min(input.page, 10)));
14
+ const data = await requestJson<any>(url.toString(), { headers: { "X-API-Key": this.key }, signal, timeoutMs: 10000 });
15
+ const list = data.results ?? data.data ?? data.web ?? [];
16
+ return { provider: "tinyfish" as const, query: input.query, results: list.map((r: any, i: number) => ({
17
+ title: r.title, url: r.url ?? r.link, snippet: asSnippet(r.snippet ?? r.description ?? r.text), siteName: r.siteName ?? r.source, position: r.position ?? i + 1,
18
+ })).filter((r: any) => r.url) };
19
+ }
20
+
21
+ async fetch(input: FetchInput, signal?: AbortSignal) {
22
+ const urls = normalizeUrls(input);
23
+ if (urls.length > 10) throw new Error("TinyFish fetch supports a maximum of 10 URLs per request.");
24
+ const data = await requestJson<any>("https://api.fetch.tinyfish.ai", {
25
+ method: "POST",
26
+ headers: { "content-type": "application/json", "X-API-Key": this.key },
27
+ body: JSON.stringify({ urls, format: input.format ?? "markdown", links: input.links, image_links: input.imageLinks }),
28
+ signal,
29
+ timeoutMs: 150000,
30
+ });
31
+ const list = data.results ?? data.data ?? data.pages ?? [];
32
+ return { provider: "tinyfish" as const, results: urls.map((url, i) => {
33
+ const r = list.find((x: any) => (x.url ?? x.source_url) === url) ?? list[i];
34
+ if (!r) return { url, error: "No content returned by TinyFish." };
35
+ const content = r.text ?? r.content ?? r.markdown ?? r.html;
36
+ return { url, content: typeof content === "string" ? content : JSON.stringify(content, null, 2), format: input.format ?? r.format ?? "markdown", title: r.title, metadata: r, error: r.error };
37
+ }) };
38
+ }
39
+ }
package/src/types.ts ADDED
@@ -0,0 +1,47 @@
1
+ export type SearchProviderName = "exa_mcp" | "exa" | "tinyfish" | "brave" | "firecrawl";
2
+ export type FetchProviderName = "exa_mcp" | "exa" | "tinyfish" | "markdown_new" | "firecrawl";
3
+ export type FetchFormat = "markdown" | "html" | "json";
4
+
5
+ export interface WebKitConfig {
6
+ provider_search: SearchProviderName;
7
+ provider_fetch: FetchProviderName;
8
+ apiKeys: Partial<Record<"exa" | "tinyfish" | "brave" | "firecrawl", string>>;
9
+ markdownNew: { method: "auto" | "ai" | "browser"; retainImages: boolean };
10
+ }
11
+
12
+ export interface SearchInput {
13
+ query: string;
14
+ numResults?: number;
15
+ [key: string]: unknown;
16
+ }
17
+
18
+ export interface WebSearchResult {
19
+ provider: SearchProviderName;
20
+ query: string;
21
+ results: Array<{ title?: string; url: string; snippet?: string; siteName?: string; position?: number }>;
22
+ }
23
+
24
+ export interface FetchInput {
25
+ url?: string;
26
+ urls?: string[];
27
+ offset?: number;
28
+ limit?: number;
29
+ refresh?: boolean;
30
+ format?: FetchFormat;
31
+ links?: boolean;
32
+ imageLinks?: boolean;
33
+ [key: string]: unknown;
34
+ }
35
+
36
+ export interface WebFetchResult {
37
+ provider: FetchProviderName;
38
+ results: Array<{ url: string; content?: string; format?: FetchFormat; title?: string; metadata?: Record<string, unknown>; error?: string }>;
39
+ }
40
+
41
+ export interface SearchProvider {
42
+ search(input: SearchInput, signal?: AbortSignal): Promise<WebSearchResult>;
43
+ }
44
+
45
+ export interface FetchProvider {
46
+ fetch(input: FetchInput, signal?: AbortSignal): Promise<WebFetchResult>;
47
+ }
package/src/urls.ts ADDED
@@ -0,0 +1,40 @@
1
+ import { MAX_URL_COUNT, MAX_URL_LENGTH } from "./limits.js";
2
+
3
+ export function normalizeWebUrl(value: string): string {
4
+ if (value.length > MAX_URL_LENGTH) throw new Error(`URL length must be <= ${MAX_URL_LENGTH} characters.`);
5
+ let parsed: URL;
6
+ try {
7
+ parsed = new URL(value.trim());
8
+ } catch {
9
+ throw new Error(`Malformed URL: ${value}`);
10
+ }
11
+ if (parsed.protocol !== "http:" && parsed.protocol !== "https:") throw new Error(`URL scheme must be http or https: ${value}`);
12
+ if (parsed.username || parsed.password) throw new Error(`URL credentials are not allowed: ${value}`);
13
+ parsed.hash = "";
14
+ return parsed.toString();
15
+ }
16
+
17
+ export function canonicalWebUrl(value: string): string {
18
+ const parsed = new URL(normalizeWebUrl(value));
19
+ parsed.hostname = parsed.hostname.toLowerCase();
20
+ if ((parsed.protocol === "https:" && parsed.port === "443") || (parsed.protocol === "http:" && parsed.port === "80")) parsed.port = "";
21
+ return parsed.toString();
22
+ }
23
+
24
+ export function normalizeUrlInput(input: { url?: string; urls?: string[] }, maxCount = MAX_URL_COUNT): string[] {
25
+ const raw = [...(Array.isArray(input.urls) ? input.urls : []), ...(input.url ? [input.url] : [])].map((url) => String(url).trim()).filter(Boolean);
26
+ const unique = [...new Set(raw.map(normalizeWebUrl))];
27
+ if (unique.length > maxCount) throw new Error(`Too many URLs: maximum is ${maxCount}.`);
28
+ return unique;
29
+ }
30
+
31
+ export function urlsMatch(a: string | undefined, b: string | undefined): boolean {
32
+ if (!a || !b) return false;
33
+ try {
34
+ if (canonicalWebUrl(a) === canonicalWebUrl(b)) return true;
35
+ const trimSlash = (s: string) => s.endsWith("/") ? s.slice(0, -1) : s;
36
+ return trimSlash(canonicalWebUrl(a)) === trimSlash(canonicalWebUrl(b));
37
+ } catch {
38
+ return a === b;
39
+ }
40
+ }