pi-web-kit 0.2.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,7 @@
1
- import { asSnippet, normalizeUrls, requestJson } from "../http.js";
2
- import { MAX_URL_COUNT } from "../limits.js";
3
- import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
1
+ import { asSnippet, normalizeUrls, requestJson, withoutContent } from "../http.js";
2
+ import { DEFAULT_NUM_RESULTS, MAX_URL_COUNT, TINYFISH_MAX_PAGE } from "../limits.js";
3
+ import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig, WebSearchResult } from "../types.js";
4
+ import { canonicalWebUrl, urlsMatch } from "../urls.js";
4
5
  import { requireKey } from "../config.js";
5
6
 
6
7
  export class TinyFishProvider implements SearchProvider, FetchProvider {
@@ -8,15 +9,81 @@ export class TinyFishProvider implements SearchProvider, FetchProvider {
8
9
  constructor(config: WebKitConfig) { this.key = requireKey(config, "tinyfish"); }
9
10
 
10
11
  async search(input: SearchInput, signal?: AbortSignal) {
11
- const url = new URL("https://api.search.tinyfish.ai/");
12
- url.searchParams.set("query", input.query);
13
- if (input.numResults) url.searchParams.set("limit", String(input.numResults));
14
- if (typeof input.page === "number") url.searchParams.set("page", String(Math.min(input.page, 10)));
15
- const data = await requestJson<any>(url.toString(), { headers: { "X-API-Key": this.key }, signal, timeoutMs: 10000 });
16
- const list = data.results ?? data.data ?? data.web ?? [];
17
- return { provider: "tinyfish" as const, query: input.query, results: list.map((r: any, i: number) => ({
18
- title: r.title, url: r.url ?? r.link, snippet: asSnippet(r.snippet ?? r.description ?? r.text), siteName: r.siteName ?? r.source, position: r.position ?? i + 1,
19
- })).filter((r: any) => r.url) };
12
+ const desired = input.numResults ?? DEFAULT_NUM_RESULTS;
13
+ const firstPage = typeof input.page === "number" ? Math.max(0, Math.min(input.page, TINYFISH_MAX_PAGE)) : 0;
14
+ const research = input.domainType === "research_paper";
15
+ const dates = orderedStrings(input.afterDate, input.beforeDate);
16
+ const years = orderedNumbers(input.pubYearMin, input.pubYearMax);
17
+ const results: WebSearchResult["results"] = [];
18
+ const seen = new Set<string>();
19
+ let lastPage = firstPage - 1;
20
+
21
+ for (let page = firstPage; page <= TINYFISH_MAX_PAGE && results.length < desired; page++) {
22
+ lastPage = page;
23
+ const url = new URL("https://api.search.tinyfish.ai/");
24
+ url.searchParams.set("query", input.query);
25
+ url.searchParams.set("page", String(page));
26
+ setString(url, "purpose", input.purpose);
27
+ setString(url, "location", input.location);
28
+ setString(url, "language", input.language);
29
+ setDomains(url, "include_domains", input.includeDomains);
30
+ setDomains(url, "exclude_domains", input.excludeDomains);
31
+ setString(url, "domain_type", input.domainType);
32
+ if (research) {
33
+ setNumber(url, "pub_year_min", years[0]);
34
+ setNumber(url, "pub_year_max", years[1]);
35
+ } else if (typeof input.recencyMinutes === "number") {
36
+ setNumber(url, "recency_minutes", input.recencyMinutes);
37
+ } else {
38
+ setString(url, "after_date", dates[0]);
39
+ setString(url, "before_date", dates[1]);
40
+ }
41
+ const data = await requestJson<any>(url.toString(), { headers: { "X-API-Key": this.key }, signal, timeoutMs: 10000 });
42
+ const list = data.results ?? data.data ?? data.web ?? [];
43
+ if (!Array.isArray(list) || list.length === 0) break;
44
+ for (const r of list) {
45
+ const resultUrl = r.url ?? r.link;
46
+ const key = typeof resultUrl === "string" ? canonicalResultUrl(resultUrl) : undefined;
47
+ if (!key || seen.has(key)) continue;
48
+ seen.add(key);
49
+ results.push({
50
+ title: r.title,
51
+ url: resultUrl,
52
+ snippet: asSnippet(r.snippet ?? r.description ?? r.text),
53
+ siteName: r.siteName ?? r.source,
54
+ position: results.length + 1,
55
+ });
56
+ if (results.length === desired) break;
57
+ }
58
+ if (typeof data.total_results === "number" && results.length >= data.total_results) break;
59
+ }
60
+
61
+ if (input.contextTokens && results.length) {
62
+ const contextCharacters = input.contextTokens * 4;
63
+ let extractedCharacters = 0;
64
+ for (let i = 0; i < results.length && extractedCharacters < contextCharacters;) {
65
+ // ponytail: budget for about 512 useful tokens per page; fetch another batch when pages are short.
66
+ const batchSize = Math.min(MAX_URL_COUNT, Math.max(1, Math.ceil((contextCharacters - extractedCharacters) / 2_048)));
67
+ const batch = results.slice(i, i + batchSize);
68
+ const fetched = await this.fetch({ urls: batch.map((item) => item.url), format: "markdown", purpose: input.purpose }, signal);
69
+ for (const item of batch) {
70
+ const page = fetched.results.find((candidate) => urlsMatch(candidate.url, item.url) || urlsMatch(candidate.metadata?.requestedUrl, item.url));
71
+ if (page?.content) {
72
+ item.content = page.content;
73
+ item.contentFormat = "markdown";
74
+ extractedCharacters += page.content.length;
75
+ }
76
+ }
77
+ i += batch.length;
78
+ }
79
+ }
80
+
81
+ return {
82
+ provider: "tinyfish" as const,
83
+ query: input.query,
84
+ effectiveResultLimit: lastPage === TINYFISH_MAX_PAGE && results.length < desired ? results.length : desired,
85
+ results,
86
+ };
20
87
  }
21
88
 
22
89
  async fetch(input: FetchInput, signal?: AbortSignal) {
@@ -25,20 +92,73 @@ export class TinyFishProvider implements SearchProvider, FetchProvider {
25
92
  const data = await requestJson<any>("https://api.fetch.tinyfish.ai", {
26
93
  method: "POST",
27
94
  headers: { "content-type": "application/json", "X-API-Key": this.key },
28
- body: JSON.stringify({ urls, format: input.format ?? "markdown", links: input.links, image_links: input.imageLinks }),
95
+ body: JSON.stringify({
96
+ urls,
97
+ purpose: input.purpose,
98
+ format: input.format ?? "markdown",
99
+ links: input.links,
100
+ image_links: input.imageLinks,
101
+ ttl: input.refresh === true
102
+ ? 0
103
+ : input.ttl ?? (typeof input.maxAgeMs === "number" ? Math.floor(input.maxAgeMs / 1_000) : undefined),
104
+ per_url_timeout_ms: input.perUrlTimeoutMs,
105
+ include_selectors: input.includeSelectors,
106
+ exclude_selectors: input.excludeSelectors,
107
+ }),
29
108
  signal,
30
109
  timeoutMs: 150000,
31
110
  });
32
111
  const list = data.results ?? data.data ?? data.pages ?? [];
33
- return { provider: "tinyfish" as const, results: urls.map((url, i) => {
34
- const r = list.find((x: any) => (x.url ?? x.source_url) === url) ?? list[i];
35
- if (!r) return { url, error: "No content returned by TinyFish." };
112
+ const errors = data.errors ?? [];
113
+ return { provider: "tinyfish" as const, results: urls.map((url) => {
114
+ const r = list.find((x: any) => urlsMatch(x.url ?? x.source_url, url));
115
+ const failure = errors.find((x: any) => urlsMatch(x.url, url));
116
+ if (!r) return { url, error: failure ? `${failure.error}${failure.status ? ` (${failure.status})` : ""}` : "No content returned by TinyFish." };
36
117
  const content = r.text ?? r.content ?? r.markdown ?? r.html;
37
- return { url, content: stringifyContent(content), format: input.format ?? r.format ?? "markdown", title: r.title, metadata: r, error: r.error };
118
+ return {
119
+ url: r.final_url ?? r.url ?? url,
120
+ content: stringifyContent(content),
121
+ format: input.format ?? r.format ?? "markdown",
122
+ title: r.title,
123
+ metadata: { ...withoutContent(r), requestedUrl: url, finalUrl: r.final_url },
124
+ error: r.error,
125
+ };
38
126
  }) };
39
127
  }
40
128
  }
41
129
 
130
+ function setString(url: URL, name: string, value: unknown) {
131
+ if (typeof value === "string" && value) url.searchParams.set(name, value);
132
+ }
133
+
134
+ function setNumber(url: URL, name: string, value: unknown) {
135
+ if (typeof value === "number") url.searchParams.set(name, String(value));
136
+ }
137
+
138
+ function setDomains(url: URL, name: string, value: unknown) {
139
+ if (Array.isArray(value) && value.length) url.searchParams.set(name, value.join(","));
140
+ else setString(url, name, value);
141
+ }
142
+
143
+ function orderedStrings(a: unknown, b: unknown): [string | undefined, string | undefined] {
144
+ const values = [a, b].filter((value): value is string => typeof value === "string").sort();
145
+ return [values[0], values[1]];
146
+ }
147
+
148
+ function orderedNumbers(a: unknown, b: unknown): [number | undefined, number | undefined] {
149
+ const values = [a, b].filter((value): value is number => typeof value === "number").sort((x, y) => x - y);
150
+ return [values[0], values[1]];
151
+ }
152
+
42
153
  function stringifyContent(value: unknown): string | undefined {
43
154
  return typeof value === "string" ? value : JSON.stringify(value, null, 2);
44
155
  }
156
+
157
+ function canonicalResultUrl(value: string): string {
158
+ try {
159
+ const canonical = canonicalWebUrl(value);
160
+ return canonical.endsWith("/") ? canonical.slice(0, -1) : canonical;
161
+ } catch {
162
+ return value;
163
+ }
164
+ }
package/src/types.ts CHANGED
@@ -1,5 +1,5 @@
1
- export type SearchProviderName = "exa_mcp" | "exa" | "tinyfish" | "brave" | "firecrawl";
2
- export type FetchProviderName = "exa_mcp" | "exa" | "tinyfish" | "markdown_new" | "firecrawl";
1
+ export type SearchProviderName = "exa" | "tinyfish" | "brave" | "firecrawl";
2
+ export type FetchProviderName = "exa" | "tinyfish" | "markdown_new" | "firecrawl";
3
3
  export type FetchFormat = "markdown" | "html" | "json";
4
4
 
5
5
  export interface WebKitConfig {
@@ -12,13 +12,16 @@ export interface WebKitConfig {
12
12
  export interface SearchInput {
13
13
  query: string;
14
14
  numResults?: number;
15
+ contextTokens?: number;
16
+ purpose?: string;
15
17
  [key: string]: unknown;
16
18
  }
17
19
 
18
20
  export interface WebSearchResult {
19
21
  provider: SearchProviderName;
20
22
  query: string;
21
- results: Array<{ title?: string; url: string; snippet?: string; siteName?: string; position?: number }>;
23
+ effectiveResultLimit?: number;
24
+ results: Array<{ title?: string; url: string; snippet?: string; content?: string; contentFormat?: "markdown" | "text"; siteName?: string; position?: number }>;
22
25
  }
23
26
 
24
27
  export interface FetchInput {
@@ -27,9 +30,15 @@ export interface FetchInput {
27
30
  offset?: number;
28
31
  limit?: number;
29
32
  refresh?: boolean;
33
+ maxAgeMs?: number;
30
34
  format?: FetchFormat;
31
35
  links?: boolean;
32
36
  imageLinks?: boolean;
37
+ purpose?: string;
38
+ ttl?: number;
39
+ perUrlTimeoutMs?: number;
40
+ includeSelectors?: string[];
41
+ excludeSelectors?: string[];
33
42
  [key: string]: unknown;
34
43
  }
35
44
 
package/src/urls.ts CHANGED
@@ -6,10 +6,10 @@ export function normalizeWebUrl(value: string): string {
6
6
  try {
7
7
  parsed = new URL(value.trim());
8
8
  } catch {
9
- throw new Error(`Malformed URL: ${value}`);
9
+ throw new Error("Malformed URL.");
10
10
  }
11
- if (parsed.protocol !== "http:" && parsed.protocol !== "https:") throw new Error(`URL scheme must be http or https: ${value}`);
12
- if (parsed.username || parsed.password) throw new Error(`URL credentials are not allowed: ${value}`);
11
+ if (parsed.protocol !== "http:" && parsed.protocol !== "https:") throw new Error("URL scheme must be http or https.");
12
+ if (parsed.username || parsed.password) throw new Error("URL credentials are not allowed.");
13
13
  parsed.hash = "";
14
14
  return parsed.toString();
15
15
  }
@@ -1,102 +0,0 @@
1
- import { asSnippet, fetchWithTimeout, normalizeUrls } from "../http.js";
2
- import { DEFAULT_NUM_RESULTS } from "../limits.js";
3
- import { urlsMatch } from "../urls.js";
4
- import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
5
- import { applyExaFetchFallbacks } from "./fallback.js";
6
-
7
- export class ExaMcpProvider implements SearchProvider, FetchProvider {
8
- private sessionId?: string;
9
- private nextId = 1;
10
- constructor(private config: WebKitConfig) {}
11
-
12
- async search(input: SearchInput, signal?: AbortSignal) {
13
- const result = await this.callTool("web_search_exa", { query: input.query, numResults: input.numResults ?? DEFAULT_NUM_RESULTS }, signal);
14
- return { provider: "exa_mcp" as const, query: input.query, results: normalizeSearch(result) };
15
- }
16
-
17
- async fetch(input: FetchInput, signal?: AbortSignal) {
18
- const urls = normalizeUrls(input);
19
- const result = await this.callTool("web_fetch_exa", { urls }, signal);
20
- const pages = normalizeFetch(result, urls);
21
- return applyExaFetchFallbacks(this.config, input, urls, { provider: "exa_mcp" as const, results: pages }, signal);
22
- }
23
-
24
- private async callTool(name: string, args: Record<string, unknown>, signal?: AbortSignal): Promise<any> {
25
- await this.initialize(signal);
26
- return this.rpc("tools/call", { name, arguments: args }, signal);
27
- }
28
-
29
- private async initialize(signal?: AbortSignal) {
30
- if (this.sessionId) return;
31
- const data = await this.rpcRaw("initialize", { protocolVersion: "2025-06-18", capabilities: {}, clientInfo: { name: "pi-web-kit", version: "0.1.0" } }, signal);
32
- this.sessionId = data.sessionId ?? data.payload?.sessionId;
33
- if (!this.sessionId) throw new Error("Exa MCP initialize did not return an mcp-session-id response header.");
34
- await this.rpcRaw("notifications/initialized", undefined, signal, true);
35
- }
36
-
37
- private async rpc(method: string, params: unknown, signal?: AbortSignal) {
38
- const data = await this.rpcRaw(method, params, signal);
39
- if (data.payload?.error) throw new Error(data.payload.error.message ?? JSON.stringify(data.payload.error));
40
- return data.payload?.result ?? data.payload;
41
- }
42
-
43
- private async rpcRaw(method: string, params: unknown, signal?: AbortSignal, notification = false): Promise<{ payload?: any; sessionId?: string }> {
44
- const headers: Record<string, string> = { "content-type": "application/json", accept: "application/json, text/event-stream" };
45
- if (this.sessionId) headers["mcp-session-id"] = this.sessionId;
46
- if (this.config.apiKeys.exa) headers["x-api-key"] = this.config.apiKeys.exa;
47
- const body = notification ? { jsonrpc: "2.0", method, params } : { jsonrpc: "2.0", id: this.nextId++, method, params };
48
- const res = await fetchWithTimeout("https://mcp.exa.ai/mcp", { method: "POST", headers, body: JSON.stringify(body), signal, timeoutMs: 45_000 });
49
- const text = await res.text();
50
- if (!res.ok) throw new Error(`Exa MCP ${method} failed: ${res.status} ${res.statusText}: ${text.slice(0, 1000)}`);
51
- if (notification) return { sessionId: this.sessionId };
52
- const sessionId = res.headers.get("mcp-session-id") ?? undefined;
53
- const payload = parseMcpResponse(text, res.headers.get("content-type") ?? "");
54
- return { payload, sessionId };
55
- }
56
- }
57
-
58
- function parseMcpResponse(text: string, contentType: string): any {
59
- if (!text.trim()) return undefined;
60
- if (contentType.includes("text/event-stream") || text.startsWith("event:") || text.startsWith("data:")) {
61
- const data = text.split(/\r?\n/).filter((line) => line.startsWith("data:")).map((line) => line.slice(5).trim()).join("\n");
62
- return data ? JSON.parse(data) : undefined;
63
- }
64
- return JSON.parse(text);
65
- }
66
-
67
- function textFromContent(result: any): string {
68
- const content = result?.content ?? result;
69
- if (Array.isArray(content)) return content.map((c) => typeof c === "string" ? c : c.text ?? JSON.stringify(c)).join("\n");
70
- return typeof content === "string" ? content : JSON.stringify(content);
71
- }
72
-
73
- function normalizeSearch(result: any) {
74
- const structured = result?.structuredContent ?? result?.result ?? result;
75
- const list = structured.results ?? structured.data ?? structured.items;
76
- if (Array.isArray(list)) return list.map(toSearchResult).filter((r: any) => r.url);
77
- const text = textFromContent(result);
78
- const urls = [...text.matchAll(/https?:\/\/[^\s)\]}>"']+/g)].map((m) => m[0]);
79
- return [...new Set(urls)].map((url, i) => ({ url, snippet: i === 0 ? text.slice(0, 1000) : undefined, position: i + 1 }));
80
- }
81
-
82
- function toSearchResult(r: any, index: number) {
83
- return {
84
- title: r.title,
85
- url: r.url,
86
- snippet: asSnippet(r.snippet ?? r.text ?? r.summary ?? r.highlights),
87
- siteName: r.siteName,
88
- position: index + 1,
89
- };
90
- }
91
-
92
- function normalizeFetch(result: any, urls: string[]) {
93
- const structured = result?.structuredContent ?? result?.result ?? result;
94
- const list = structured.results ?? structured.data ?? structured.pages;
95
- if (Array.isArray(list)) return urls.map((url, i) => {
96
- const r = list.find((x: any) => urlsMatch(x.url, url)) ?? list[i];
97
- if (!r) return { url, error: "No content returned by Exa MCP." };
98
- return { url, title: r.title, content: r.markdown ?? r.text ?? r.content ?? r.html, format: "markdown" as const, metadata: r, error: r.error };
99
- });
100
- const text = textFromContent(result);
101
- return urls.length <= 1 ? [{ url: urls[0] ?? "", content: text, format: "markdown" as const }] : urls.map((url) => ({ url, content: text, format: "markdown" as const }));
102
- }