pi-web-kit 0.2.4 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/README.md +27 -8
- package/SECURITY.md +2 -2
- package/extensions/index.ts +120 -34
- package/package.json +10 -7
- package/src/config.ts +4 -4
- package/src/http.ts +12 -3
- package/src/install-telemetry.ts +48 -40
- package/src/limits.ts +39 -0
- package/src/providers/brave.ts +43 -16
- package/src/providers/exa.ts +28 -11
- package/src/providers/firecrawl.ts +15 -6
- package/src/providers/index.ts +0 -3
- package/src/providers/markdown-new.ts +16 -2
- package/src/providers/tinyfish.ts +137 -17
- package/src/types.ts +12 -3
- package/src/providers/exa-mcp.ts +0 -102
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,28 @@ This project follows the spirit of [Keep a Changelog](https://keepachangelog.com
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.3.0] - 2026-08-31
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- Share install telemetry mechanics through `@mocito/install-telemetry` while preserving Pi-specific settings and state paths.
|
|
14
|
+
- Treat `web_search.numResults` as a provider-agnostic desired limit, cap it to each provider's service constraints, and report requested, effective, returned, and omitted counts.
|
|
15
|
+
- Add provider-agnostic `contextTokens` and `purpose` search controls, preserve provider-native grounding content, and allocate one bounded context budget across ranked results.
|
|
16
|
+
- Use adaptive extraction: native Exa/Brave context is retained automatically, while TinyFish fetch and Firecrawl scrape-on-search run only when expanded context is requested.
|
|
17
|
+
- Expose documented TinyFish search/fetch controls and Brave LLM Context controls.
|
|
18
|
+
- Add provider-agnostic `maxAgeMs`, pass refresh/cache intent through to Exa, TinyFish, and Firecrawl, and request complete bounded pages from Exa fetch.
|
|
19
|
+
|
|
20
|
+
### Fixed
|
|
21
|
+
|
|
22
|
+
- Let `enableInstallTelemetry: false` override an enabled `PI_TELEMETRY` environment flag.
|
|
23
|
+
- Paginate TinyFish searches instead of sending its ignored `limit` parameter.
|
|
24
|
+
- Send markdown.new's documented `retain_images` field and report its response metadata.
|
|
25
|
+
- Match TinyFish fetch responses and per-URL errors by canonical URL instead of response position.
|
|
26
|
+
|
|
27
|
+
### Removed
|
|
28
|
+
|
|
29
|
+
- Remove the Exa MCP search/fetch surface and make the direct Exa API the default. Existing `exa_mcp` configurations must select `exa` and provide `EXA_API_KEY`.
|
|
30
|
+
|
|
9
31
|
## [0.2.4] - 2026-07-28
|
|
10
32
|
|
|
11
33
|
### Changed
|
package/README.md
CHANGED
|
@@ -70,11 +70,10 @@ Pi chooses `web_search` or `web_fetch` automatically when the request calls for
|
|
|
70
70
|
|
|
71
71
|
## Providers
|
|
72
72
|
|
|
73
|
-
Defaults: `provider_search = "
|
|
73
|
+
Defaults: `provider_search = "exa"`, `provider_fetch = "exa"`.
|
|
74
74
|
|
|
75
75
|
| Provider | Search | Fetch | Key |
|
|
76
76
|
|---|---:|---:|---|
|
|
77
|
-
| `exa_mcp` | yes | yes | optional `EXA_API_KEY` |
|
|
78
77
|
| `exa` | yes | yes | `EXA_API_KEY` |
|
|
79
78
|
| `tinyfish` | yes | yes | `TINYFISH_API_KEY` |
|
|
80
79
|
| `brave` | yes | no | `BRAVE_SEARCH_API_KEY` |
|
|
@@ -83,6 +82,18 @@ Defaults: `provider_search = "exa_mcp"`, `provider_fetch = "exa_mcp"`.
|
|
|
83
82
|
|
|
84
83
|
Tool schemas are tailored to the configured providers at startup/reload, so only supported provider-specific fields are exposed. Restart/reload Pi after changing provider config.
|
|
85
84
|
|
|
85
|
+
Provider-native efficiencies are used as follows:
|
|
86
|
+
|
|
87
|
+
| Service | Efficient path |
|
|
88
|
+
|---|---|
|
|
89
|
+
| Exa API | Keep included highlights; request bounded text in the same search only when requested; batch `/contents` fetches with freshness controls. |
|
|
90
|
+
| TinyFish | Paginate only as needed; use dedicated filters; batch only enough top-result fetches to fill requested context; pass TTL and intent. |
|
|
91
|
+
| Brave | Return pre-extracted LLM Context in one search call with native URL, token, snippet, threshold, and Goggles controls. |
|
|
92
|
+
| Firecrawl | Return search metadata by default; use scrape-on-search only when requested; pass cache/main-content options. |
|
|
93
|
+
| markdown.new | Keep `method: auto` so native Markdown falls back to AI and browser rendering only as needed; keep images opt-in. |
|
|
94
|
+
| Context7 | Support `fast` mode to skip reranking and reduce latency. |
|
|
95
|
+
| Exa Code | Send the requested/dynamic context token target directly to Exa Code Context. |
|
|
96
|
+
|
|
86
97
|
## Configuration
|
|
87
98
|
|
|
88
99
|
Resolution order: defaults < environment variables < global config < trusted project config < CLI flags. Project config is ignored unless Pi trusts the current project, including in print, JSON, and RPC modes.
|
|
@@ -92,8 +103,8 @@ Resolution order: defaults < environment variables < global config < trusted pro
|
|
|
92
103
|
```bash
|
|
93
104
|
PI_OFFLINE=1 # disables install/update telemetry
|
|
94
105
|
PI_TELEMETRY=0 # disables install/update telemetry
|
|
95
|
-
PI_WEB_KIT_PROVIDER_SEARCH=
|
|
96
|
-
PI_WEB_KIT_PROVIDER_FETCH=
|
|
106
|
+
PI_WEB_KIT_PROVIDER_SEARCH=exa|tinyfish|brave|firecrawl
|
|
107
|
+
PI_WEB_KIT_PROVIDER_FETCH=exa|tinyfish|markdown_new|firecrawl
|
|
97
108
|
EXA_API_KEY=... # enables Exa provider and code_search
|
|
98
109
|
CONTEXT7_API_KEY=... # enables library_search and library_docs
|
|
99
110
|
TINYFISH_API_KEY=...
|
|
@@ -148,9 +159,15 @@ Searches with the active search provider and returns compact results grouped by
|
|
|
148
159
|
|---|---|---|
|
|
149
160
|
| `query` | string | Single search query. |
|
|
150
161
|
| `queries` | string[] | Multiple related search queries. Max 5 after de-duplication. |
|
|
151
|
-
| `numResults` | integer |
|
|
162
|
+
| `numResults` | integer | Desired results per query. Default: 10. Values above the active provider's limit are capped rather than rejected. |
|
|
163
|
+
| `contextTokens` | integer | Desired extracted context across the result set. Omitted native context uses an 8,192-token output budget; an explicit value can enable extraction. Any positive value; capped at 10,000. |
|
|
164
|
+
| `purpose` | string | Optional task/use-case hint for providers that support separate intent. |
|
|
165
|
+
|
|
166
|
+
`numResults` controls source breadth. `contextTokens` controls grounding depth. The tool automatically keeps native/included Exa highlights and Brave LLM Context. Explicit `contextTokens` enables extra extraction for Exa, TinyFish, and Firecrawl; this can add provider calls or provider cost. Search results keep a compact `snippet` plus ranked `content` and `contentFormat`, with one shared context budget and the existing 50KB tool-output limit.
|
|
167
|
+
|
|
168
|
+
Provider caps are Exa 100, Brave 50, and Firecrawl 100. TinyFish is paginated internally through its service maximum of page 10 and may make up to 11 search requests for one query. Search output reports requested, effective, returned, and omitted result/context counts. Brave `maxUrls` remains as a deprecated alias for `numResults`.
|
|
152
169
|
|
|
153
|
-
|
|
170
|
+
Other provider-specific parameters are exposed only for the configured provider. These include Exa date/domain filters; TinyFish domain, date, geography, language, and publication filters; Brave locale, freshness, spellcheck, Goggles, and LLM Context controls; and Firecrawl scrape/search options.
|
|
154
171
|
|
|
155
172
|
### `web_fetch`
|
|
156
173
|
|
|
@@ -163,8 +180,9 @@ Fetches page content with the active fetch provider. Results are cached in memor
|
|
|
163
180
|
| `offset` | integer | Character offset for cached/ranged reads. Single URL only. |
|
|
164
181
|
| `limit` | integer | Maximum characters to return. Default: 30,000 for one URL, 8,000 for multiple URLs. |
|
|
165
182
|
| `refresh` | boolean | Refetch even if cached. |
|
|
183
|
+
| `maxAgeMs` | integer | Desired maximum local/provider-cached page age in milliseconds. `0` requests live content where supported. |
|
|
166
184
|
|
|
167
|
-
Provider-specific parameters are exposed only for the configured provider
|
|
185
|
+
Provider-specific parameters are exposed only for the configured provider. TinyFish supports `purpose`, `format`, links/images, selectors, per-URL timeout, and a seconds-based `ttl` alias. Exa supports the hours-based `maxAgeHours` alias. markdown.new supports `method` / `retainImages`. Firecrawl supports `format`, `waitFor`, `mobile`, structured `location`, and its existing `maxAge` alias. `refresh: true` also requests live content from Exa, TinyFish, and Firecrawl instead of only bypassing the local cache.
|
|
168
186
|
|
|
169
187
|
### `library_search`
|
|
170
188
|
|
|
@@ -210,7 +228,8 @@ Finds practical code examples, implementation context, setup snippets, migration
|
|
|
210
228
|
| Max cached bytes | 20 MiB |
|
|
211
229
|
| Max URLs per call | 10 |
|
|
212
230
|
| Max queries per call | 5 |
|
|
213
|
-
|
|
|
231
|
+
| Provider `numResults` caps | Exa 100; Brave 50; Firecrawl 100 |
|
|
232
|
+
| Search context budget | 10,000 tokens |
|
|
214
233
|
| Max URL length | 2048 characters |
|
|
215
234
|
|
|
216
235
|
Cache keys include the provider, canonical URL, fetch-affecting parameters, relevant provider defaults, and an opaque SHA-256 API-key/account scope. Internal cache keys are never returned in tool output. `refresh: true` bypasses and replaces the cached entry.
|
package/SECURITY.md
CHANGED
|
@@ -21,8 +21,8 @@ The maintainer will acknowledge reports as soon as practical and coordinate disc
|
|
|
21
21
|
|
|
22
22
|
`pi-web-kit` is a Pi package. Pi extensions execute with the same permissions as the local user running Pi. Users should review installed Pi packages and only install packages from sources they trust.
|
|
23
23
|
|
|
24
|
-
`pi-web-kit` sends search queries and fetched URLs to the configured third-party provider. Page content returned by providers is cached in memory for the lifetime of the Pi process, subject to TTL and size limits. API keys are read from environment variables or local config files and are used only for provider requests. Do not commit API keys, tokens, or config files containing secrets.
|
|
24
|
+
`pi-web-kit` sends search queries and fetched URLs to the configured third-party provider. A TinyFish search may send the same query in up to 11 paginated requests to satisfy the requested result limit. When `contextTokens` is explicit, TinyFish also receives the selected result URLs for batched extraction, while Firecrawl may scrape search results. Page content returned by providers is cached in memory for the lifetime of the Pi process, subject to TTL and size limits. API keys are read from environment variables or local config files and are used only for provider requests. Do not commit API keys, tokens, or config files containing secrets.
|
|
25
25
|
|
|
26
|
-
On startup,
|
|
26
|
+
On startup, `@mocito/install-telemetry` sends a best-effort install/update telemetry ping to the configured telemetry endpoint once per package version unless Pi telemetry is disabled or offline mode is enabled. The ping includes only the package name, version, and parsed platform/runtime/architecture from its User-Agent; it does not include prompts, queries, fetched URLs, file paths, config values, or API keys.
|
|
27
27
|
|
|
28
28
|
The extension validates URLs before fetch calls and only accepts `http:` and `https:` URLs without embedded credentials. This validation reduces accidental misuse but does not sandbox provider responses or the local Pi process.
|
package/extensions/index.ts
CHANGED
|
@@ -4,7 +4,7 @@ import { Text } from "@earendil-works/pi-tui";
|
|
|
4
4
|
import { Type } from "typebox";
|
|
5
5
|
import { fetchCache, type CachedPage } from "../src/cache.js";
|
|
6
6
|
import { resolveConfig } from "../src/config.js";
|
|
7
|
-
import { DEFAULT_FETCH_LIMIT, DEFAULT_NUM_RESULTS, MAX_LIMIT, MAX_NUM_RESULTS, MAX_OFFSET, MAX_QUERY_COUNT, MAX_URL_COUNT, MULTI_FETCH_LIMIT } from "../src/limits.js";
|
|
7
|
+
import { applySearchContextBudget, capSearchResultLimit, DEFAULT_FETCH_LIMIT, DEFAULT_NUM_RESULTS, DEFAULT_SEARCH_CONTEXT_TOKENS, MAX_LIMIT, MAX_NUM_RESULTS, MAX_OFFSET, MAX_QUERY_COUNT, MAX_SEARCH_CONTEXT_TOKENS, MAX_URL_COUNT, MULTI_FETCH_LIMIT, safePrefix, TINYFISH_MAX_PAGE } from "../src/limits.js";
|
|
8
8
|
import { createCodeSearchProvider, createContext7Provider, createFetchProvider, createSearchProvider } from "../src/providers/index.js";
|
|
9
9
|
import { mapFetchResults } from "../src/providers/fallback.js";
|
|
10
10
|
import type { FetchProviderName, SearchProviderName, WebFetchResult } from "../src/types.js";
|
|
@@ -14,11 +14,11 @@ import { reportInstallTelemetry } from "../src/install-telemetry.js";
|
|
|
14
14
|
export default function (pi: ExtensionAPI) {
|
|
15
15
|
void reportInstallTelemetry();
|
|
16
16
|
pi.registerFlag("web-provider-search", {
|
|
17
|
-
description: "Temporary pi-web-kit search provider override (
|
|
17
|
+
description: "Temporary pi-web-kit search provider override (exa, tinyfish, brave, firecrawl)",
|
|
18
18
|
type: "string",
|
|
19
19
|
});
|
|
20
20
|
pi.registerFlag("web-provider-fetch", {
|
|
21
|
-
description: "Temporary pi-web-kit fetch provider override (
|
|
21
|
+
description: "Temporary pi-web-kit fetch provider override (exa, tinyfish, markdown_new, firecrawl)",
|
|
22
22
|
type: "string",
|
|
23
23
|
});
|
|
24
24
|
|
|
@@ -53,11 +53,19 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
|
|
|
53
53
|
async execute(_toolCallId, rawParams, signal, onUpdate, ctx) {
|
|
54
54
|
const params = rawParams as Record<string, any>;
|
|
55
55
|
const queries = normalizeQueries(params);
|
|
56
|
-
const numResults = parseInteger(params.numResults, DEFAULT_NUM_RESULTS, "numResults", 1, MAX_NUM_RESULTS);
|
|
57
56
|
if (queries.length === 0) throw new Error("web_search requires query or queries.");
|
|
58
57
|
|
|
59
58
|
const config = runtimeConfig(pi, ctx.cwd, projectIsTrusted(ctx));
|
|
60
59
|
assertProviderUnchanged("web_search", startupConfig.provider_search, config.provider_search);
|
|
60
|
+
const requestedResultLimit = parseInteger(
|
|
61
|
+
params.numResults ?? (config.provider_search === "brave" ? params.maxUrls : undefined),
|
|
62
|
+
DEFAULT_NUM_RESULTS,
|
|
63
|
+
"numResults",
|
|
64
|
+
1,
|
|
65
|
+
);
|
|
66
|
+
const effectiveResultLimit = capSearchResultLimit(config.provider_search, requestedResultLimit);
|
|
67
|
+
const requestedContextTokens = params.contextTokens == null ? undefined : parseInteger(params.contextTokens, 0, "contextTokens", 1);
|
|
68
|
+
const effectiveContextTokens = Math.min(requestedContextTokens ?? DEFAULT_SEARCH_CONTEXT_TOKENS, MAX_SEARCH_CONTEXT_TOKENS);
|
|
61
69
|
const provider = createSearchProvider(config);
|
|
62
70
|
const grouped = [];
|
|
63
71
|
const progress = createProgress("search", config.provider_search, queries);
|
|
@@ -65,9 +73,20 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
|
|
|
65
73
|
for (const query of queries) {
|
|
66
74
|
markProgressCurrent(progress, query);
|
|
67
75
|
emitProgress(onUpdate, progress);
|
|
68
|
-
const result = await provider.search({ ...params, query, numResults }, signal);
|
|
69
|
-
|
|
70
|
-
|
|
76
|
+
const result = await provider.search({ ...params, query, numResults: effectiveResultLimit, contextTokens: requestedContextTokens == null ? undefined : effectiveContextTokens }, signal);
|
|
77
|
+
const bounded = applySearchContextBudget(result.results, effectiveContextTokens * 4);
|
|
78
|
+
grouped.push({
|
|
79
|
+
query,
|
|
80
|
+
requestedResultLimit,
|
|
81
|
+
effectiveResultLimit: result.effectiveResultLimit ?? effectiveResultLimit,
|
|
82
|
+
requestedContextTokens,
|
|
83
|
+
effectiveContextTokens,
|
|
84
|
+
contextCharacters: bounded.contextCharacters,
|
|
85
|
+
...(bounded.omittedContextCharacters ? { omittedContextCharacters: bounded.omittedContextCharacters } : {}),
|
|
86
|
+
resultCount: bounded.results.length,
|
|
87
|
+
results: bounded.results,
|
|
88
|
+
});
|
|
89
|
+
markProgressDone(progress, query, `${bounded.results.length} results`);
|
|
71
90
|
emitProgress(onUpdate, progress);
|
|
72
91
|
}
|
|
73
92
|
const result = { provider: config.provider_search, queries: grouped };
|
|
@@ -212,7 +231,9 @@ export function buildSearchSchema(provider: SearchProviderName) {
|
|
|
212
231
|
const props: Record<string, any> = {
|
|
213
232
|
query: Type.Optional(Type.String({ description: "Single search query" })),
|
|
214
233
|
queries: Type.Optional(Type.Array(Type.String(), { description: `Multiple related search queries (max ${MAX_QUERY_COUNT})`, maxItems: MAX_QUERY_COUNT })),
|
|
215
|
-
numResults: Type.Optional(int("
|
|
234
|
+
numResults: Type.Optional(int("Desired results per query; capped by the active provider", 1)),
|
|
235
|
+
contextTokens: Type.Optional(int("Desired extracted context tokens; capped to the tool output budget and may enable provider extraction", 1)),
|
|
236
|
+
purpose: Type.Optional(Type.String({ description: "Optional task/use-case hint when supported by the provider", minLength: 1, maxLength: 2_000 })),
|
|
216
237
|
};
|
|
217
238
|
if (provider === "exa") Object.assign(props, {
|
|
218
239
|
includeDomains: Type.Optional(Type.Array(Type.String())),
|
|
@@ -223,13 +244,32 @@ export function buildSearchSchema(provider: SearchProviderName) {
|
|
|
223
244
|
endCrawlDate: Type.Optional(Type.String()),
|
|
224
245
|
type: Type.Optional(Type.String()),
|
|
225
246
|
category: Type.Optional(Type.String()),
|
|
247
|
+
maxAgeHours: Type.Optional(int("Maximum Exa cached content age in hours; 0 forces live crawl, -1 disables live crawl", -1)),
|
|
248
|
+
});
|
|
249
|
+
if (provider === "tinyfish") Object.assign(props, {
|
|
250
|
+
page: Type.Optional(int("Deprecated starting result page", 0, TINYFISH_MAX_PAGE)),
|
|
251
|
+
location: Type.Optional(Type.String()), language: Type.Optional(Type.String()),
|
|
252
|
+
includeDomains: Type.Optional(Type.Array(Type.String())), excludeDomains: Type.Optional(Type.Array(Type.String())),
|
|
253
|
+
domainType: Type.Optional(Type.Union([Type.Literal("web"), Type.Literal("news"), Type.Literal("research_paper")])),
|
|
254
|
+
afterDate: Type.Optional(Type.String({ pattern: "^\\d{4}-\\d{2}-\\d{2}$" })), beforeDate: Type.Optional(Type.String({ pattern: "^\\d{4}-\\d{2}-\\d{2}$" })),
|
|
255
|
+
recencyMinutes: Type.Optional(int("Freshness window in minutes", 1, 5_256_000)),
|
|
256
|
+
pubYearMin: Type.Optional(int("Minimum publication year", 0, 9_999)), pubYearMax: Type.Optional(int("Maximum publication year", 0, 9_999)),
|
|
226
257
|
});
|
|
227
|
-
if (provider === "tinyfish") Object.assign(props, { page: Type.Optional(int("Result page", 1, 10)) });
|
|
228
258
|
if (provider === "brave") Object.assign(props, {
|
|
229
|
-
country: Type.Optional(Type.String(
|
|
259
|
+
country: Type.Optional(Type.String({ minLength: 2, maxLength: 2 })), searchLang: Type.Optional(Type.String({ minLength: 2, maxLength: 10 })),
|
|
260
|
+
safesearch: Type.Optional(Type.Union([Type.Literal("off"), Type.Literal("moderate"), Type.Literal("strict")])),
|
|
261
|
+
freshness: Type.Optional(Type.String()), spellcheck: Type.Optional(Type.Boolean()),
|
|
262
|
+
contextThresholdMode: Type.Optional(Type.Union([Type.Literal("disabled"), Type.Literal("strict"), Type.Literal("balanced"), Type.Literal("lenient")])),
|
|
263
|
+
maxSnippets: Type.Optional(int("Maximum context snippets", 1, 256)),
|
|
264
|
+
maxTokensPerUrl: Type.Optional(int("Maximum context tokens per URL", 512, 8_192)),
|
|
265
|
+
maxSnippetsPerUrl: Type.Optional(int("Maximum context snippets per URL", 1, 100)),
|
|
266
|
+
goggles: Type.Optional(Type.Union([Type.String(), Type.Array(Type.String(), { maxItems: 3 })])),
|
|
267
|
+
maxUrls: Type.Optional(int("Deprecated alias for numResults", 1)),
|
|
230
268
|
});
|
|
231
269
|
if (provider === "firecrawl") Object.assign(props, {
|
|
232
|
-
location: Type.Optional(Type.String()), country: Type.Optional(Type.String()), includeDomains: Type.Optional(Type.Array(Type.String())), excludeDomains: Type.Optional(Type.Array(Type.String())),
|
|
270
|
+
location: Type.Optional(Type.String()), country: Type.Optional(Type.String()), includeDomains: Type.Optional(Type.Array(Type.String())), excludeDomains: Type.Optional(Type.Array(Type.String())),
|
|
271
|
+
categories: Type.Optional(Type.Array(Type.Union([Type.Literal("research"), Type.Literal("pdf"), Type.Literal("developer")]))),
|
|
272
|
+
tbs: Type.Optional(Type.String()), scrape: Type.Optional(Type.Boolean({ description: "Enable default markdown scrape-on-search" })), scrapeOptions: Type.Optional(Type.Object({}, { additionalProperties: true, description: "Firecrawl scrapeOptions for search." })),
|
|
233
273
|
});
|
|
234
274
|
return Type.Object(props, { additionalProperties: false });
|
|
235
275
|
}
|
|
@@ -241,12 +281,23 @@ export function buildFetchSchema(provider: FetchProviderName) {
|
|
|
241
281
|
offset: Type.Optional(int("Character offset for cached/ranged reads", 0, MAX_OFFSET)),
|
|
242
282
|
limit: Type.Optional(int("Maximum characters to return", 1, MAX_LIMIT)),
|
|
243
283
|
refresh: Type.Optional(Type.Boolean({ description: "Refetch even if cached" })),
|
|
284
|
+
maxAgeMs: Type.Optional(int("Maximum local/provider-cached page age in milliseconds; 0 requests live content where supported", 0)),
|
|
244
285
|
};
|
|
286
|
+
if (provider === "exa") Object.assign(props, { maxAgeHours: Type.Optional(int("Maximum Exa cached content age in hours; 0 forces live crawl, -1 disables live crawl", -1)) });
|
|
245
287
|
if (provider === "tinyfish") Object.assign(props, { format: Type.Optional(Type.Union([Type.Literal("markdown"), Type.Literal("html"), Type.Literal("json")])), links: Type.Optional(Type.Boolean()), imageLinks: Type.Optional(Type.Boolean()) });
|
|
288
|
+
if (provider === "tinyfish") Object.assign(props, {
|
|
289
|
+
purpose: Type.Optional(Type.String({ minLength: 1, maxLength: 2_000 })),
|
|
290
|
+
ttl: Type.Optional(int("Provider cache freshness tolerance in seconds; 0 prefers live fetch", 0)),
|
|
291
|
+
perUrlTimeoutMs: Type.Optional(int("Per-URL timeout in milliseconds", 1, 110_000)),
|
|
292
|
+
includeSelectors: Type.Optional(Type.Array(Type.String({ minLength: 1, maxLength: 1_000 }), { minItems: 1, maxItems: 20 })),
|
|
293
|
+
excludeSelectors: Type.Optional(Type.Array(Type.String({ minLength: 1, maxLength: 1_000 }), { minItems: 1, maxItems: 20 })),
|
|
294
|
+
});
|
|
246
295
|
if (provider === "markdown_new") Object.assign(props, { method: Type.Optional(Type.Union([Type.Literal("auto"), Type.Literal("ai"), Type.Literal("browser")])), retainImages: Type.Optional(Type.Boolean()) });
|
|
247
296
|
if (provider === "firecrawl") Object.assign(props, {
|
|
248
297
|
format: Type.Optional(Type.Union([Type.Literal("markdown"), Type.Literal("html"), Type.Literal("json")])),
|
|
249
|
-
onlyMainContent: Type.Optional(Type.Boolean()), waitFor: Type.Optional(int("Milliseconds to wait", 0, 60_000)), mobile: Type.Optional(Type.Boolean()),
|
|
298
|
+
onlyMainContent: Type.Optional(Type.Boolean()), waitFor: Type.Optional(int("Milliseconds to wait", 0, 60_000)), mobile: Type.Optional(Type.Boolean()),
|
|
299
|
+
location: Type.Optional(Type.Object({ country: Type.Optional(Type.String()), languages: Type.Optional(Type.Array(Type.String())) }, { additionalProperties: false })),
|
|
300
|
+
maxAge: Type.Optional(int("Maximum cached page age in milliseconds", 0)),
|
|
250
301
|
});
|
|
251
302
|
return Type.Object(props, { additionalProperties: false });
|
|
252
303
|
}
|
|
@@ -304,6 +355,7 @@ export async function fetchWithCache(providerName: FetchProviderName, params: Re
|
|
|
304
355
|
const defaultLimit = urls.length > 1 ? MULTI_FETCH_LIMIT : DEFAULT_FETCH_LIMIT;
|
|
305
356
|
const limit = parseInteger(params.limit, defaultLimit, "limit", 1, MAX_LIMIT);
|
|
306
357
|
const refresh = params.refresh === true;
|
|
358
|
+
const maxAgeMs = localCacheMaxAge(providerName, params);
|
|
307
359
|
|
|
308
360
|
const pages = new Map<string, { page?: CachedPage; cached: boolean; refreshed: boolean; error?: string }>();
|
|
309
361
|
const cacheKeys = new Map<string, string>();
|
|
@@ -312,7 +364,7 @@ export async function fetchWithCache(providerName: FetchProviderName, params: Re
|
|
|
312
364
|
const cacheKey = buildCacheKey(providerName, url, params, config);
|
|
313
365
|
cacheKeys.set(url, cacheKey);
|
|
314
366
|
const cached = fetchCache.get(cacheKey);
|
|
315
|
-
if (cached && !refresh) {
|
|
367
|
+
if (cached && !refresh && (maxAgeMs == null || (maxAgeMs > 0 && Date.now() - cached.fetchedAt <= maxAgeMs))) {
|
|
316
368
|
pages.set(url, { page: cached, cached: true, refreshed: false });
|
|
317
369
|
onProgress?.({ status: "done", url, note: "cached" });
|
|
318
370
|
} else {
|
|
@@ -356,6 +408,14 @@ export async function fetchWithCache(providerName: FetchProviderName, params: Re
|
|
|
356
408
|
}) };
|
|
357
409
|
}
|
|
358
410
|
|
|
411
|
+
function localCacheMaxAge(provider: FetchProviderName, params: Record<string, any>): number | undefined {
|
|
412
|
+
if (params.maxAgeMs != null) return parseInteger(params.maxAgeMs, 0, "maxAgeMs", 0);
|
|
413
|
+
if (provider === "exa" && typeof params.maxAgeHours === "number" && params.maxAgeHours >= 0) return params.maxAgeHours * 3_600_000;
|
|
414
|
+
if (provider === "tinyfish" && typeof params.ttl === "number") return params.ttl * 1_000;
|
|
415
|
+
if (provider === "firecrawl" && typeof params.maxAge === "number") return params.maxAge;
|
|
416
|
+
return undefined;
|
|
417
|
+
}
|
|
418
|
+
|
|
359
419
|
|
|
360
420
|
export function pageSlice(page: CachedPage, offset: number, limit: number, cached: boolean, refreshed: boolean) {
|
|
361
421
|
const total = page.content.length;
|
|
@@ -436,8 +496,20 @@ function isSearchResult(value: any): value is { provider?: string; queries: Arra
|
|
|
436
496
|
}
|
|
437
497
|
|
|
438
498
|
function boundSearchResult(value: { provider?: string; queries: Array<{ query?: string; results?: any[] }> }) {
|
|
439
|
-
const compact =
|
|
440
|
-
|
|
499
|
+
const compact = {
|
|
500
|
+
...(limitStrings(value, 1_000) as Record<string, unknown>),
|
|
501
|
+
queries: value.queries.map((group) => ({
|
|
502
|
+
...(limitStrings(group, 1_000) as Record<string, unknown>),
|
|
503
|
+
results: (group.results ?? []).map((item) => ({
|
|
504
|
+
...(limitStrings(item, 1_000) as Record<string, unknown>),
|
|
505
|
+
...(typeof item?.content === "string" ? { content: item.content } : {}),
|
|
506
|
+
})),
|
|
507
|
+
})),
|
|
508
|
+
} as typeof value;
|
|
509
|
+
const maxSnippet = Math.max(0, ...compact.queries.flatMap((group) => (group.results ?? []).flatMap((item) => [
|
|
510
|
+
typeof item?.snippet === "string" ? item.snippet.length : 0,
|
|
511
|
+
typeof item?.content === "string" ? item.content.length : 0,
|
|
512
|
+
])));
|
|
441
513
|
let low = 0;
|
|
442
514
|
let high = maxSnippet;
|
|
443
515
|
let best = searchResultWithLimits(compact, 0);
|
|
@@ -473,17 +545,29 @@ function boundSearchResult(value: { provider?: string; queries: Array<{ query?:
|
|
|
473
545
|
function searchResultWithLimits(value: { provider?: string; queries: Array<{ query?: string; results?: any[] }> }, snippetChars: number, resultsPerQuery?: number) {
|
|
474
546
|
return {
|
|
475
547
|
...value,
|
|
476
|
-
queries: value.queries.map((group) =>
|
|
477
|
-
|
|
478
|
-
results
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
548
|
+
queries: value.queries.map((group) => {
|
|
549
|
+
const originalContextCharacters = (group.results ?? []).reduce((total, item) => total + (typeof item?.content === "string" ? item.content.length : 0), 0);
|
|
550
|
+
const results = (group.results ?? []).slice(0, resultsPerQuery).map((item) => {
|
|
551
|
+
const bounded = { ...item };
|
|
552
|
+
for (const key of ["snippet", "content"] as const) {
|
|
553
|
+
if (typeof bounded[key] !== "string") continue;
|
|
554
|
+
if (snippetChars === 0) delete bounded[key];
|
|
555
|
+
else bounded[key] = safePrefix(bounded[key], snippetChars);
|
|
483
556
|
}
|
|
484
|
-
return
|
|
485
|
-
})
|
|
486
|
-
|
|
557
|
+
return bounded;
|
|
558
|
+
});
|
|
559
|
+
const omittedResultCount = Math.max(0, (group.results?.length ?? 0) - results.length);
|
|
560
|
+
const contextCharacters = results.reduce((total, item) => total + (typeof item?.content === "string" ? item.content.length : 0), 0);
|
|
561
|
+
const omittedContextCharacters = (typeof (group as any).omittedContextCharacters === "number" ? (group as any).omittedContextCharacters : 0) + originalContextCharacters - contextCharacters;
|
|
562
|
+
return {
|
|
563
|
+
...group,
|
|
564
|
+
contextCharacters,
|
|
565
|
+
...(omittedContextCharacters ? { omittedContextCharacters } : {}),
|
|
566
|
+
resultCount: results.length,
|
|
567
|
+
...(omittedResultCount ? { omittedResultCount } : {}),
|
|
568
|
+
results,
|
|
569
|
+
};
|
|
570
|
+
}),
|
|
487
571
|
};
|
|
488
572
|
}
|
|
489
573
|
|
|
@@ -521,12 +605,6 @@ function fetchResultWithContentLimit(value: { provider?: string; results: any[]
|
|
|
521
605
|
};
|
|
522
606
|
}
|
|
523
607
|
|
|
524
|
-
function safePrefix(value: string, maxChars: number): string {
|
|
525
|
-
let end = Math.min(value.length, maxChars);
|
|
526
|
-
if (end > 0 && /[\uD800-\uDBFF]/.test(value[end - 1])) end--;
|
|
527
|
-
return value.slice(0, end);
|
|
528
|
-
}
|
|
529
|
-
|
|
530
608
|
function limitStrings(value: unknown, maxChars: number): unknown {
|
|
531
609
|
if (typeof value === "string") return safePrefix(value, maxChars);
|
|
532
610
|
if (Array.isArray(value)) return value.map((item) => limitStrings(item, maxChars));
|
|
@@ -534,9 +612,11 @@ function limitStrings(value: unknown, maxChars: number): unknown {
|
|
|
534
612
|
return value;
|
|
535
613
|
}
|
|
536
614
|
|
|
537
|
-
function parseInteger(value: unknown, defaultValue: number, name: string, min: number, max
|
|
615
|
+
function parseInteger(value: unknown, defaultValue: number, name: string, min: number, max?: number): number {
|
|
538
616
|
if (value == null) return defaultValue;
|
|
539
|
-
if (typeof value !== "number" || !Number.isInteger(value) || !Number.isFinite(value) || value < min ||
|
|
617
|
+
if (typeof value !== "number" || !Number.isInteger(value) || !Number.isFinite(value) || value < min || (max != null && value > max)) {
|
|
618
|
+
throw new Error(`${name} must be a finite integer ${max == null ? `>= ${min}` : `between ${min} and ${max}`}.`);
|
|
619
|
+
}
|
|
540
620
|
return value;
|
|
541
621
|
}
|
|
542
622
|
|
|
@@ -565,7 +645,6 @@ function fetchConfigDefaults(provider: FetchProviderName, config?: any): Record<
|
|
|
565
645
|
function providerScope(provider: FetchProviderName, config?: any): string {
|
|
566
646
|
const keyMap: Partial<Record<FetchProviderName, string | undefined>> = {
|
|
567
647
|
exa: config?.apiKeys?.exa,
|
|
568
|
-
exa_mcp: config?.apiKeys?.exa,
|
|
569
648
|
tinyfish: config?.apiKeys?.tinyfish,
|
|
570
649
|
firecrawl: config?.apiKeys?.firecrawl,
|
|
571
650
|
};
|
|
@@ -591,7 +670,14 @@ function searchDetails(value: any) {
|
|
|
591
670
|
provider: value.provider,
|
|
592
671
|
queries: value.queries.map((q: any) => ({
|
|
593
672
|
query: q.query,
|
|
673
|
+
requestedResultLimit: q.requestedResultLimit,
|
|
674
|
+
effectiveResultLimit: q.effectiveResultLimit,
|
|
675
|
+
requestedContextTokens: q.requestedContextTokens,
|
|
676
|
+
effectiveContextTokens: q.effectiveContextTokens,
|
|
677
|
+
contextCharacters: q.contextCharacters,
|
|
678
|
+
omittedContextCharacters: q.omittedContextCharacters,
|
|
594
679
|
resultCount: (q.results ?? []).length,
|
|
680
|
+
omittedResultCount: q.omittedResultCount,
|
|
595
681
|
results: (q.results ?? []).map((r: any) => ({ title: r.title, url: r.url, siteName: r.siteName, position: r.position })),
|
|
596
682
|
})),
|
|
597
683
|
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-web-kit",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"description": "Context-efficient web search and fetch tools for Pi.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -58,12 +58,12 @@
|
|
|
58
58
|
"typebox": "*"
|
|
59
59
|
},
|
|
60
60
|
"devDependencies": {
|
|
61
|
-
"@earendil-works/pi-ai": "^0.
|
|
62
|
-
"@earendil-works/pi-coding-agent": "^0.
|
|
63
|
-
"@earendil-works/pi-tui": "^0.
|
|
64
|
-
"@types/node": "^26.
|
|
65
|
-
"tsx": "^4.23.
|
|
66
|
-
"typebox": "^1.3.
|
|
61
|
+
"@earendil-works/pi-ai": "^0.84.1",
|
|
62
|
+
"@earendil-works/pi-coding-agent": "^0.84.1",
|
|
63
|
+
"@earendil-works/pi-tui": "^0.84.1",
|
|
64
|
+
"@types/node": "^26.2.0",
|
|
65
|
+
"tsx": "^4.23.5",
|
|
66
|
+
"typebox": "^1.3.10",
|
|
67
67
|
"typescript": "^7.0.2"
|
|
68
68
|
},
|
|
69
69
|
"publishConfig": {
|
|
@@ -71,5 +71,8 @@
|
|
|
71
71
|
},
|
|
72
72
|
"engines": {
|
|
73
73
|
"node": ">=20.6.0"
|
|
74
|
+
},
|
|
75
|
+
"dependencies": {
|
|
76
|
+
"@mocito/install-telemetry": "0.1.1"
|
|
74
77
|
}
|
|
75
78
|
}
|
package/src/config.ts
CHANGED
|
@@ -3,11 +3,11 @@ import { homedir } from "node:os";
|
|
|
3
3
|
import { join } from "node:path";
|
|
4
4
|
import type { FetchProviderName, SearchProviderName, WebKitConfig } from "./types.js";
|
|
5
5
|
|
|
6
|
-
const SEARCH = ["
|
|
7
|
-
const FETCH = ["
|
|
6
|
+
const SEARCH = ["exa", "tinyfish", "brave", "firecrawl"] as const;
|
|
7
|
+
const FETCH = ["exa", "tinyfish", "markdown_new", "firecrawl"] as const;
|
|
8
8
|
const DEFAULT_CONFIG: WebKitConfig = {
|
|
9
|
-
provider_search: "
|
|
10
|
-
provider_fetch: "
|
|
9
|
+
provider_search: "exa",
|
|
10
|
+
provider_fetch: "exa",
|
|
11
11
|
apiKeys: {},
|
|
12
12
|
markdownNew: { method: "auto", retainImages: false },
|
|
13
13
|
};
|
package/src/http.ts
CHANGED
|
@@ -32,10 +32,19 @@ export async function requestJson<T>(url: string, init: RequestInit & { timeoutM
|
|
|
32
32
|
}
|
|
33
33
|
|
|
34
34
|
export function asSnippet(value: unknown): string | undefined {
|
|
35
|
+
return asText(value)?.slice(0, 1000) || undefined;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export function asText(value: unknown): string | undefined {
|
|
35
39
|
if (value == null) return undefined;
|
|
36
|
-
if (Array.isArray(value)) return value.filter(Boolean).join("\n")
|
|
37
|
-
if (typeof value === "string") return value
|
|
38
|
-
return JSON.stringify(value)
|
|
40
|
+
if (Array.isArray(value)) return value.filter(Boolean).map((item) => typeof item === "string" ? item : JSON.stringify(item)).join("\n") || undefined;
|
|
41
|
+
if (typeof value === "string") return value || undefined;
|
|
42
|
+
return JSON.stringify(value);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function withoutContent(value: any): Record<string, unknown> {
|
|
46
|
+
const { text: _text, content: _content, markdown: _markdown, html: _html, summary: _summary, highlights: _highlights, ...metadata } = value ?? {};
|
|
47
|
+
return metadata;
|
|
39
48
|
}
|
|
40
49
|
|
|
41
50
|
export function normalizeUrls(input: { url?: string; urls?: string[] }, maxCount?: number): string[] {
|
package/src/install-telemetry.ts
CHANGED
|
@@ -1,17 +1,35 @@
|
|
|
1
1
|
import { readFileSync } from "node:fs";
|
|
2
|
-
import { mkdir, writeFile } from "node:fs/promises";
|
|
3
2
|
import { join } from "node:path";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import { reportInstallTelemetry as report } from "@mocito/install-telemetry";
|
|
4
5
|
import { getAgentDir } from "@earendil-works/pi-coding-agent";
|
|
5
6
|
|
|
6
7
|
const PACKAGE_NAME = "pi-web-kit";
|
|
7
|
-
const
|
|
8
|
-
const
|
|
8
|
+
const INSTALL_TELEMETRY_ENDPOINT = "https://mocito.dev/api/report-install";
|
|
9
|
+
const CI_ENVIRONMENT_VARIABLES = [
|
|
10
|
+
"APPVEYOR",
|
|
11
|
+
"BITBUCKET_BUILD_NUMBER",
|
|
12
|
+
"BUILDKITE",
|
|
13
|
+
"CIRCLECI",
|
|
14
|
+
"CODESPACES",
|
|
15
|
+
"DRONE",
|
|
16
|
+
"GITHUB_ACTIONS",
|
|
17
|
+
"GITLAB_CI",
|
|
18
|
+
"JENKINS_URL",
|
|
19
|
+
"NETLIFY",
|
|
20
|
+
"TEAMCITY_VERSION",
|
|
21
|
+
"TF_BUILD",
|
|
22
|
+
"TRAVIS",
|
|
23
|
+
"VERCEL",
|
|
24
|
+
];
|
|
9
25
|
|
|
10
|
-
|
|
26
|
+
interface PiSettingsDocument {
|
|
27
|
+
enableInstallTelemetry?: unknown;
|
|
28
|
+
}
|
|
11
29
|
|
|
12
30
|
function readJsonFile(path: string): unknown {
|
|
13
31
|
try {
|
|
14
|
-
return JSON.parse(readFileSync(path, "utf8"));
|
|
32
|
+
return JSON.parse(readFileSync(path, "utf8")) as unknown;
|
|
15
33
|
} catch {
|
|
16
34
|
return {};
|
|
17
35
|
}
|
|
@@ -22,48 +40,38 @@ function isTruthyEnvFlag(value: string | undefined): boolean {
|
|
|
22
40
|
return value === "1" || value.toLowerCase() === "true" || value.toLowerCase() === "yes";
|
|
23
41
|
}
|
|
24
42
|
|
|
25
|
-
function
|
|
26
|
-
if (
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
const settings = readJsonFile(join(getAgentDir(), "settings.json")) as { enableInstallTelemetry?: unknown };
|
|
30
|
-
return settings.enableInstallTelemetry !== false;
|
|
43
|
+
function isPresentEnvFlag(value: string | undefined): boolean {
|
|
44
|
+
if (!value) return false;
|
|
45
|
+
const normalized = value.toLowerCase();
|
|
46
|
+
return normalized !== "0" && normalized !== "false" && normalized !== "no";
|
|
31
47
|
}
|
|
32
48
|
|
|
33
|
-
function
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
49
|
+
export function isInstallTelemetryEnabled(env: NodeJS.ProcessEnv = process.env, settingsPath = join(getAgentDir(), "settings.json")): boolean {
|
|
50
|
+
if (isTruthyEnvFlag(env.CI)) return false;
|
|
51
|
+
if (CI_ENVIRONMENT_VARIABLES.some((name) => isPresentEnvFlag(env[name]))) return false;
|
|
52
|
+
if (isTruthyEnvFlag(env.PI_OFFLINE)) return false;
|
|
53
|
+
|
|
54
|
+
const settings = readJsonFile(settingsPath) as PiSettingsDocument;
|
|
55
|
+
if (settings.enableInstallTelemetry === false) return false;
|
|
56
|
+
if (env.PI_TELEMETRY !== undefined) return isTruthyEnvFlag(env.PI_TELEMETRY);
|
|
57
|
+
return true;
|
|
40
58
|
}
|
|
41
59
|
|
|
42
|
-
function
|
|
43
|
-
const
|
|
44
|
-
|
|
45
|
-
return `${PACKAGE_NAME}/${version} (${process.platform}; ${runtime}; ${process.arch})`;
|
|
60
|
+
function getPackageVersion(): string {
|
|
61
|
+
const packageJson = readJsonFile(fileURLToPath(new URL("../package.json", import.meta.url))) as { version?: unknown };
|
|
62
|
+
return typeof packageJson.version === "string" && packageJson.version.length > 0 ? packageJson.version : "0.0.0";
|
|
46
63
|
}
|
|
47
64
|
|
|
48
|
-
export
|
|
65
|
+
export function reportInstallTelemetry(): void {
|
|
49
66
|
try {
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
await mkdir(telemetryDir, { recursive: true });
|
|
59
|
-
await writeFile(statePath, `${JSON.stringify({ lastReportedVersion: version }, null, 2)}\n`);
|
|
60
|
-
|
|
61
|
-
const params = new URLSearchParams({ tool: PACKAGE_NAME, version });
|
|
62
|
-
await fetch(`${INSTALL_TELEMETRY_URL}?${params.toString()}`, {
|
|
63
|
-
headers: { "User-Agent": getInstallTelemetryUserAgent(version) },
|
|
64
|
-
signal: AbortSignal.timeout(INSTALL_TELEMETRY_TIMEOUT_MS),
|
|
65
|
-
});
|
|
67
|
+
void report({
|
|
68
|
+
endpoint: INSTALL_TELEMETRY_ENDPOINT,
|
|
69
|
+
tool: PACKAGE_NAME,
|
|
70
|
+
version: getPackageVersion(),
|
|
71
|
+
statePath: join(getAgentDir(), "extensions", "pi-web-kit-install.json"),
|
|
72
|
+
enabled: isInstallTelemetryEnabled(),
|
|
73
|
+
}).catch(() => undefined);
|
|
66
74
|
} catch {
|
|
67
|
-
// Best-effort
|
|
75
|
+
// Best-effort telemetry: ignore local policy and filesystem failures.
|
|
68
76
|
}
|
|
69
77
|
}
|
package/src/limits.ts
CHANGED
|
@@ -1,8 +1,19 @@
|
|
|
1
|
+
import type { SearchProviderName } from "./types.js";
|
|
2
|
+
|
|
1
3
|
export const MAX_QUERY_COUNT = 5;
|
|
2
4
|
export const MAX_URL_COUNT = 10;
|
|
3
5
|
export const MAX_URL_LENGTH = 2_048;
|
|
4
6
|
export const MAX_NUM_RESULTS = 20;
|
|
5
7
|
export const DEFAULT_NUM_RESULTS = 10;
|
|
8
|
+
export const DEFAULT_SEARCH_CONTEXT_TOKENS = 8_192;
|
|
9
|
+
export const MAX_SEARCH_CONTEXT_TOKENS = 10_000;
|
|
10
|
+
export const SEARCH_RESULT_LIMITS: Readonly<Record<SearchProviderName, number | undefined>> = {
|
|
11
|
+
exa: 100,
|
|
12
|
+
brave: 50,
|
|
13
|
+
firecrawl: 100,
|
|
14
|
+
tinyfish: undefined,
|
|
15
|
+
};
|
|
16
|
+
export const TINYFISH_MAX_PAGE = 10;
|
|
6
17
|
export const MAX_OFFSET = 10_000_000;
|
|
7
18
|
export const MAX_LIMIT = 100_000;
|
|
8
19
|
export const DEFAULT_FETCH_LIMIT = 30_000;
|
|
@@ -12,3 +23,31 @@ export const FETCH_CACHE_MAX_ENTRIES = 100;
|
|
|
12
23
|
export const FETCH_CACHE_MAX_BYTES = 20 * 1024 * 1024;
|
|
13
24
|
export const FETCH_CACHE_TTL_MS = 30 * 60 * 1000;
|
|
14
25
|
export const FETCH_CONCURRENCY = 3;
|
|
26
|
+
|
|
27
|
+
export function capSearchResultLimit(provider: SearchProviderName, requested: number): number {
|
|
28
|
+
return Math.min(requested, SEARCH_RESULT_LIMITS[provider] ?? requested);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function applySearchContextBudget(results: Array<{ content?: string; [key: string]: unknown }>, maxCharacters: number) {
|
|
32
|
+
let remaining = maxCharacters;
|
|
33
|
+
let remainingContentResults = results.filter((result) => typeof result.content === "string").length;
|
|
34
|
+
let contextCharacters = 0;
|
|
35
|
+
let omittedContextCharacters = 0;
|
|
36
|
+
const bounded = results.map((result) => {
|
|
37
|
+
if (typeof result.content !== "string") return result;
|
|
38
|
+
const share = Math.max(0, Math.floor(remaining / remainingContentResults));
|
|
39
|
+
const content = safePrefix(result.content, share);
|
|
40
|
+
remaining -= content.length;
|
|
41
|
+
remainingContentResults--;
|
|
42
|
+
contextCharacters += content.length;
|
|
43
|
+
omittedContextCharacters += result.content.length - content.length;
|
|
44
|
+
return { ...result, content };
|
|
45
|
+
});
|
|
46
|
+
return { results: bounded, contextCharacters, omittedContextCharacters };
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export function safePrefix(value: string, maxChars: number): string {
|
|
50
|
+
let end = Math.min(value.length, maxChars);
|
|
51
|
+
if (end > 0 && /[\uD800-\uDBFF]/.test(value[end - 1])) end--;
|
|
52
|
+
return value.slice(0, end);
|
|
53
|
+
}
|
package/src/providers/brave.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { asSnippet, requestJson } from "../http.js";
|
|
1
|
+
import { asSnippet, asText, requestJson } from "../http.js";
|
|
2
2
|
import type { SearchInput, SearchProvider, WebKitConfig } from "../types.js";
|
|
3
3
|
import { requireKey } from "../config.js";
|
|
4
4
|
|
|
@@ -7,24 +7,51 @@ export class BraveProvider implements SearchProvider {
|
|
|
7
7
|
constructor(config: WebKitConfig) { this.key = requireKey(config, "brave"); }
|
|
8
8
|
|
|
9
9
|
async search(input: SearchInput, signal?: AbortSignal) {
|
|
10
|
-
const
|
|
11
|
-
url.searchParams.set("q", input.query);
|
|
10
|
+
const params: Record<string, string | number | boolean | string[]> = { q: boundedQuery(input.query) };
|
|
12
11
|
if (input.numResults) {
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
12
|
+
params.count = input.numResults;
|
|
13
|
+
params.maximum_number_of_urls = input.numResults;
|
|
14
|
+
}
|
|
15
|
+
if (input.contextTokens) params.maximum_number_of_tokens = Math.max(1_024, Math.min(input.contextTokens, 32_768));
|
|
16
|
+
if (typeof input.country === "string") params.country = input.country;
|
|
17
|
+
if (typeof input.searchLang === "string") params.search_lang = input.searchLang;
|
|
18
|
+
if (typeof input.safesearch === "string") params.safesearch = input.safesearch;
|
|
19
|
+
if (typeof input.freshness === "string") params.freshness = input.freshness;
|
|
20
|
+
if (typeof input.spellcheck === "boolean") params.spellcheck = input.spellcheck;
|
|
21
|
+
if (typeof input.contextThresholdMode === "string") params.context_threshold_mode = input.contextThresholdMode;
|
|
22
|
+
if (typeof input.maxSnippets === "number") params.maximum_number_of_snippets = input.maxSnippets;
|
|
23
|
+
if (typeof input.maxTokensPerUrl === "number") params.maximum_number_of_tokens_per_url = input.maxTokensPerUrl;
|
|
24
|
+
if (typeof input.maxSnippetsPerUrl === "number") params.maximum_number_of_snippets_per_url = input.maxSnippetsPerUrl;
|
|
25
|
+
if (typeof input.goggles === "string" || Array.isArray(input.goggles)) params.goggles = input.goggles;
|
|
26
|
+
|
|
27
|
+
const url = new URL("https://api.search.brave.com/res/v1/llm/context");
|
|
28
|
+
for (const [name, value] of Object.entries(params)) {
|
|
29
|
+
if (Array.isArray(value)) for (const item of value) url.searchParams.append(name, item);
|
|
30
|
+
else url.searchParams.set(name, String(value));
|
|
16
31
|
}
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
32
|
+
const usePost = input.goggles != null || url.toString().length > 2_000;
|
|
33
|
+
const data = await requestJson<any>(usePost ? `${url.origin}${url.pathname}` : url.toString(), {
|
|
34
|
+
method: usePost ? "POST" : undefined,
|
|
35
|
+
headers: { "X-Subscription-Token": this.key, accept: "application/json", ...(usePost ? { "content-type": "application/json" } : {}) },
|
|
36
|
+
body: usePost ? JSON.stringify(params) : undefined,
|
|
37
|
+
signal,
|
|
38
|
+
timeoutMs: 30000,
|
|
39
|
+
});
|
|
23
40
|
const sources = data.sources ?? data.web?.results ?? [];
|
|
24
41
|
const snippets = data.grounding?.generic ?? [];
|
|
25
|
-
const
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
42
|
+
const sourceRows = Array.isArray(sources)
|
|
43
|
+
? sources
|
|
44
|
+
: Object.entries(sources).map(([url, source]: [string, any]) => ({ ...source, url }));
|
|
45
|
+
const rows = snippets.length ? snippets : sourceRows;
|
|
46
|
+
return { provider: "brave" as const, query: input.query, results: rows.map((r: any, i: number) => {
|
|
47
|
+
const content = asText(r.snippets ?? r.snippet ?? r.description ?? snippets[i]?.snippets);
|
|
48
|
+
return {
|
|
49
|
+
title: r.title ?? r.name, url: r.url, snippet: asSnippet(content), content, contentFormat: content ? "markdown" as const : undefined, siteName: r.site_name ?? r.source, position: i + 1,
|
|
50
|
+
};
|
|
51
|
+
}).filter((r: any) => r.url) };
|
|
29
52
|
}
|
|
30
53
|
}
|
|
54
|
+
|
|
55
|
+
function boundedQuery(query: string): string {
|
|
56
|
+
return query.trim().split(/\s+/).slice(0, 50).join(" ").slice(0, 400);
|
|
57
|
+
}
|
package/src/providers/exa.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { asSnippet, normalizeUrls, requestJson } from "../http.js";
|
|
1
|
+
import { asSnippet, asText, normalizeUrls, requestJson, withoutContent } from "../http.js";
|
|
2
2
|
import { DEFAULT_NUM_RESULTS } from "../limits.js";
|
|
3
3
|
import { urlsMatch } from "../urls.js";
|
|
4
4
|
import type { ExaCodeInput, ExaCodeResult, FetchInput, FetchProvider, SearchInput, SearchProvider, WebFetchResult, WebKitConfig } from "../types.js";
|
|
@@ -14,10 +14,15 @@ export class ExaProvider implements SearchProvider, FetchProvider {
|
|
|
14
14
|
}
|
|
15
15
|
|
|
16
16
|
async search(input: SearchInput, signal?: AbortSignal) {
|
|
17
|
+
const textCharacters = input.contextTokens ? Math.min(100_000, Math.max(1, Math.floor(input.contextTokens * 4 / (input.numResults ?? DEFAULT_NUM_RESULTS)))) : undefined;
|
|
17
18
|
const body = {
|
|
18
19
|
query: input.query,
|
|
19
20
|
numResults: input.numResults ?? DEFAULT_NUM_RESULTS,
|
|
20
|
-
contents: input.contents ?? {
|
|
21
|
+
contents: input.contents ?? {
|
|
22
|
+
highlights: input.purpose ? { query: input.purpose } : true,
|
|
23
|
+
...(textCharacters ? { text: { maxCharacters: textCharacters } } : {}),
|
|
24
|
+
...(typeof input.maxAgeHours === "number" ? { maxAgeHours: input.maxAgeHours } : {}),
|
|
25
|
+
},
|
|
21
26
|
includeDomains: input.includeDomains,
|
|
22
27
|
excludeDomains: input.excludeDomains,
|
|
23
28
|
startPublishedDate: input.startPublishedDate,
|
|
@@ -37,13 +42,18 @@ export class ExaProvider implements SearchProvider, FetchProvider {
|
|
|
37
42
|
return {
|
|
38
43
|
provider: "exa" as const,
|
|
39
44
|
query: input.query,
|
|
40
|
-
results: (data.results ?? []).map((r: any, i: number) =>
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
45
|
+
results: (data.results ?? []).map((r: any, i: number) => {
|
|
46
|
+
const content = asText(r.text ?? r.highlights ?? r.summary);
|
|
47
|
+
return {
|
|
48
|
+
title: r.title,
|
|
49
|
+
url: r.url,
|
|
50
|
+
snippet: asSnippet(content),
|
|
51
|
+
content,
|
|
52
|
+
contentFormat: content ? "markdown" as const : undefined,
|
|
53
|
+
siteName: r.author ?? r.publishedDate,
|
|
54
|
+
position: i + 1,
|
|
55
|
+
};
|
|
56
|
+
}).filter((r: any) => r.url),
|
|
47
57
|
};
|
|
48
58
|
}
|
|
49
59
|
|
|
@@ -53,7 +63,14 @@ export class ExaProvider implements SearchProvider, FetchProvider {
|
|
|
53
63
|
const data = await requestJson<any>("https://api.exa.ai/contents", {
|
|
54
64
|
method: "POST",
|
|
55
65
|
headers: this.headers(),
|
|
56
|
-
body: JSON.stringify({
|
|
66
|
+
body: JSON.stringify({
|
|
67
|
+
urls,
|
|
68
|
+
text: { maxCharacters: 100_000 },
|
|
69
|
+
highlights: false,
|
|
70
|
+
maxAgeHours: input.refresh === true
|
|
71
|
+
? 0
|
|
72
|
+
: input.maxAgeHours ?? (typeof input.maxAgeMs === "number" ? Math.floor(input.maxAgeMs / 3_600_000) : undefined),
|
|
73
|
+
}),
|
|
57
74
|
signal,
|
|
58
75
|
timeoutMs: 45000,
|
|
59
76
|
});
|
|
@@ -61,7 +78,7 @@ export class ExaProvider implements SearchProvider, FetchProvider {
|
|
|
61
78
|
const primary: WebFetchResult = { provider: "exa", results: urls.map((url, i) => {
|
|
62
79
|
const r: any = list.find((item: any) => urlsMatch(item.url, url)) ?? list[i];
|
|
63
80
|
if (!r) return { url, error: "No content returned by Exa contents endpoint." };
|
|
64
|
-
return { url: r.url ?? url, title: r.title, content: r.text ?? r.summary ?? "", format: "markdown" as const, metadata: r };
|
|
81
|
+
return { url: r.url ?? url, title: r.title, content: r.text ?? r.summary ?? "", format: "markdown" as const, metadata: withoutContent(r) };
|
|
65
82
|
}) };
|
|
66
83
|
return applyExaFetchFallbacks(this.config, input, urls, primary, signal);
|
|
67
84
|
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { asSnippet, mapConcurrent, normalizeUrls, requestJson } from "../http.js";
|
|
1
|
+
import { asSnippet, asText, mapConcurrent, normalizeUrls, requestJson } from "../http.js";
|
|
2
2
|
import { DEFAULT_NUM_RESULTS, FETCH_CONCURRENCY } from "../limits.js";
|
|
3
3
|
import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
|
|
4
4
|
import { requireKey } from "../config.js";
|
|
@@ -20,13 +20,22 @@ export class FirecrawlProvider implements SearchProvider, FetchProvider {
|
|
|
20
20
|
excludeDomains: input.excludeDomains,
|
|
21
21
|
categories: input.categories,
|
|
22
22
|
tbs: input.tbs,
|
|
23
|
-
scrapeOptions: input.scrapeOptions ?? (input.scrape === true ? {
|
|
23
|
+
scrapeOptions: input.scrapeOptions ?? (input.scrape === true || input.contextTokens ? {
|
|
24
|
+
formats: ["markdown"],
|
|
25
|
+
onlyMainContent: true,
|
|
26
|
+
maxAge: input.maxAge,
|
|
27
|
+
} : undefined),
|
|
24
28
|
}),
|
|
25
29
|
});
|
|
26
30
|
const list = data.data?.web ?? data.web ?? data.data ?? [];
|
|
27
|
-
return { provider: "firecrawl" as const, query: input.query, results: list.map((r: any, i: number) =>
|
|
28
|
-
|
|
29
|
-
|
|
31
|
+
return { provider: "firecrawl" as const, query: input.query, results: list.map((r: any, i: number) => {
|
|
32
|
+
const content = asText(r.markdown ?? r.content ?? r.highlights);
|
|
33
|
+
return {
|
|
34
|
+
title: r.title, url: r.url, snippet: asSnippet(r.highlights ?? r.description ?? content), content,
|
|
35
|
+
contentFormat: content ? (r.markdown ? "markdown" as const : "text" as const) : undefined,
|
|
36
|
+
siteName: r.siteName, position: i + 1,
|
|
37
|
+
};
|
|
38
|
+
}).filter((r: any) => r.url) };
|
|
30
39
|
}
|
|
31
40
|
|
|
32
41
|
async fetch(input: FetchInput, signal?: AbortSignal) {
|
|
@@ -44,7 +53,7 @@ export class FirecrawlProvider implements SearchProvider, FetchProvider {
|
|
|
44
53
|
waitFor: input.waitFor,
|
|
45
54
|
mobile: input.mobile,
|
|
46
55
|
location: input.location,
|
|
47
|
-
maxAge: input.maxAge,
|
|
56
|
+
maxAge: input.refresh === true ? 0 : input.maxAge ?? input.maxAgeMs,
|
|
48
57
|
}),
|
|
49
58
|
});
|
|
50
59
|
const d = data.data ?? data;
|
package/src/providers/index.ts
CHANGED
|
@@ -2,7 +2,6 @@ import { validateFetchProvider, validateSearchProvider } from "../config.js";
|
|
|
2
2
|
import type { FetchProvider, SearchProvider, WebKitConfig } from "../types.js";
|
|
3
3
|
import { BraveProvider } from "./brave.js";
|
|
4
4
|
import { Context7Provider } from "./context7.js";
|
|
5
|
-
import { ExaMcpProvider } from "./exa-mcp.js";
|
|
6
5
|
import { ExaProvider } from "./exa.js";
|
|
7
6
|
import { FirecrawlProvider } from "./firecrawl.js";
|
|
8
7
|
import { MarkdownNewProvider } from "./markdown-new.js";
|
|
@@ -11,7 +10,6 @@ import { TinyFishProvider } from "./tinyfish.js";
|
|
|
11
10
|
export function createSearchProvider(config: WebKitConfig): SearchProvider {
|
|
12
11
|
validateSearchProvider(config.provider_search);
|
|
13
12
|
switch (config.provider_search) {
|
|
14
|
-
case "exa_mcp": return new ExaMcpProvider(config);
|
|
15
13
|
case "exa": return new ExaProvider(config);
|
|
16
14
|
case "tinyfish": return new TinyFishProvider(config);
|
|
17
15
|
case "brave": return new BraveProvider(config);
|
|
@@ -22,7 +20,6 @@ export function createSearchProvider(config: WebKitConfig): SearchProvider {
|
|
|
22
20
|
export function createFetchProvider(config: WebKitConfig): FetchProvider {
|
|
23
21
|
validateFetchProvider(config.provider_fetch);
|
|
24
22
|
switch (config.provider_fetch) {
|
|
25
|
-
case "exa_mcp": return new ExaMcpProvider(config);
|
|
26
23
|
case "exa": return new ExaProvider(config);
|
|
27
24
|
case "tinyfish": return new TinyFishProvider(config);
|
|
28
25
|
case "markdown_new": return new MarkdownNewProvider(config);
|
|
@@ -15,14 +15,22 @@ export class MarkdownNewProvider implements FetchProvider {
|
|
|
15
15
|
body: JSON.stringify({
|
|
16
16
|
url,
|
|
17
17
|
method: input.method ?? this.config.markdownNew.method,
|
|
18
|
-
|
|
18
|
+
retain_images: input.retainImages ?? this.config.markdownNew.retainImages,
|
|
19
19
|
}),
|
|
20
20
|
signal,
|
|
21
21
|
timeoutMs: 45_000,
|
|
22
22
|
});
|
|
23
23
|
const text = await res.text();
|
|
24
24
|
if (!res.ok) throw new Error(`${res.status} ${res.statusText}: ${text.slice(0, 1000)}`);
|
|
25
|
-
return {
|
|
25
|
+
return {
|
|
26
|
+
url,
|
|
27
|
+
content: text,
|
|
28
|
+
format: "markdown" as const,
|
|
29
|
+
metadata: {
|
|
30
|
+
estimatedTokens: numberHeader(res.headers.get("x-markdown-tokens")),
|
|
31
|
+
conversionMethod: res.headers.get("x-markdown-method") ?? undefined,
|
|
32
|
+
},
|
|
33
|
+
};
|
|
26
34
|
} catch (e) {
|
|
27
35
|
return { url, error: e instanceof Error ? e.message : String(e) };
|
|
28
36
|
}
|
|
@@ -30,3 +38,9 @@ export class MarkdownNewProvider implements FetchProvider {
|
|
|
30
38
|
return { provider: "markdown_new" as const, results };
|
|
31
39
|
}
|
|
32
40
|
}
|
|
41
|
+
|
|
42
|
+
function numberHeader(value: string | null): number | undefined {
|
|
43
|
+
if (value == null) return undefined;
|
|
44
|
+
const parsed = Number(value);
|
|
45
|
+
return Number.isFinite(parsed) ? parsed : undefined;
|
|
46
|
+
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { asSnippet, normalizeUrls, requestJson } from "../http.js";
|
|
2
|
-
import { MAX_URL_COUNT } from "../limits.js";
|
|
3
|
-
import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
|
|
1
|
+
import { asSnippet, normalizeUrls, requestJson, withoutContent } from "../http.js";
|
|
2
|
+
import { DEFAULT_NUM_RESULTS, MAX_URL_COUNT, TINYFISH_MAX_PAGE } from "../limits.js";
|
|
3
|
+
import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig, WebSearchResult } from "../types.js";
|
|
4
|
+
import { canonicalWebUrl, urlsMatch } from "../urls.js";
|
|
4
5
|
import { requireKey } from "../config.js";
|
|
5
6
|
|
|
6
7
|
export class TinyFishProvider implements SearchProvider, FetchProvider {
|
|
@@ -8,15 +9,81 @@ export class TinyFishProvider implements SearchProvider, FetchProvider {
|
|
|
8
9
|
constructor(config: WebKitConfig) { this.key = requireKey(config, "tinyfish"); }
|
|
9
10
|
|
|
10
11
|
async search(input: SearchInput, signal?: AbortSignal) {
|
|
11
|
-
const
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
const
|
|
16
|
-
const
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
12
|
+
const desired = input.numResults ?? DEFAULT_NUM_RESULTS;
|
|
13
|
+
const firstPage = typeof input.page === "number" ? Math.max(0, Math.min(input.page, TINYFISH_MAX_PAGE)) : 0;
|
|
14
|
+
const research = input.domainType === "research_paper";
|
|
15
|
+
const dates = orderedStrings(input.afterDate, input.beforeDate);
|
|
16
|
+
const years = orderedNumbers(input.pubYearMin, input.pubYearMax);
|
|
17
|
+
const results: WebSearchResult["results"] = [];
|
|
18
|
+
const seen = new Set<string>();
|
|
19
|
+
let lastPage = firstPage - 1;
|
|
20
|
+
|
|
21
|
+
for (let page = firstPage; page <= TINYFISH_MAX_PAGE && results.length < desired; page++) {
|
|
22
|
+
lastPage = page;
|
|
23
|
+
const url = new URL("https://api.search.tinyfish.ai/");
|
|
24
|
+
url.searchParams.set("query", input.query);
|
|
25
|
+
url.searchParams.set("page", String(page));
|
|
26
|
+
setString(url, "purpose", input.purpose);
|
|
27
|
+
setString(url, "location", input.location);
|
|
28
|
+
setString(url, "language", input.language);
|
|
29
|
+
setDomains(url, "include_domains", input.includeDomains);
|
|
30
|
+
setDomains(url, "exclude_domains", input.excludeDomains);
|
|
31
|
+
setString(url, "domain_type", input.domainType);
|
|
32
|
+
if (research) {
|
|
33
|
+
setNumber(url, "pub_year_min", years[0]);
|
|
34
|
+
setNumber(url, "pub_year_max", years[1]);
|
|
35
|
+
} else if (typeof input.recencyMinutes === "number") {
|
|
36
|
+
setNumber(url, "recency_minutes", input.recencyMinutes);
|
|
37
|
+
} else {
|
|
38
|
+
setString(url, "after_date", dates[0]);
|
|
39
|
+
setString(url, "before_date", dates[1]);
|
|
40
|
+
}
|
|
41
|
+
const data = await requestJson<any>(url.toString(), { headers: { "X-API-Key": this.key }, signal, timeoutMs: 10000 });
|
|
42
|
+
const list = data.results ?? data.data ?? data.web ?? [];
|
|
43
|
+
if (!Array.isArray(list) || list.length === 0) break;
|
|
44
|
+
for (const r of list) {
|
|
45
|
+
const resultUrl = r.url ?? r.link;
|
|
46
|
+
const key = typeof resultUrl === "string" ? canonicalResultUrl(resultUrl) : undefined;
|
|
47
|
+
if (!key || seen.has(key)) continue;
|
|
48
|
+
seen.add(key);
|
|
49
|
+
results.push({
|
|
50
|
+
title: r.title,
|
|
51
|
+
url: resultUrl,
|
|
52
|
+
snippet: asSnippet(r.snippet ?? r.description ?? r.text),
|
|
53
|
+
siteName: r.siteName ?? r.source,
|
|
54
|
+
position: results.length + 1,
|
|
55
|
+
});
|
|
56
|
+
if (results.length === desired) break;
|
|
57
|
+
}
|
|
58
|
+
if (typeof data.total_results === "number" && results.length >= data.total_results) break;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
if (input.contextTokens && results.length) {
|
|
62
|
+
const contextCharacters = input.contextTokens * 4;
|
|
63
|
+
let extractedCharacters = 0;
|
|
64
|
+
for (let i = 0; i < results.length && extractedCharacters < contextCharacters;) {
|
|
65
|
+
// ponytail: budget for about 512 useful tokens per page; fetch another batch when pages are short.
|
|
66
|
+
const batchSize = Math.min(MAX_URL_COUNT, Math.max(1, Math.ceil((contextCharacters - extractedCharacters) / 2_048)));
|
|
67
|
+
const batch = results.slice(i, i + batchSize);
|
|
68
|
+
const fetched = await this.fetch({ urls: batch.map((item) => item.url), format: "markdown", purpose: input.purpose }, signal);
|
|
69
|
+
for (const item of batch) {
|
|
70
|
+
const page = fetched.results.find((candidate) => urlsMatch(candidate.url, item.url) || urlsMatch(candidate.metadata?.requestedUrl, item.url));
|
|
71
|
+
if (page?.content) {
|
|
72
|
+
item.content = page.content;
|
|
73
|
+
item.contentFormat = "markdown";
|
|
74
|
+
extractedCharacters += page.content.length;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
i += batch.length;
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
return {
|
|
82
|
+
provider: "tinyfish" as const,
|
|
83
|
+
query: input.query,
|
|
84
|
+
effectiveResultLimit: lastPage === TINYFISH_MAX_PAGE && results.length < desired ? results.length : desired,
|
|
85
|
+
results,
|
|
86
|
+
};
|
|
20
87
|
}
|
|
21
88
|
|
|
22
89
|
async fetch(input: FetchInput, signal?: AbortSignal) {
|
|
@@ -25,20 +92,73 @@ export class TinyFishProvider implements SearchProvider, FetchProvider {
|
|
|
25
92
|
const data = await requestJson<any>("https://api.fetch.tinyfish.ai", {
|
|
26
93
|
method: "POST",
|
|
27
94
|
headers: { "content-type": "application/json", "X-API-Key": this.key },
|
|
28
|
-
body: JSON.stringify({
|
|
95
|
+
body: JSON.stringify({
|
|
96
|
+
urls,
|
|
97
|
+
purpose: input.purpose,
|
|
98
|
+
format: input.format ?? "markdown",
|
|
99
|
+
links: input.links,
|
|
100
|
+
image_links: input.imageLinks,
|
|
101
|
+
ttl: input.refresh === true
|
|
102
|
+
? 0
|
|
103
|
+
: input.ttl ?? (typeof input.maxAgeMs === "number" ? Math.floor(input.maxAgeMs / 1_000) : undefined),
|
|
104
|
+
per_url_timeout_ms: input.perUrlTimeoutMs,
|
|
105
|
+
include_selectors: input.includeSelectors,
|
|
106
|
+
exclude_selectors: input.excludeSelectors,
|
|
107
|
+
}),
|
|
29
108
|
signal,
|
|
30
109
|
timeoutMs: 150000,
|
|
31
110
|
});
|
|
32
111
|
const list = data.results ?? data.data ?? data.pages ?? [];
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
112
|
+
const errors = data.errors ?? [];
|
|
113
|
+
return { provider: "tinyfish" as const, results: urls.map((url) => {
|
|
114
|
+
const r = list.find((x: any) => urlsMatch(x.url ?? x.source_url, url));
|
|
115
|
+
const failure = errors.find((x: any) => urlsMatch(x.url, url));
|
|
116
|
+
if (!r) return { url, error: failure ? `${failure.error}${failure.status ? ` (${failure.status})` : ""}` : "No content returned by TinyFish." };
|
|
36
117
|
const content = r.text ?? r.content ?? r.markdown ?? r.html;
|
|
37
|
-
return {
|
|
118
|
+
return {
|
|
119
|
+
url: r.final_url ?? r.url ?? url,
|
|
120
|
+
content: stringifyContent(content),
|
|
121
|
+
format: input.format ?? r.format ?? "markdown",
|
|
122
|
+
title: r.title,
|
|
123
|
+
metadata: { ...withoutContent(r), requestedUrl: url, finalUrl: r.final_url },
|
|
124
|
+
error: r.error,
|
|
125
|
+
};
|
|
38
126
|
}) };
|
|
39
127
|
}
|
|
40
128
|
}
|
|
41
129
|
|
|
130
|
+
function setString(url: URL, name: string, value: unknown) {
|
|
131
|
+
if (typeof value === "string" && value) url.searchParams.set(name, value);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function setNumber(url: URL, name: string, value: unknown) {
|
|
135
|
+
if (typeof value === "number") url.searchParams.set(name, String(value));
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function setDomains(url: URL, name: string, value: unknown) {
|
|
139
|
+
if (Array.isArray(value) && value.length) url.searchParams.set(name, value.join(","));
|
|
140
|
+
else setString(url, name, value);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
function orderedStrings(a: unknown, b: unknown): [string | undefined, string | undefined] {
|
|
144
|
+
const values = [a, b].filter((value): value is string => typeof value === "string").sort();
|
|
145
|
+
return [values[0], values[1]];
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
function orderedNumbers(a: unknown, b: unknown): [number | undefined, number | undefined] {
|
|
149
|
+
const values = [a, b].filter((value): value is number => typeof value === "number").sort((x, y) => x - y);
|
|
150
|
+
return [values[0], values[1]];
|
|
151
|
+
}
|
|
152
|
+
|
|
42
153
|
function stringifyContent(value: unknown): string | undefined {
|
|
43
154
|
return typeof value === "string" ? value : JSON.stringify(value, null, 2);
|
|
44
155
|
}
|
|
156
|
+
|
|
157
|
+
function canonicalResultUrl(value: string): string {
|
|
158
|
+
try {
|
|
159
|
+
const canonical = canonicalWebUrl(value);
|
|
160
|
+
return canonical.endsWith("/") ? canonical.slice(0, -1) : canonical;
|
|
161
|
+
} catch {
|
|
162
|
+
return value;
|
|
163
|
+
}
|
|
164
|
+
}
|
package/src/types.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
export type SearchProviderName = "
|
|
2
|
-
export type FetchProviderName = "
|
|
1
|
+
export type SearchProviderName = "exa" | "tinyfish" | "brave" | "firecrawl";
|
|
2
|
+
export type FetchProviderName = "exa" | "tinyfish" | "markdown_new" | "firecrawl";
|
|
3
3
|
export type FetchFormat = "markdown" | "html" | "json";
|
|
4
4
|
|
|
5
5
|
export interface WebKitConfig {
|
|
@@ -12,13 +12,16 @@ export interface WebKitConfig {
|
|
|
12
12
|
export interface SearchInput {
|
|
13
13
|
query: string;
|
|
14
14
|
numResults?: number;
|
|
15
|
+
contextTokens?: number;
|
|
16
|
+
purpose?: string;
|
|
15
17
|
[key: string]: unknown;
|
|
16
18
|
}
|
|
17
19
|
|
|
18
20
|
export interface WebSearchResult {
|
|
19
21
|
provider: SearchProviderName;
|
|
20
22
|
query: string;
|
|
21
|
-
|
|
23
|
+
effectiveResultLimit?: number;
|
|
24
|
+
results: Array<{ title?: string; url: string; snippet?: string; content?: string; contentFormat?: "markdown" | "text"; siteName?: string; position?: number }>;
|
|
22
25
|
}
|
|
23
26
|
|
|
24
27
|
export interface FetchInput {
|
|
@@ -27,9 +30,15 @@ export interface FetchInput {
|
|
|
27
30
|
offset?: number;
|
|
28
31
|
limit?: number;
|
|
29
32
|
refresh?: boolean;
|
|
33
|
+
maxAgeMs?: number;
|
|
30
34
|
format?: FetchFormat;
|
|
31
35
|
links?: boolean;
|
|
32
36
|
imageLinks?: boolean;
|
|
37
|
+
purpose?: string;
|
|
38
|
+
ttl?: number;
|
|
39
|
+
perUrlTimeoutMs?: number;
|
|
40
|
+
includeSelectors?: string[];
|
|
41
|
+
excludeSelectors?: string[];
|
|
33
42
|
[key: string]: unknown;
|
|
34
43
|
}
|
|
35
44
|
|
package/src/providers/exa-mcp.ts
DELETED
|
@@ -1,102 +0,0 @@
|
|
|
1
|
-
import { asSnippet, fetchWithTimeout, normalizeUrls } from "../http.js";
|
|
2
|
-
import { DEFAULT_NUM_RESULTS } from "../limits.js";
|
|
3
|
-
import { urlsMatch } from "../urls.js";
|
|
4
|
-
import type { FetchInput, FetchProvider, SearchInput, SearchProvider, WebKitConfig } from "../types.js";
|
|
5
|
-
import { applyExaFetchFallbacks } from "./fallback.js";
|
|
6
|
-
|
|
7
|
-
export class ExaMcpProvider implements SearchProvider, FetchProvider {
|
|
8
|
-
private sessionId?: string;
|
|
9
|
-
private nextId = 1;
|
|
10
|
-
constructor(private config: WebKitConfig) {}
|
|
11
|
-
|
|
12
|
-
async search(input: SearchInput, signal?: AbortSignal) {
|
|
13
|
-
const result = await this.callTool("web_search_exa", { query: input.query, numResults: input.numResults ?? DEFAULT_NUM_RESULTS }, signal);
|
|
14
|
-
return { provider: "exa_mcp" as const, query: input.query, results: normalizeSearch(result) };
|
|
15
|
-
}
|
|
16
|
-
|
|
17
|
-
async fetch(input: FetchInput, signal?: AbortSignal) {
|
|
18
|
-
const urls = normalizeUrls(input);
|
|
19
|
-
const result = await this.callTool("web_fetch_exa", { urls }, signal);
|
|
20
|
-
const pages = normalizeFetch(result, urls);
|
|
21
|
-
return applyExaFetchFallbacks(this.config, input, urls, { provider: "exa_mcp" as const, results: pages }, signal);
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
private async callTool(name: string, args: Record<string, unknown>, signal?: AbortSignal): Promise<any> {
|
|
25
|
-
await this.initialize(signal);
|
|
26
|
-
return this.rpc("tools/call", { name, arguments: args }, signal);
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
private async initialize(signal?: AbortSignal) {
|
|
30
|
-
if (this.sessionId) return;
|
|
31
|
-
const data = await this.rpcRaw("initialize", { protocolVersion: "2025-06-18", capabilities: {}, clientInfo: { name: "pi-web-kit", version: "0.1.0" } }, signal);
|
|
32
|
-
this.sessionId = data.sessionId ?? data.payload?.sessionId;
|
|
33
|
-
if (!this.sessionId) throw new Error("Exa MCP initialize did not return an mcp-session-id response header.");
|
|
34
|
-
await this.rpcRaw("notifications/initialized", undefined, signal, true);
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
private async rpc(method: string, params: unknown, signal?: AbortSignal) {
|
|
38
|
-
const data = await this.rpcRaw(method, params, signal);
|
|
39
|
-
if (data.payload?.error) throw new Error(data.payload.error.message ?? JSON.stringify(data.payload.error));
|
|
40
|
-
return data.payload?.result ?? data.payload;
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
private async rpcRaw(method: string, params: unknown, signal?: AbortSignal, notification = false): Promise<{ payload?: any; sessionId?: string }> {
|
|
44
|
-
const headers: Record<string, string> = { "content-type": "application/json", accept: "application/json, text/event-stream" };
|
|
45
|
-
if (this.sessionId) headers["mcp-session-id"] = this.sessionId;
|
|
46
|
-
if (this.config.apiKeys.exa) headers["x-api-key"] = this.config.apiKeys.exa;
|
|
47
|
-
const body = notification ? { jsonrpc: "2.0", method, params } : { jsonrpc: "2.0", id: this.nextId++, method, params };
|
|
48
|
-
const res = await fetchWithTimeout("https://mcp.exa.ai/mcp", { method: "POST", headers, body: JSON.stringify(body), signal, timeoutMs: 45_000 });
|
|
49
|
-
const text = await res.text();
|
|
50
|
-
if (!res.ok) throw new Error(`Exa MCP ${method} failed: ${res.status} ${res.statusText}: ${text.slice(0, 1000)}`);
|
|
51
|
-
if (notification) return { sessionId: this.sessionId };
|
|
52
|
-
const sessionId = res.headers.get("mcp-session-id") ?? undefined;
|
|
53
|
-
const payload = parseMcpResponse(text, res.headers.get("content-type") ?? "");
|
|
54
|
-
return { payload, sessionId };
|
|
55
|
-
}
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
function parseMcpResponse(text: string, contentType: string): any {
|
|
59
|
-
if (!text.trim()) return undefined;
|
|
60
|
-
if (contentType.includes("text/event-stream") || text.startsWith("event:") || text.startsWith("data:")) {
|
|
61
|
-
const data = text.split(/\r?\n/).filter((line) => line.startsWith("data:")).map((line) => line.slice(5).trim()).join("\n");
|
|
62
|
-
return data ? JSON.parse(data) : undefined;
|
|
63
|
-
}
|
|
64
|
-
return JSON.parse(text);
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
function textFromContent(result: any): string {
|
|
68
|
-
const content = result?.content ?? result;
|
|
69
|
-
if (Array.isArray(content)) return content.map((c) => typeof c === "string" ? c : c.text ?? JSON.stringify(c)).join("\n");
|
|
70
|
-
return typeof content === "string" ? content : JSON.stringify(content);
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
function normalizeSearch(result: any) {
|
|
74
|
-
const structured = result?.structuredContent ?? result?.result ?? result;
|
|
75
|
-
const list = structured.results ?? structured.data ?? structured.items;
|
|
76
|
-
if (Array.isArray(list)) return list.map(toSearchResult).filter((r: any) => r.url);
|
|
77
|
-
const text = textFromContent(result);
|
|
78
|
-
const urls = [...text.matchAll(/https?:\/\/[^\s)\]}>"']+/g)].map((m) => m[0]);
|
|
79
|
-
return [...new Set(urls)].map((url, i) => ({ url, snippet: i === 0 ? text.slice(0, 1000) : undefined, position: i + 1 }));
|
|
80
|
-
}
|
|
81
|
-
|
|
82
|
-
function toSearchResult(r: any, index: number) {
|
|
83
|
-
return {
|
|
84
|
-
title: r.title,
|
|
85
|
-
url: r.url,
|
|
86
|
-
snippet: asSnippet(r.snippet ?? r.text ?? r.summary ?? r.highlights),
|
|
87
|
-
siteName: r.siteName,
|
|
88
|
-
position: index + 1,
|
|
89
|
-
};
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
function normalizeFetch(result: any, urls: string[]) {
|
|
93
|
-
const structured = result?.structuredContent ?? result?.result ?? result;
|
|
94
|
-
const list = structured.results ?? structured.data ?? structured.pages;
|
|
95
|
-
if (Array.isArray(list)) return urls.map((url, i) => {
|
|
96
|
-
const r = list.find((x: any) => urlsMatch(x.url, url)) ?? list[i];
|
|
97
|
-
if (!r) return { url, error: "No content returned by Exa MCP." };
|
|
98
|
-
return { url, title: r.title, content: r.markdown ?? r.text ?? r.content ?? r.html, format: "markdown" as const, metadata: r, error: r.error };
|
|
99
|
-
});
|
|
100
|
-
const text = textFromContent(result);
|
|
101
|
-
return urls.length <= 1 ? [{ url: urls[0] ?? "", content: text, format: "markdown" as const }] : urls.map((url) => ({ url, content: text, format: "markdown" as const }));
|
|
102
|
-
}
|