pi-web-kit 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,13 +1,41 @@
1
1
  # Changelog
2
2
 
3
- ## 0.2.3 - 2026-07-17
4
-
5
3
  All notable changes to this project will be documented in this file.
6
4
 
7
5
  This project follows the spirit of [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and uses semantic versioning for releases.
8
6
 
9
7
  ## [Unreleased]
10
8
 
9
+ ## [0.3.0] - 2026-08-31
10
+
11
+ ### Changed
12
+
13
+ - Share install telemetry mechanics through `@mocito/install-telemetry` while preserving Pi-specific settings and state paths.
14
+ - Treat `web_search.numResults` as a provider-agnostic desired limit, cap it to each provider's service constraints, and report requested, effective, returned, and omitted counts.
15
+ - Add provider-agnostic `contextTokens` and `purpose` search controls, preserve provider-native grounding content, and allocate one bounded context budget across ranked results.
16
+ - Use adaptive extraction: native Exa/Brave context is retained automatically, while TinyFish fetch and Firecrawl scrape-on-search run only when expanded context is requested.
17
+ - Expose documented TinyFish search/fetch controls and Brave LLM Context controls.
18
+ - Add provider-agnostic `maxAgeMs`, pass refresh/cache intent through to Exa, TinyFish, and Firecrawl, and request complete bounded pages from Exa fetch.
19
+
20
+ ### Fixed
21
+
22
+ - Let `enableInstallTelemetry: false` override an enabled `PI_TELEMETRY` environment flag.
23
+ - Paginate TinyFish searches instead of sending its ignored `limit` parameter.
24
+ - Send markdown.new's documented `retain_images` field and report its response metadata.
25
+ - Match TinyFish fetch responses and per-URL errors by canonical URL instead of response position.
26
+
27
+ ### Removed
28
+
29
+ - Remove the Exa MCP search/fetch surface and make the direct Exa API the default. Existing `exa_mcp` configurations must select `exa` and provide `EXA_API_KEY`.
30
+
31
+ ## [0.2.4] - 2026-07-28
32
+
33
+ ### Changed
34
+
35
+ - Rewrite `library_search`, `library_docs`, `web_search`, and `code_search` prompts so `library_docs` is the default for versioned library/framework/SDK docs (even familiar libraries), `library_search` is reserved for inspecting candidate matches, and `web_search`/`code_search` defer to it. Removes the "ambiguous library" wording that suppressed calls to well-known libraries. The `web_search`/`code_search` deferrals are emitted only when Context7 is configured, so installations without `CONTEXT7_API_KEY` are unaffected. Provider behavior and tool schemas are unchanged.
36
+
37
+ ## [0.2.3] - 2026-07-17
38
+
11
39
  ### Fixed
12
40
 
13
41
  - Honor project-local config only after Pi confirms project trust.
@@ -24,6 +52,10 @@ This project follows the spirit of [Keep a Changelog](https://keepachangelog.com
24
52
 
25
53
  ## [0.2.1] - 2026-06-14
26
54
 
55
+ ### Changed
56
+
57
+ - Simplify provider, configuration, cache, and URL handling while preserving the public tool surface.
58
+
27
59
  ## [0.2.0] - 2026-06-13
28
60
 
29
61
  ### Added
package/README.md CHANGED
@@ -1,20 +1,16 @@
1
1
  # pi-web-kit
2
2
 
3
- Context-efficient web and developer search tools for [Pi](https://pi.dev): `web_search`, `web_fetch`, `library_search`, `library_docs`, and `code_search`.
3
+ Give [Pi](https://pi.dev) current web knowledge, authoritative library docs, and real-world code examples without flooding model context.
4
4
 
5
- `pi-web-kit` provides provider-backed search, page fetching, library docs lookup, and code-context search with bounded output, chunked reads, URL validation, and an in-memory fetch cache designed for agent workflows.
5
+ `pi-web-kit` combines search, page reading, version-aware documentation, and code research behind five agent-ready tools with bounded, cache-aware output.
6
6
 
7
7
  ## Features
8
8
 
9
- - `web_search` for current/external web information, including multi-query searches.
10
- - `web_fetch` for reading one or more URLs, with `offset` / `limit` chunk reads for long pages.
11
- - `library_search` and `library_docs` for library resolution and current, version-aware documentation/code examples.
12
- - `code_search` for practical examples and implementation context.
13
- - Multiple provider backends: Exa MCP, Exa API, TinyFish, Brave Search, Firecrawl, markdown.new, Context7, and Exa Code.
14
- - Provider-tailored tool schemas at Pi startup/reload.
15
- - URL validation: HTTP(S)-only, no embedded credentials, fragment stripping, duplicate removal, and length/count limits.
16
- - In-memory fetch cache with TTL, LRU eviction, max entry count, max byte count, and cache keys based on provider/config/fetch-affecting options.
17
- - Bounded provider concurrency and network timeouts.
9
+ - **Research the live web** — search current information and read single or multiple pages without leaving Pi.
10
+ - **Use docs that match the task** — resolve libraries and retrieve focused, version-aware documentation with code examples.
11
+ - **Find proven implementation patterns** — search practical usage, setup, migrations, and error context across real code.
12
+ - **Spend context wisely** — compact search results, chunked page reads, bounded output, and fetch caching keep research useful without overwhelming the model.
13
+ - **Choose your providers** — mix Exa, TinyFish, Brave, Firecrawl, markdown.new, Context7, and Exa Code based on coverage, cost, and credentials.
18
14
 
19
15
  ## Installation
20
16
 
@@ -74,11 +70,10 @@ Pi chooses `web_search` or `web_fetch` automatically when the request calls for
74
70
 
75
71
  ## Providers
76
72
 
77
- Defaults: `provider_search = "exa_mcp"`, `provider_fetch = "exa_mcp"`.
73
+ Defaults: `provider_search = "exa"`, `provider_fetch = "exa"`.
78
74
 
79
75
  | Provider | Search | Fetch | Key |
80
76
  |---|---:|---:|---|
81
- | `exa_mcp` | yes | yes | optional `EXA_API_KEY` |
82
77
  | `exa` | yes | yes | `EXA_API_KEY` |
83
78
  | `tinyfish` | yes | yes | `TINYFISH_API_KEY` |
84
79
  | `brave` | yes | no | `BRAVE_SEARCH_API_KEY` |
@@ -87,6 +82,18 @@ Defaults: `provider_search = "exa_mcp"`, `provider_fetch = "exa_mcp"`.
87
82
 
88
83
  Tool schemas are tailored to the configured providers at startup/reload, so only supported provider-specific fields are exposed. Restart/reload Pi after changing provider config.
89
84
 
85
+ Provider-native efficiencies are used as follows:
86
+
87
+ | Service | Efficient path |
88
+ |---|---|
89
+ | Exa API | Keep included highlights; request bounded text in the same search only when requested; batch `/contents` fetches with freshness controls. |
90
+ | TinyFish | Paginate only as needed; use dedicated filters; batch only enough top-result fetches to fill requested context; pass TTL and intent. |
91
+ | Brave | Return pre-extracted LLM Context in one search call with native URL, token, snippet, threshold, and Goggles controls. |
92
+ | Firecrawl | Return search metadata by default; use scrape-on-search only when requested; pass cache/main-content options. |
93
+ | markdown.new | Keep `method: auto` so native Markdown falls back to AI and browser rendering only as needed; keep images opt-in. |
94
+ | Context7 | Support `fast` mode to skip reranking and reduce latency. |
95
+ | Exa Code | Send the requested/dynamic context token target directly to Exa Code Context. |
96
+
90
97
  ## Configuration
91
98
 
92
99
  Resolution order: defaults < environment variables < global config < trusted project config < CLI flags. Project config is ignored unless Pi trusts the current project, including in print, JSON, and RPC modes.
@@ -96,8 +103,8 @@ Resolution order: defaults < environment variables < global config < trusted pro
96
103
  ```bash
97
104
  PI_OFFLINE=1 # disables install/update telemetry
98
105
  PI_TELEMETRY=0 # disables install/update telemetry
99
- PI_WEB_KIT_PROVIDER_SEARCH=exa_mcp|exa|tinyfish|brave|firecrawl
100
- PI_WEB_KIT_PROVIDER_FETCH=exa_mcp|exa|tinyfish|markdown_new|firecrawl
106
+ PI_WEB_KIT_PROVIDER_SEARCH=exa|tinyfish|brave|firecrawl
107
+ PI_WEB_KIT_PROVIDER_FETCH=exa|tinyfish|markdown_new|firecrawl
101
108
  EXA_API_KEY=... # enables Exa provider and code_search
102
109
  CONTEXT7_API_KEY=... # enables library_search and library_docs
103
110
  TINYFISH_API_KEY=...
@@ -152,9 +159,15 @@ Searches with the active search provider and returns compact results grouped by
152
159
  |---|---|---|
153
160
  | `query` | string | Single search query. |
154
161
  | `queries` | string[] | Multiple related search queries. Max 5 after de-duplication. |
155
- | `numResults` | integer | Results per query. Range: 1-20. Default: 10. |
162
+ | `numResults` | integer | Desired results per query. Default: 10. Values above the active provider's limit are capped rather than rejected. |
163
+ | `contextTokens` | integer | Desired extracted context across the result set. Omitted native context uses an 8,192-token output budget; an explicit value can enable extraction. Any positive value; capped at 10,000. |
164
+ | `purpose` | string | Optional task/use-case hint for providers that support separate intent. |
165
+
166
+ `numResults` controls source breadth. `contextTokens` controls grounding depth. The tool automatically keeps native/included Exa highlights and Brave LLM Context. Explicit `contextTokens` enables extra extraction for Exa, TinyFish, and Firecrawl; this can add provider calls or provider cost. Search results keep a compact `snippet` plus ranked `content` and `contentFormat`, with one shared context budget and the existing 50KB tool-output limit.
167
+
168
+ Provider caps are Exa 100, Brave 50, and Firecrawl 100. TinyFish is paginated internally through its service maximum of page 10 and may make up to 11 search requests for one query. Search output reports requested, effective, returned, and omitted result/context counts. Brave `maxUrls` remains as a deprecated alias for `numResults`.
156
169
 
157
- Provider-specific parameters are exposed only for the configured provider, such as Exa date/domain filters, TinyFish `page`, Brave locale/freshness options, or Firecrawl scrape/search options.
170
+ Other provider-specific parameters are exposed only for the configured provider. These include Exa date/domain filters; TinyFish domain, date, geography, language, and publication filters; Brave locale, freshness, spellcheck, Goggles, and LLM Context controls; and Firecrawl scrape/search options.
158
171
 
159
172
  ### `web_fetch`
160
173
 
@@ -167,8 +180,9 @@ Fetches page content with the active fetch provider. Results are cached in memor
167
180
  | `offset` | integer | Character offset for cached/ranged reads. Single URL only. |
168
181
  | `limit` | integer | Maximum characters to return. Default: 30,000 for one URL, 8,000 for multiple URLs. |
169
182
  | `refresh` | boolean | Refetch even if cached. |
183
+ | `maxAgeMs` | integer | Desired maximum local/provider-cached page age in milliseconds. `0` requests live content where supported. |
170
184
 
171
- Provider-specific parameters are exposed only for the configured provider, such as TinyFish `format`, markdown.new `method` / `retainImages`, or Firecrawl `format`, `waitFor`, `mobile`, `location`, and `maxAge`.
185
+ Provider-specific parameters are exposed only for the configured provider. TinyFish supports `purpose`, `format`, links/images, selectors, per-URL timeout, and a seconds-based `ttl` alias. Exa supports the hours-based `maxAgeHours` alias. markdown.new supports `method` / `retainImages`. Firecrawl supports `format`, `waitFor`, `mobile`, structured `location`, and its existing `maxAge` alias. `refresh: true` also requests live content from Exa, TinyFish, and Firecrawl instead of only bypassing the local cache.
172
186
 
173
187
  ### `library_search`
174
188
 
@@ -214,7 +228,8 @@ Finds practical code examples, implementation context, setup snippets, migration
214
228
  | Max cached bytes | 20 MiB |
215
229
  | Max URLs per call | 10 |
216
230
  | Max queries per call | 5 |
217
- | Max `numResults` | 20 |
231
+ | Provider `numResults` caps | Exa 100; Brave 50; Firecrawl 100 |
232
+ | Search context budget | 10,000 tokens |
218
233
  | Max URL length | 2048 characters |
219
234
 
220
235
  Cache keys include the provider, canonical URL, fetch-affecting parameters, relevant provider defaults, and an opaque SHA-256 API-key/account scope. Internal cache keys are never returned in tool output. `refresh: true` bypasses and replaces the cached entry.
package/SECURITY.md CHANGED
@@ -21,8 +21,8 @@ The maintainer will acknowledge reports as soon as practical and coordinate disc
21
21
 
22
22
  `pi-web-kit` is a Pi package. Pi extensions execute with the same permissions as the local user running Pi. Users should review installed Pi packages and only install packages from sources they trust.
23
23
 
24
- `pi-web-kit` sends search queries and fetched URLs to the configured third-party provider. Page content returned by providers is cached in memory for the lifetime of the Pi process, subject to TTL and size limits. API keys are read from environment variables or local config files and are used only for provider requests. Do not commit API keys, tokens, or config files containing secrets.
24
+ `pi-web-kit` sends search queries and fetched URLs to the configured third-party provider. A TinyFish search may send the same query in up to 11 paginated requests to satisfy the requested result limit. When `contextTokens` is explicit, TinyFish also receives the selected result URLs for batched extraction, while Firecrawl may scrape search results. Page content returned by providers is cached in memory for the lifetime of the Pi process, subject to TTL and size limits. API keys are read from environment variables or local config files and are used only for provider requests. Do not commit API keys, tokens, or config files containing secrets.
25
25
 
26
- On startup, the extension sends a best-effort install/update telemetry ping to `mocito.dev` once per package version unless Pi telemetry is disabled or offline mode is enabled. The ping includes only the package name, version, and parsed platform/runtime/architecture from its User-Agent; it does not include prompts, queries, fetched URLs, file paths, config values, or API keys.
26
+ On startup, `@mocito/install-telemetry` sends a best-effort install/update telemetry ping to the configured telemetry endpoint once per package version unless Pi telemetry is disabled or offline mode is enabled. The ping includes only the package name, version, and parsed platform/runtime/architecture from its User-Agent; it does not include prompts, queries, fetched URLs, file paths, config values, or API keys.
27
27
 
28
28
  The extension validates URLs before fetch calls and only accepts `http:` and `https:` URLs without embedded credentials. This validation reduces accidental misuse but does not sandbox provider responses or the local Pi process.
@@ -4,7 +4,7 @@ import { Text } from "@earendil-works/pi-tui";
4
4
  import { Type } from "typebox";
5
5
  import { fetchCache, type CachedPage } from "../src/cache.js";
6
6
  import { resolveConfig } from "../src/config.js";
7
- import { DEFAULT_FETCH_LIMIT, DEFAULT_NUM_RESULTS, MAX_LIMIT, MAX_NUM_RESULTS, MAX_OFFSET, MAX_QUERY_COUNT, MAX_URL_COUNT, MULTI_FETCH_LIMIT } from "../src/limits.js";
7
+ import { applySearchContextBudget, capSearchResultLimit, DEFAULT_FETCH_LIMIT, DEFAULT_NUM_RESULTS, DEFAULT_SEARCH_CONTEXT_TOKENS, MAX_LIMIT, MAX_NUM_RESULTS, MAX_OFFSET, MAX_QUERY_COUNT, MAX_SEARCH_CONTEXT_TOKENS, MAX_URL_COUNT, MULTI_FETCH_LIMIT, safePrefix, TINYFISH_MAX_PAGE } from "../src/limits.js";
8
8
  import { createCodeSearchProvider, createContext7Provider, createFetchProvider, createSearchProvider } from "../src/providers/index.js";
9
9
  import { mapFetchResults } from "../src/providers/fallback.js";
10
10
  import type { FetchProviderName, SearchProviderName, WebFetchResult } from "../src/types.js";
@@ -14,11 +14,11 @@ import { reportInstallTelemetry } from "../src/install-telemetry.js";
14
14
  export default function (pi: ExtensionAPI) {
15
15
  void reportInstallTelemetry();
16
16
  pi.registerFlag("web-provider-search", {
17
- description: "Temporary pi-web-kit search provider override (exa_mcp, exa, tinyfish, brave, firecrawl)",
17
+ description: "Temporary pi-web-kit search provider override (exa, tinyfish, brave, firecrawl)",
18
18
  type: "string",
19
19
  });
20
20
  pi.registerFlag("web-provider-fetch", {
21
- description: "Temporary pi-web-kit fetch provider override (exa_mcp, exa, tinyfish, markdown_new, firecrawl)",
21
+ description: "Temporary pi-web-kit fetch provider override (exa, tinyfish, markdown_new, firecrawl)",
22
22
  type: "string",
23
23
  });
24
24
 
@@ -38,17 +38,34 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
38
38
  name: "web_search",
39
39
  label: "Web Search",
40
40
  description: buildSearchDescription(startupConfig.provider_search),
41
- promptSnippet: "Find current or external web information.",
42
- promptGuidelines: ["Use web_search to find current or external web information."],
41
+ promptSnippet: startupConfig.apiKeys.context7
42
+ ? "Search the live web for current events, news, and non-library topics."
43
+ : "Find current or external web information.",
44
+ promptGuidelines: [
45
+ startupConfig.apiKeys.context7
46
+ ? "Use web_search for current events, news, and non-library topics."
47
+ : "Use web_search to find current or external web information.",
48
+ ...(startupConfig.apiKeys.context7
49
+ ? ["Prefer library_docs over web_search for versioned library/framework/SDK documentation."]
50
+ : []),
51
+ ],
43
52
  parameters: buildSearchSchema(startupConfig.provider_search),
44
53
  async execute(_toolCallId, rawParams, signal, onUpdate, ctx) {
45
54
  const params = rawParams as Record<string, any>;
46
55
  const queries = normalizeQueries(params);
47
- const numResults = parseInteger(params.numResults, DEFAULT_NUM_RESULTS, "numResults", 1, MAX_NUM_RESULTS);
48
56
  if (queries.length === 0) throw new Error("web_search requires query or queries.");
49
57
 
50
58
  const config = runtimeConfig(pi, ctx.cwd, projectIsTrusted(ctx));
51
59
  assertProviderUnchanged("web_search", startupConfig.provider_search, config.provider_search);
60
+ const requestedResultLimit = parseInteger(
61
+ params.numResults ?? (config.provider_search === "brave" ? params.maxUrls : undefined),
62
+ DEFAULT_NUM_RESULTS,
63
+ "numResults",
64
+ 1,
65
+ );
66
+ const effectiveResultLimit = capSearchResultLimit(config.provider_search, requestedResultLimit);
67
+ const requestedContextTokens = params.contextTokens == null ? undefined : parseInteger(params.contextTokens, 0, "contextTokens", 1);
68
+ const effectiveContextTokens = Math.min(requestedContextTokens ?? DEFAULT_SEARCH_CONTEXT_TOKENS, MAX_SEARCH_CONTEXT_TOKENS);
52
69
  const provider = createSearchProvider(config);
53
70
  const grouped = [];
54
71
  const progress = createProgress("search", config.provider_search, queries);
@@ -56,9 +73,20 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
56
73
  for (const query of queries) {
57
74
  markProgressCurrent(progress, query);
58
75
  emitProgress(onUpdate, progress);
59
- const result = await provider.search({ ...params, query, numResults }, signal);
60
- grouped.push({ query, results: result.results });
61
- markProgressDone(progress, query, `${result.results.length} results`);
76
+ const result = await provider.search({ ...params, query, numResults: effectiveResultLimit, contextTokens: requestedContextTokens == null ? undefined : effectiveContextTokens }, signal);
77
+ const bounded = applySearchContextBudget(result.results, effectiveContextTokens * 4);
78
+ grouped.push({
79
+ query,
80
+ requestedResultLimit,
81
+ effectiveResultLimit: result.effectiveResultLimit ?? effectiveResultLimit,
82
+ requestedContextTokens,
83
+ effectiveContextTokens,
84
+ contextCharacters: bounded.contextCharacters,
85
+ ...(bounded.omittedContextCharacters ? { omittedContextCharacters: bounded.omittedContextCharacters } : {}),
86
+ resultCount: bounded.results.length,
87
+ results: bounded.results,
88
+ });
89
+ markProgressDone(progress, query, `${bounded.results.length} results`);
62
90
  emitProgress(onUpdate, progress);
63
91
  }
64
92
  const result = { provider: config.provider_search, queries: grouped };
@@ -108,9 +136,9 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
108
136
  pi.registerTool({
109
137
  name: "library_search",
110
138
  label: "Library Search",
111
- description: "Resolve library, package, framework, SDK, API, or CLI names to canonical library IDs.",
112
- promptSnippet: "Resolve a library name to a canonical library ID before querying docs.",
113
- promptGuidelines: ["Use library_search when a library/framework/package is ambiguous or you need a canonical library ID."],
139
+ description: "Resolve library, package, framework, SDK, API, or CLI names to canonical library IDs with version, trust, and snippet metadata. Use when you need to inspect candidate matches (official sources, versions, forks); library_docs resolves names automatically.",
140
+ promptSnippet: "List candidate library IDs; library_docs resolves names on its own.",
141
+ promptGuidelines: ["Use library_search only to inspect matching libraries (official source, versions, forks) before a precise library_docs call.", "Not needed for ordinary doc lookups — library_docs resolves a libraryName for you."],
114
142
  parameters: buildLibrarySearchSchema(),
115
143
  async execute(_toolCallId, rawParams, signal, _onUpdate, ctx) {
116
144
  const params = rawParams as Record<string, any>;
@@ -126,9 +154,9 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
126
154
  pi.registerTool({
127
155
  name: "library_docs",
128
156
  label: "Library Docs",
129
- description: "Fetch current, version-aware documentation and code examples for a library.",
130
- promptSnippet: "Get current library documentation and code examples.",
131
- promptGuidelines: ["Use library_docs for current APIs, framework behavior, SDK examples, package docs, and version-specific library questions."],
157
+ description: "Fetch current, version-aware documentation and code examples for a library. Pass libraryName to resolve it automatically, or a known libraryId.",
158
+ promptSnippet: "Get current, version-aware library/framework/SDK docs and code examples.",
159
+ promptGuidelines: ["Use library_docs for any library, framework, SDK, API, or CLI — even familiar ones like React, Next.js, Prisma, or Express. Training data may be outdated; this returns version-specific docs.", "Prefer library_docs over web_search or your own knowledge for library/API questions. Pass libraryName and query; you rarely need library_search first."],
132
160
  parameters: buildLibraryDocsSchema(),
133
161
  async execute(_toolCallId, rawParams, signal, _onUpdate, ctx) {
134
162
  const params = rawParams as Record<string, any>;
@@ -154,7 +182,10 @@ function registerTools(pi: ExtensionAPI, startupConfig: ReturnType<typeof resolv
154
182
  label: "Code Search",
155
183
  description: "Find practical code examples, usage patterns, setup snippets, migrations, and error context.",
156
184
  promptSnippet: "Find real-world code examples, usage patterns, migrations, and error context.",
157
- promptGuidelines: ["Use code_search for real-world code examples, GitHub/open-source usage patterns, API syntax examples, setup snippets, migrations, and error messages."],
185
+ promptGuidelines: [
186
+ "Use code_search for real-world code examples, usage patterns, setup snippets, migrations, and error context from open source.",
187
+ ...(startupConfig.apiKeys.context7 ? ["Prefer library_docs for official API/reference documentation."] : []),
188
+ ],
158
189
  parameters: buildCodeSearchSchema(),
159
190
  async execute(_toolCallId, rawParams, signal, _onUpdate, ctx) {
160
191
  const params = rawParams as Record<string, any>;
@@ -200,7 +231,9 @@ export function buildSearchSchema(provider: SearchProviderName) {
200
231
  const props: Record<string, any> = {
201
232
  query: Type.Optional(Type.String({ description: "Single search query" })),
202
233
  queries: Type.Optional(Type.Array(Type.String(), { description: `Multiple related search queries (max ${MAX_QUERY_COUNT})`, maxItems: MAX_QUERY_COUNT })),
203
- numResults: Type.Optional(int("Results per query", 1, MAX_NUM_RESULTS)),
234
+ numResults: Type.Optional(int("Desired results per query; capped by the active provider", 1)),
235
+ contextTokens: Type.Optional(int("Desired extracted context tokens; capped to the tool output budget and may enable provider extraction", 1)),
236
+ purpose: Type.Optional(Type.String({ description: "Optional task/use-case hint when supported by the provider", minLength: 1, maxLength: 2_000 })),
204
237
  };
205
238
  if (provider === "exa") Object.assign(props, {
206
239
  includeDomains: Type.Optional(Type.Array(Type.String())),
@@ -211,13 +244,32 @@ export function buildSearchSchema(provider: SearchProviderName) {
211
244
  endCrawlDate: Type.Optional(Type.String()),
212
245
  type: Type.Optional(Type.String()),
213
246
  category: Type.Optional(Type.String()),
247
+ maxAgeHours: Type.Optional(int("Maximum Exa cached content age in hours; 0 forces live crawl, -1 disables live crawl", -1)),
248
+ });
249
+ if (provider === "tinyfish") Object.assign(props, {
250
+ page: Type.Optional(int("Deprecated starting result page", 0, TINYFISH_MAX_PAGE)),
251
+ location: Type.Optional(Type.String()), language: Type.Optional(Type.String()),
252
+ includeDomains: Type.Optional(Type.Array(Type.String())), excludeDomains: Type.Optional(Type.Array(Type.String())),
253
+ domainType: Type.Optional(Type.Union([Type.Literal("web"), Type.Literal("news"), Type.Literal("research_paper")])),
254
+ afterDate: Type.Optional(Type.String({ pattern: "^\\d{4}-\\d{2}-\\d{2}$" })), beforeDate: Type.Optional(Type.String({ pattern: "^\\d{4}-\\d{2}-\\d{2}$" })),
255
+ recencyMinutes: Type.Optional(int("Freshness window in minutes", 1, 5_256_000)),
256
+ pubYearMin: Type.Optional(int("Minimum publication year", 0, 9_999)), pubYearMax: Type.Optional(int("Maximum publication year", 0, 9_999)),
214
257
  });
215
- if (provider === "tinyfish") Object.assign(props, { page: Type.Optional(int("Result page", 1, 10)) });
216
258
  if (provider === "brave") Object.assign(props, {
217
- country: Type.Optional(Type.String()), searchLang: Type.Optional(Type.String()), uiLang: Type.Optional(Type.String()), safesearch: Type.Optional(Type.String()), freshness: Type.Optional(Type.String()), maxUrls: Type.Optional(int("Maximum URLs", 1, MAX_NUM_RESULTS)),
259
+ country: Type.Optional(Type.String({ minLength: 2, maxLength: 2 })), searchLang: Type.Optional(Type.String({ minLength: 2, maxLength: 10 })),
260
+ safesearch: Type.Optional(Type.Union([Type.Literal("off"), Type.Literal("moderate"), Type.Literal("strict")])),
261
+ freshness: Type.Optional(Type.String()), spellcheck: Type.Optional(Type.Boolean()),
262
+ contextThresholdMode: Type.Optional(Type.Union([Type.Literal("disabled"), Type.Literal("strict"), Type.Literal("balanced"), Type.Literal("lenient")])),
263
+ maxSnippets: Type.Optional(int("Maximum context snippets", 1, 256)),
264
+ maxTokensPerUrl: Type.Optional(int("Maximum context tokens per URL", 512, 8_192)),
265
+ maxSnippetsPerUrl: Type.Optional(int("Maximum context snippets per URL", 1, 100)),
266
+ goggles: Type.Optional(Type.Union([Type.String(), Type.Array(Type.String(), { maxItems: 3 })])),
267
+ maxUrls: Type.Optional(int("Deprecated alias for numResults", 1)),
218
268
  });
219
269
  if (provider === "firecrawl") Object.assign(props, {
220
- location: Type.Optional(Type.String()), country: Type.Optional(Type.String()), includeDomains: Type.Optional(Type.Array(Type.String())), excludeDomains: Type.Optional(Type.Array(Type.String())), categories: Type.Optional(Type.Array(Type.String())), tbs: Type.Optional(Type.String()), scrape: Type.Optional(Type.Boolean({ description: "Enable default markdown scrape-on-search" })), scrapeOptions: Type.Optional(Type.Object({}, { additionalProperties: true, description: "Firecrawl scrapeOptions for search." })),
270
+ location: Type.Optional(Type.String()), country: Type.Optional(Type.String()), includeDomains: Type.Optional(Type.Array(Type.String())), excludeDomains: Type.Optional(Type.Array(Type.String())),
271
+ categories: Type.Optional(Type.Array(Type.Union([Type.Literal("research"), Type.Literal("pdf"), Type.Literal("developer")]))),
272
+ tbs: Type.Optional(Type.String()), scrape: Type.Optional(Type.Boolean({ description: "Enable default markdown scrape-on-search" })), scrapeOptions: Type.Optional(Type.Object({}, { additionalProperties: true, description: "Firecrawl scrapeOptions for search." })),
221
273
  });
222
274
  return Type.Object(props, { additionalProperties: false });
223
275
  }
@@ -229,12 +281,23 @@ export function buildFetchSchema(provider: FetchProviderName) {
229
281
  offset: Type.Optional(int("Character offset for cached/ranged reads", 0, MAX_OFFSET)),
230
282
  limit: Type.Optional(int("Maximum characters to return", 1, MAX_LIMIT)),
231
283
  refresh: Type.Optional(Type.Boolean({ description: "Refetch even if cached" })),
284
+ maxAgeMs: Type.Optional(int("Maximum local/provider-cached page age in milliseconds; 0 requests live content where supported", 0)),
232
285
  };
286
+ if (provider === "exa") Object.assign(props, { maxAgeHours: Type.Optional(int("Maximum Exa cached content age in hours; 0 forces live crawl, -1 disables live crawl", -1)) });
233
287
  if (provider === "tinyfish") Object.assign(props, { format: Type.Optional(Type.Union([Type.Literal("markdown"), Type.Literal("html"), Type.Literal("json")])), links: Type.Optional(Type.Boolean()), imageLinks: Type.Optional(Type.Boolean()) });
288
+ if (provider === "tinyfish") Object.assign(props, {
289
+ purpose: Type.Optional(Type.String({ minLength: 1, maxLength: 2_000 })),
290
+ ttl: Type.Optional(int("Provider cache freshness tolerance in seconds; 0 prefers live fetch", 0)),
291
+ perUrlTimeoutMs: Type.Optional(int("Per-URL timeout in milliseconds", 1, 110_000)),
292
+ includeSelectors: Type.Optional(Type.Array(Type.String({ minLength: 1, maxLength: 1_000 }), { minItems: 1, maxItems: 20 })),
293
+ excludeSelectors: Type.Optional(Type.Array(Type.String({ minLength: 1, maxLength: 1_000 }), { minItems: 1, maxItems: 20 })),
294
+ });
234
295
  if (provider === "markdown_new") Object.assign(props, { method: Type.Optional(Type.Union([Type.Literal("auto"), Type.Literal("ai"), Type.Literal("browser")])), retainImages: Type.Optional(Type.Boolean()) });
235
296
  if (provider === "firecrawl") Object.assign(props, {
236
297
  format: Type.Optional(Type.Union([Type.Literal("markdown"), Type.Literal("html"), Type.Literal("json")])),
237
- onlyMainContent: Type.Optional(Type.Boolean()), waitFor: Type.Optional(int("Milliseconds to wait", 0, 60_000)), mobile: Type.Optional(Type.Boolean()), location: Type.Optional(Type.String()), maxAge: Type.Optional(int("Maximum cached page age", 0)),
298
+ onlyMainContent: Type.Optional(Type.Boolean()), waitFor: Type.Optional(int("Milliseconds to wait", 0, 60_000)), mobile: Type.Optional(Type.Boolean()),
299
+ location: Type.Optional(Type.Object({ country: Type.Optional(Type.String()), languages: Type.Optional(Type.Array(Type.String())) }, { additionalProperties: false })),
300
+ maxAge: Type.Optional(int("Maximum cached page age in milliseconds", 0)),
238
301
  });
239
302
  return Type.Object(props, { additionalProperties: false });
240
303
  }
@@ -292,6 +355,7 @@ export async function fetchWithCache(providerName: FetchProviderName, params: Re
292
355
  const defaultLimit = urls.length > 1 ? MULTI_FETCH_LIMIT : DEFAULT_FETCH_LIMIT;
293
356
  const limit = parseInteger(params.limit, defaultLimit, "limit", 1, MAX_LIMIT);
294
357
  const refresh = params.refresh === true;
358
+ const maxAgeMs = localCacheMaxAge(providerName, params);
295
359
 
296
360
  const pages = new Map<string, { page?: CachedPage; cached: boolean; refreshed: boolean; error?: string }>();
297
361
  const cacheKeys = new Map<string, string>();
@@ -300,7 +364,7 @@ export async function fetchWithCache(providerName: FetchProviderName, params: Re
300
364
  const cacheKey = buildCacheKey(providerName, url, params, config);
301
365
  cacheKeys.set(url, cacheKey);
302
366
  const cached = fetchCache.get(cacheKey);
303
- if (cached && !refresh) {
367
+ if (cached && !refresh && (maxAgeMs == null || (maxAgeMs > 0 && Date.now() - cached.fetchedAt <= maxAgeMs))) {
304
368
  pages.set(url, { page: cached, cached: true, refreshed: false });
305
369
  onProgress?.({ status: "done", url, note: "cached" });
306
370
  } else {
@@ -344,6 +408,14 @@ export async function fetchWithCache(providerName: FetchProviderName, params: Re
344
408
  }) };
345
409
  }
346
410
 
411
+ function localCacheMaxAge(provider: FetchProviderName, params: Record<string, any>): number | undefined {
412
+ if (params.maxAgeMs != null) return parseInteger(params.maxAgeMs, 0, "maxAgeMs", 0);
413
+ if (provider === "exa" && typeof params.maxAgeHours === "number" && params.maxAgeHours >= 0) return params.maxAgeHours * 3_600_000;
414
+ if (provider === "tinyfish" && typeof params.ttl === "number") return params.ttl * 1_000;
415
+ if (provider === "firecrawl" && typeof params.maxAge === "number") return params.maxAge;
416
+ return undefined;
417
+ }
418
+
347
419
 
348
420
  export function pageSlice(page: CachedPage, offset: number, limit: number, cached: boolean, refreshed: boolean) {
349
421
  const total = page.content.length;
@@ -424,8 +496,20 @@ function isSearchResult(value: any): value is { provider?: string; queries: Arra
424
496
  }
425
497
 
426
498
  function boundSearchResult(value: { provider?: string; queries: Array<{ query?: string; results?: any[] }> }) {
427
- const compact = limitStrings(value, 1_000) as typeof value;
428
- const maxSnippet = Math.max(0, ...compact.queries.flatMap((group) => (group.results ?? []).map((item) => typeof item?.snippet === "string" ? item.snippet.length : 0)));
499
+ const compact = {
500
+ ...(limitStrings(value, 1_000) as Record<string, unknown>),
501
+ queries: value.queries.map((group) => ({
502
+ ...(limitStrings(group, 1_000) as Record<string, unknown>),
503
+ results: (group.results ?? []).map((item) => ({
504
+ ...(limitStrings(item, 1_000) as Record<string, unknown>),
505
+ ...(typeof item?.content === "string" ? { content: item.content } : {}),
506
+ })),
507
+ })),
508
+ } as typeof value;
509
+ const maxSnippet = Math.max(0, ...compact.queries.flatMap((group) => (group.results ?? []).flatMap((item) => [
510
+ typeof item?.snippet === "string" ? item.snippet.length : 0,
511
+ typeof item?.content === "string" ? item.content.length : 0,
512
+ ])));
429
513
  let low = 0;
430
514
  let high = maxSnippet;
431
515
  let best = searchResultWithLimits(compact, 0);
@@ -461,17 +545,29 @@ function boundSearchResult(value: { provider?: string; queries: Array<{ query?:
461
545
  function searchResultWithLimits(value: { provider?: string; queries: Array<{ query?: string; results?: any[] }> }, snippetChars: number, resultsPerQuery?: number) {
462
546
  return {
463
547
  ...value,
464
- queries: value.queries.map((group) => ({
465
- ...group,
466
- results: (group.results ?? []).slice(0, resultsPerQuery).map((item) => {
467
- if (typeof item?.snippet !== "string") return item;
468
- if (snippetChars === 0) {
469
- const { snippet: _snippet, ...metadata } = item;
470
- return metadata;
548
+ queries: value.queries.map((group) => {
549
+ const originalContextCharacters = (group.results ?? []).reduce((total, item) => total + (typeof item?.content === "string" ? item.content.length : 0), 0);
550
+ const results = (group.results ?? []).slice(0, resultsPerQuery).map((item) => {
551
+ const bounded = { ...item };
552
+ for (const key of ["snippet", "content"] as const) {
553
+ if (typeof bounded[key] !== "string") continue;
554
+ if (snippetChars === 0) delete bounded[key];
555
+ else bounded[key] = safePrefix(bounded[key], snippetChars);
471
556
  }
472
- return { ...item, snippet: safePrefix(item.snippet, snippetChars) };
473
- }),
474
- })),
557
+ return bounded;
558
+ });
559
+ const omittedResultCount = Math.max(0, (group.results?.length ?? 0) - results.length);
560
+ const contextCharacters = results.reduce((total, item) => total + (typeof item?.content === "string" ? item.content.length : 0), 0);
561
+ const omittedContextCharacters = (typeof (group as any).omittedContextCharacters === "number" ? (group as any).omittedContextCharacters : 0) + originalContextCharacters - contextCharacters;
562
+ return {
563
+ ...group,
564
+ contextCharacters,
565
+ ...(omittedContextCharacters ? { omittedContextCharacters } : {}),
566
+ resultCount: results.length,
567
+ ...(omittedResultCount ? { omittedResultCount } : {}),
568
+ results,
569
+ };
570
+ }),
475
571
  };
476
572
  }
477
573
 
@@ -509,12 +605,6 @@ function fetchResultWithContentLimit(value: { provider?: string; results: any[]
509
605
  };
510
606
  }
511
607
 
512
- function safePrefix(value: string, maxChars: number): string {
513
- let end = Math.min(value.length, maxChars);
514
- if (end > 0 && /[\uD800-\uDBFF]/.test(value[end - 1])) end--;
515
- return value.slice(0, end);
516
- }
517
-
518
608
  function limitStrings(value: unknown, maxChars: number): unknown {
519
609
  if (typeof value === "string") return safePrefix(value, maxChars);
520
610
  if (Array.isArray(value)) return value.map((item) => limitStrings(item, maxChars));
@@ -522,9 +612,11 @@ function limitStrings(value: unknown, maxChars: number): unknown {
522
612
  return value;
523
613
  }
524
614
 
525
- function parseInteger(value: unknown, defaultValue: number, name: string, min: number, max: number): number {
615
+ function parseInteger(value: unknown, defaultValue: number, name: string, min: number, max?: number): number {
526
616
  if (value == null) return defaultValue;
527
- if (typeof value !== "number" || !Number.isInteger(value) || !Number.isFinite(value) || value < min || value > max) throw new Error(`${name} must be a finite integer between ${min} and ${max}.`);
617
+ if (typeof value !== "number" || !Number.isInteger(value) || !Number.isFinite(value) || value < min || (max != null && value > max)) {
618
+ throw new Error(`${name} must be a finite integer ${max == null ? `>= ${min}` : `between ${min} and ${max}`}.`);
619
+ }
528
620
  return value;
529
621
  }
530
622
 
@@ -553,7 +645,6 @@ function fetchConfigDefaults(provider: FetchProviderName, config?: any): Record<
553
645
  function providerScope(provider: FetchProviderName, config?: any): string {
554
646
  const keyMap: Partial<Record<FetchProviderName, string | undefined>> = {
555
647
  exa: config?.apiKeys?.exa,
556
- exa_mcp: config?.apiKeys?.exa,
557
648
  tinyfish: config?.apiKeys?.tinyfish,
558
649
  firecrawl: config?.apiKeys?.firecrawl,
559
650
  };
@@ -579,7 +670,14 @@ function searchDetails(value: any) {
579
670
  provider: value.provider,
580
671
  queries: value.queries.map((q: any) => ({
581
672
  query: q.query,
673
+ requestedResultLimit: q.requestedResultLimit,
674
+ effectiveResultLimit: q.effectiveResultLimit,
675
+ requestedContextTokens: q.requestedContextTokens,
676
+ effectiveContextTokens: q.effectiveContextTokens,
677
+ contextCharacters: q.contextCharacters,
678
+ omittedContextCharacters: q.omittedContextCharacters,
582
679
  resultCount: (q.results ?? []).length,
680
+ omittedResultCount: q.omittedResultCount,
583
681
  results: (q.results ?? []).map((r: any) => ({ title: r.title, url: r.url, siteName: r.siteName, position: r.position })),
584
682
  })),
585
683
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-web-kit",
3
- "version": "0.2.3",
3
+ "version": "0.3.0",
4
4
  "description": "Context-efficient web search and fetch tools for Pi.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -58,18 +58,21 @@
58
58
  "typebox": "*"
59
59
  },
60
60
  "devDependencies": {
61
- "@earendil-works/pi-ai": "^0.80.0",
62
- "@earendil-works/pi-coding-agent": "^0.80.0",
63
- "@earendil-works/pi-tui": "^0.80.0",
64
- "@types/node": "^26.1.0",
65
- "tsx": "^4.23.0",
66
- "typebox": "^1.3.4",
67
- "typescript": "^6.0.3"
61
+ "@earendil-works/pi-ai": "^0.84.1",
62
+ "@earendil-works/pi-coding-agent": "^0.84.1",
63
+ "@earendil-works/pi-tui": "^0.84.1",
64
+ "@types/node": "^26.2.0",
65
+ "tsx": "^4.23.5",
66
+ "typebox": "^1.3.10",
67
+ "typescript": "^7.0.2"
68
68
  },
69
69
  "publishConfig": {
70
70
  "access": "public"
71
71
  },
72
72
  "engines": {
73
73
  "node": ">=20.6.0"
74
+ },
75
+ "dependencies": {
76
+ "@mocito/install-telemetry": "0.1.1"
74
77
  }
75
78
  }
package/src/config.ts CHANGED
@@ -3,11 +3,11 @@ import { homedir } from "node:os";
3
3
  import { join } from "node:path";
4
4
  import type { FetchProviderName, SearchProviderName, WebKitConfig } from "./types.js";
5
5
 
6
- const SEARCH = ["exa_mcp", "exa", "tinyfish", "brave", "firecrawl"] as const;
7
- const FETCH = ["exa_mcp", "exa", "tinyfish", "markdown_new", "firecrawl"] as const;
6
+ const SEARCH = ["exa", "tinyfish", "brave", "firecrawl"] as const;
7
+ const FETCH = ["exa", "tinyfish", "markdown_new", "firecrawl"] as const;
8
8
  const DEFAULT_CONFIG: WebKitConfig = {
9
- provider_search: "exa_mcp",
10
- provider_fetch: "exa_mcp",
9
+ provider_search: "exa",
10
+ provider_fetch: "exa",
11
11
  apiKeys: {},
12
12
  markdownNew: { method: "auto", retainImages: false },
13
13
  };
package/src/http.ts CHANGED
@@ -32,10 +32,19 @@ export async function requestJson<T>(url: string, init: RequestInit & { timeoutM
32
32
  }
33
33
 
34
34
  export function asSnippet(value: unknown): string | undefined {
35
+ return asText(value)?.slice(0, 1000) || undefined;
36
+ }
37
+
38
+ export function asText(value: unknown): string | undefined {
35
39
  if (value == null) return undefined;
36
- if (Array.isArray(value)) return value.filter(Boolean).join("\n").slice(0, 1000) || undefined;
37
- if (typeof value === "string") return value.slice(0, 1000) || undefined;
38
- return JSON.stringify(value).slice(0, 1000);
40
+ if (Array.isArray(value)) return value.filter(Boolean).map((item) => typeof item === "string" ? item : JSON.stringify(item)).join("\n") || undefined;
41
+ if (typeof value === "string") return value || undefined;
42
+ return JSON.stringify(value);
43
+ }
44
+
45
+ export function withoutContent(value: any): Record<string, unknown> {
46
+ const { text: _text, content: _content, markdown: _markdown, html: _html, summary: _summary, highlights: _highlights, ...metadata } = value ?? {};
47
+ return metadata;
39
48
  }
40
49
 
41
50
  export function normalizeUrls(input: { url?: string; urls?: string[] }, maxCount?: number): string[] {