bestony-pi-preset 0.0.21 → 0.0.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [2.23.0] - 2026-08-11
11
+
12
+ ### Added
13
+ - Added interactive callback URL pasting to `/mcp-auth` for OAuth flows running on remote or headless machines. Thanks @trevorleibert-mixpanel for PR #330.
14
+
15
+ ### Fixed
16
+ - Stopped load-time MCP initialization from printing a TUI startup error when Pi action methods are not bound yet. Thanks @21307369 for issue #327.
17
+ - Kept interactive OAuth authorization URLs clickable as a single terminal hyperlink. Thanks @rfccg for PR #329.
18
+
10
19
  ## [2.22.0] - 2026-08-11
11
20
 
12
21
  ### Added
@@ -265,7 +265,9 @@ The adapter owns only its client socket and closes that connection when the Pi r
265
265
 
266
266
  ### Remote/headless OAuth
267
267
 
268
- If Pi is running on a remote server and cannot open a local browser, start OAuth through the proxy tool. Persistent OAuth still requires an available OS credential store; on headless Linux that usually means an unlocked Secret Service/libsecret keyring. The adapter fails closed instead of falling back to plaintext credentials when the secure store is unavailable.
268
+ If Pi is running on a remote server, `/mcp-auth <server>` prints the authorization URL and opens a callback input. Open the URL in your local browser. After approval, the browser may fail to load the localhost callback page because localhost refers to your workstation; copy the full URL from its address bar and paste it into Pi. The input closes automatically instead when the browser can reach Pi's callback directly.
269
+
270
+ The same flow is available through the proxy tool for non-interactive clients. Persistent OAuth still requires an available OS credential store; on headless Linux that usually means an unlocked Secret Service/libsecret keyring. The adapter fails closed instead of falling back to plaintext credentials when the secure store is unavailable.
269
271
 
270
272
  On Linux, if credential access fails because Pi inherited a revoked session keyring, the adapter uses a best-effort recovery path through `keyctl session - node <packaged helper>` so explicit re-authentication can write fresh credentials without killing a long-lived tmux server. This path requires `keyctl` and `node` on `PATH`; missing, locked, or otherwise unavailable credential stores still fail closed.
271
273
 
@@ -25,6 +25,10 @@ import { loadOnboardingState, markSetupCompleted as persistSetupCompleted, markS
25
25
  import { openPath, resolveServerUrl, sanitizeTerminalText } from "./utils.ts";
26
26
  import { isAbortError } from "./runtime-owner.ts";
27
27
 
28
+ function terminalHyperlink(label: string, url: string): string {
29
+ return `\u001B]8;;${sanitizeTerminalText(url)}\u001B\\${sanitizeTerminalText(label)}\u001B]8;;\u001B\\`;
30
+ }
31
+
28
32
  export async function showStatus(state: McpExtensionState, ctx: ExtensionContext): Promise<void> {
29
33
  if (!ctx.hasUI) return;
30
34
 
@@ -273,11 +277,17 @@ export async function authenticateServer(
273
277
  ...(authStorageOptions.baseDir ? { authStorageOptions } : {}),
274
278
  onAuthorizationUrl: (authorizationUrl) => {
275
279
  ui.notify(
276
- `Open this URL to authenticate ${serverName}:\n\n${authorizationUrl}\n\n` +
277
- "After approving, return to Pi; the local callback will complete automatically.",
280
+ `Open this URL to authenticate ${serverName}:\n\n${terminalHyperlink(authorizationUrl, authorizationUrl)}\n\n` +
281
+ "After approving, Pi will complete automatically if the browser can reach its localhost callback. " +
282
+ "On a remote machine, copy the full localhost URL from the browser address bar and paste it into Pi.",
278
283
  "info"
279
284
  );
280
285
  },
286
+ onAuthorizationInput: (_authorizationUrl, inputSignal) => ui.input(
287
+ `Complete ${serverName} OAuth`,
288
+ "Paste the full callback URL, or wait for automatic completion",
289
+ { signal: inputSignal },
290
+ ),
281
291
  ...(signal ? { signal } : {}),
282
292
  ...(runtime ? { runtime } : {}),
283
293
  });
@@ -169,13 +169,23 @@ function installMcpAdapter(pi: ExtensionAPI, options: McpAdapterOptions) {
169
169
  return resolveDirectTools(config, cache, prefix, envDirectToolOverride);
170
170
  }
171
171
 
172
+ function getActiveToolsIfReady(): string[] | undefined {
173
+ try {
174
+ return pi.getActiveTools?.();
175
+ } catch (error) {
176
+ if (error instanceof Error
177
+ && error.message.includes("Action methods cannot be called during extension loading")) return undefined;
178
+ throw error;
179
+ }
180
+ }
181
+
172
182
  function deactivateTools(toolNames: string[]): string[] {
173
183
  if (toolNames.length === 0) return [];
174
184
  const unregisterTool = (pi as ExtensionAPI & { unregisterTool?: (name: string) => boolean }).unregisterTool;
175
185
  const unregistered = toolNames.filter((toolName) => unregisterTool?.(toolName) === true);
176
186
  const fallbackNames = toolNames.filter((toolName) => !unregistered.includes(toolName));
177
187
  const remove = new Set(toolNames);
178
- const activeTools = pi.getActiveTools?.();
188
+ const activeTools = getActiveToolsIfReady();
179
189
  if (!activeTools || activeTools.length === 0) {
180
190
  for (const toolName of fallbackNames) fallbackDeactivatedTools.add(toolName);
181
191
  return unregistered;
@@ -207,7 +217,7 @@ function installMcpAdapter(pi: ExtensionAPI, options: McpAdapterOptions) {
207
217
  registerDirectTool(spec);
208
218
  registeredDirectTools.set(spec.prefixedName, fingerprint);
209
219
  if (fallbackDeactivatedTools.delete(spec.prefixedName)) {
210
- const activeTools = pi.getActiveTools?.();
220
+ const activeTools = getActiveToolsIfReady();
211
221
  if (activeTools && !activeTools.includes(spec.prefixedName)) {
212
222
  pi.setActiveTools([...activeTools, spec.prefixedName]);
213
223
  }
@@ -849,7 +859,7 @@ function installMcpAdapter(pi: ExtensionAPI, options: McpAdapterOptions) {
849
859
  registerProxyTool(description);
850
860
  return;
851
861
  }
852
- const activeTools = pi.getActiveTools?.();
862
+ const activeTools = getActiveToolsIfReady();
853
863
  if (activeTools && !activeTools.includes("mcp")) {
854
864
  pi.setActiveTools([...activeTools, "mcp"]);
855
865
  }
@@ -49,6 +49,10 @@ export interface McpOAuthRuntime {
49
49
 
50
50
  export interface AuthenticateOptions {
51
51
  onAuthorizationUrl?: (authorizationUrl: string) => void | Promise<void>
52
+ onAuthorizationInput?: (
53
+ authorizationUrl: string,
54
+ signal: AbortSignal,
55
+ ) => Promise<string | undefined>
52
56
  authStorageOptions?: AuthStorageOptions
53
57
  signal?: AbortSignal
54
58
  runtime?: McpOAuthRuntime
@@ -568,6 +572,53 @@ export function parseAuthorizationCodeInput(input: string, expectedState?: strin
568
572
  return parseAuthorizationRedirectInput(input, expectedState).code
569
573
  }
570
574
 
575
+ type AuthorizationResponse = {
576
+ input: AuthorizationCodeInput
577
+ source: "callback" | "manual"
578
+ }
579
+
580
+ /**
581
+ * Wait for either the localhost callback or a manually pasted redirect URL.
582
+ * The manual input prompt is dismissed as soon as either path finishes.
583
+ */
584
+ export async function waitForAuthorizationResponse(
585
+ callbackPromise: Promise<AuthorizationCodeInput>,
586
+ authorizationUrl: string,
587
+ expectedState: string,
588
+ onAuthorizationInput?: AuthenticateOptions["onAuthorizationInput"],
589
+ signal?: AbortSignal,
590
+ ): Promise<AuthorizationResponse> {
591
+ if (!onAuthorizationInput) {
592
+ return {
593
+ input: await abortable(callbackPromise, signal),
594
+ source: "callback",
595
+ }
596
+ }
597
+
598
+ const inputController = new AbortController()
599
+ try {
600
+ const response = await abortable(Promise.race([
601
+ callbackPromise.then((input) => ({ input, source: "callback" as const })),
602
+ onAuthorizationInput(authorizationUrl, inputController.signal).then((input) => ({
603
+ input,
604
+ source: "manual" as const,
605
+ })),
606
+ ]), signal)
607
+
608
+ if (response.source === "callback") return response
609
+ if (!response.input?.trim()) throw new Error("OAuth authentication cancelled")
610
+ if (!getSearchParamsFromInput(response.input.trim())) {
611
+ throw new Error("Paste the full OAuth callback URL, including its code and state parameters")
612
+ }
613
+ return {
614
+ input: parseAuthorizationRedirectInput(response.input, expectedState),
615
+ source: "manual",
616
+ }
617
+ } finally {
618
+ inputController.abort()
619
+ }
620
+ }
621
+
571
622
  /**
572
623
  * Complete OAuth authentication from manual user input.
573
624
  */
@@ -728,12 +779,22 @@ export async function authenticate(
728
779
  console.warn(`MCP Auth: Failed to open browser for ${serverName}; waiting for manual callback`, { error })
729
780
  }
730
781
 
731
- const callbackResult = await abortable(callbackPromise, signal)
782
+ const authorizationResponse = await waitForAuthorizationResponse(
783
+ callbackPromise,
784
+ authorizationUrl,
785
+ oauthState,
786
+ options.onAuthorizationInput,
787
+ signal,
788
+ )
789
+ if (authorizationResponse.source === "manual") {
790
+ cancelPendingCallback(oauthState)
791
+ }
732
792
 
733
- // The callback server accepted only the flow-local reserved state.
793
+ // The callback server accepted only the flow-local reserved state. Manual
794
+ // input is checked against the same state before token exchange.
734
795
  throwIfAborted(signal)
735
796
 
736
- return await completeAuth(serverName, callbackResult, {
797
+ return await completeAuth(serverName, authorizationResponse.input, {
737
798
  ...options,
738
799
  ...(signal ? { signal } : {}),
739
800
  runtime,
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-mcp-adapter",
3
- "version": "2.22.0",
3
+ "version": "2.23.0",
4
4
  "description": "MCP (Model Context Protocol) adapter extension for Pi coding agent",
5
5
  "type": "module",
6
6
  "types": "./index.ts",
@@ -4,6 +4,15 @@ All notable changes to this project will be documented in this file.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.22.0] - 2026-08-11
8
+
9
+ ### Added
10
+ - Added Bocha web search provider support. Thanks @jingyulong for PR #243.
11
+ - Added `maxInlineContentChars` to configure the direct content and stored-content slice limit, with a 200,000-character maximum. Thanks @be4zad for issue #244.
12
+
13
+ ### Fixed
14
+ - Hardened the fetched-content cache against symlink traversal and unsafe permissions, and bounded it to 128 entries and 128 MiB with oldest-entry eviction. Thanks `@HerbertGao` for issue #240 and PR #241.
15
+
7
16
  ## [0.21.0] - 2026-08-10
8
17
 
9
18
  ### Added
@@ -4,7 +4,7 @@
4
4
 
5
5
  # Pi Web Access
6
6
 
7
- **Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, AnySearch, xAI/Grok, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, or bring your own API keys.**
7
+ **Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI/Grok, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, or bring your own API keys.**
8
8
 
9
9
  [![npm version](https://img.shields.io/npm/v/pi-web-access?style=for-the-badge)](https://www.npmjs.com/package/pi-web-access)
10
10
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg?style=for-the-badge)](https://opensource.org/licenses/MIT)
@@ -14,11 +14,11 @@
14
14
 
15
15
  ## Why Pi Web Access
16
16
 
17
- **Zero Config** — Works out of the box with Exa MCP (no API key needed). If you're signed into Pi with a Codex subscription, OpenAI web search can reuse that auth. Add API keys for OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, SerpBase, Exa, Perplexity, or Gemini API for more control; configure a self-hosted SearXNG endpoint for private search; or opt into browser-cookie access for Gemini Web.
17
+ **Zero Config** — Works out of the box with Exa MCP (no API key needed). If you're signed into Pi with a Codex subscription, OpenAI web search can reuse that auth. Add API keys for OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, SerpBase, Exa, Perplexity, or Gemini API for more control; configure a self-hosted SearXNG endpoint for private search; or opt into browser-cookie access for Gemini Web.
18
18
 
19
19
  **Video Understanding** — Point it at a YouTube video or local screen recording and ask questions about what's on screen. Full transcripts, visual descriptions, and frame extraction at exact timestamps.
20
20
 
21
- **Smart Fallbacks** — Every capability has a fallback chain. Search tries configured SearXNG first for local/private search, then OpenAI when suitable and available, Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, Perplexity, Gemini API, and Gemini Web when browser cookies are enabled. With no SearXNG configured, the existing zero-config order is unchanged. YouTube tries Gemini Web when enabled, then API, then Perplexity. Blocked pages try configured self-hosted Firecrawl first. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
21
+ **Smart Fallbacks** — Every capability has a fallback chain. Search tries configured SearXNG first for local/private search, then OpenAI when suitable and available, Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, and Gemini Web when browser cookies are enabled. With no SearXNG configured, the existing zero-config order is unchanged. YouTube tries Gemini Web when enabled, then API, then Perplexity. Blocked pages try configured self-hosted Firecrawl first. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
22
22
 
23
23
  **GitHub Cloning** — GitHub URLs are cloned locally instead of scraped. The agent gets real file contents and a local path to explore, not rendered HTML.
24
24
 
@@ -40,6 +40,7 @@ Works immediately with no API keys — Exa MCP provides zero-config search. If P
40
40
  "searchinfinityApiKey": "...",
41
41
  "queritApiKey": "...",
42
42
  "jinaApiKey": "jina_...",
43
+ "bochaApiKey": "sk-...",
43
44
  "perplexityApiKey": "pplx-...",
44
45
  "geminiApiKey": "AIza..."
45
46
  }
@@ -95,7 +96,7 @@ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on sc
95
96
 
96
97
  ### web_search
97
98
 
98
- Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, AnySearch, xAI, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, or Gemini. Returns a synthesized answer with source citations.
99
+ Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, or Gemini. Returns a synthesized answer with source citations.
99
100
 
100
101
  ```typescript
101
102
  web_search({ query: "rust async programming" })
@@ -116,7 +117,7 @@ web_search({ queries: ["query 1", "query 2"], workflow: "auto-summary" })
116
117
  | `numResults` | Results per query (default: 5, max: 20) |
117
118
  | `recencyFilter` | `day`, `week`, `month`, or `year` |
118
119
  | `domainFilter` | Limit to domains (prefix with `-` to exclude) |
119
- | `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase simultaneously; otherwise `openai`, `brave`, `parallel`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `jina`, `serpdive`, `kagi`, `ollama`, `anysearch`, `xai`, `brightdata`, `serpbase`, `searxng`, `duckduckgo`, `exa`, `perplexity`, or `gemini` (auto-selects when no provider or routing is configured; DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are explicit-only) |
120
+ | `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase simultaneously; otherwise `openai`, `brave`, `parallel`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `jina`, `serpdive`, `kagi`, `bocha`, `ollama`, `anysearch`, `xai`, `brightdata`, `serpbase`, `searxng`, `duckduckgo`, `exa`, `perplexity`, or `gemini` (auto-selects when no provider or routing is configured; DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are explicit-only) |
120
121
  | `includeContent` | Fetch full page content from sources in background |
121
122
  | `workflow` | `none` (skip curator), `summary-review` (open curator and auto-generate a summary draft, default), or `auto-summary` (generate a summary without opening the curator) |
122
123
 
@@ -148,7 +149,7 @@ fetch_content({ url: "https://example.com/diagram.png" })
148
149
 
149
150
  ### get_search_content
150
151
 
151
- Retrieve stored content from previous searches or fetches. Fetched URL content is stored in full in a private `web-search-cache` directory under the Pi config directory, not in the session JSONL. This includes `fetch_content` answer mode, which stores the original page content. Cache entries use the same one-hour lifetime as in-session result IDs. Use `findText` to locate bounded matching passages without paging through a large page, or use `offset` and `limit` to retrieve slices intentionally.
152
+ Retrieve stored content from previous searches or fetches. Fetched URL content is stored in full in a private `web-search-cache` directory under the Pi config directory, not in the session JSONL. This includes `fetch_content` answer mode, which stores the original page content. The cache has a one-hour lifetime and fixed limits of 128 entries and 128 MiB; when either limit is reached, the oldest entries are removed first. On macOS and Linux the cache directory and files are kept at permissions `0700` and `0600`, respectively. Use `findText` to locate bounded matching passages without paging through a large page, or use `offset` and `limit` to retrieve slices intentionally.
152
153
 
153
154
  ```typescript
154
155
  get_search_content({ responseId: "abc123", urlIndex: 0 })
@@ -158,11 +159,11 @@ get_search_content({ responseId: "abc123", urlIndex: 0, findText: "installation"
158
159
  get_search_content({ responseId: "abc123", urlIndex: 0, findText: ["timeout", "retry"], findMode: "fuzzy" })
159
160
  ```
160
161
 
161
- `findMode` supports `exact`, `case-insensitive` (default), and `fuzzy`. Finder output is capped at 20,000 characters with match counts and nearby context. `findText` cannot be combined with `offset` or `limit`.
162
+ `findMode` supports `exact`, `case-insensitive` (default), and `fuzzy`. Finder output is capped at 20,000 characters with match counts and nearby context. `findText` cannot be combined with `offset` or `limit`. The default `limit` and maximum permitted `limit` use `maxInlineContentChars`.
162
163
 
163
164
  ### source_check
164
165
 
165
- Check a claim and return a machine-readable artifact with exact passage citations. Search results are deduplicated and capped at 20 sources; `fetchContent` fetches at most 5 pages, while stored and retrieved content remains subject to the existing 30,000-character `offset`/`limit` bounds.
166
+ Check a claim and return a machine-readable artifact with exact passage citations. Search results are deduplicated and capped at 20 sources; `fetchContent` fetches at most 5 pages, while stored and retrieved content remains subject to the configured `maxInlineContentChars` `offset`/`limit` bounds.
166
167
 
167
168
  ```typescript
168
169
  source_check({ claim: "The API supports streaming responses" })
@@ -377,6 +378,7 @@ Config defaults to `~/.pi/web-search.json`, or `web-search.json` under `PI_CODIN
377
378
  "searchModel": "gemini-3.6-flash",
378
379
  "summaryModel": "anthropic/claude-haiku-4-5",
379
380
  "summaryGenerationDeadlineMs": 30000,
381
+ "maxInlineContentChars": 30000,
380
382
  "workflow": "summary-review",
381
383
  "curatorTimeoutSeconds": 20,
382
384
  "curatorRemote": {
@@ -421,7 +423,7 @@ Config defaults to `~/.pi/web-search.json`, or `web-search.json` under `PI_CODIN
421
423
  }
422
424
  ```
423
425
 
424
- All provider API-key fields (`openaiApiKey`, `braveApiKey`, `parallelApiKey`, `tinyfishApiKey`, `search1apiApiKey`, `searchinfinityApiKey`, `queritApiKey`, `tavilyApiKey`, `jinaApiKey`, `serpdiveApiKey`, `kagiApiKey`, `ollamaApiKey`, `serpbaseApiKey`, `anysearchApiKey`, `xaiApiKey`, `brightdataApiKey`, `firecrawlApiKey`, `exaApiKey`, `perplexityApiKey`, `geminiApiKey`, `datalabApiKey`, and `cloudflareApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
426
+ All provider API-key fields (`openaiApiKey`, `braveApiKey`, `parallelApiKey`, `tinyfishApiKey`, `search1apiApiKey`, `searchinfinityApiKey`, `queritApiKey`, `tavilyApiKey`, `jinaApiKey`, `serpdiveApiKey`, `kagiApiKey`, `bochaApiKey`, `ollamaApiKey`, `serpbaseApiKey`, `anysearchApiKey`, `xaiApiKey`, `brightdataApiKey`, `firecrawlApiKey`, `exaApiKey`, `perplexityApiKey`, `geminiApiKey`, `datalabApiKey`, and `cloudflareApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
425
427
 
426
428
  ```json
427
429
  {
@@ -456,7 +458,7 @@ Bright Data Web Unlocker is a paid `fetch_content` fallback after Parallel and b
456
458
 
457
459
  **SerpBase.** Set `serpbaseApiKey` or `SERPBASE_API_KEY` and select `provider: "serpbase"` to query SerpBase's Google Search Results API. SerpBase is explicit-only: it is never chosen by `auto` and never participates in `provider: "all"`, because each request can consume paid Google SERP credits. Domain filters are sent as Google `site:` clauses and reapplied locally; recency maps to Google's `tbs` time filter.
458
460
 
459
- Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"jina"`, `"serpdive"`, `"kagi"`, `"ollama"`, `"anysearch"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. AnySearch, xAI, Bright Data, and SerpBase are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `chromeProfile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
461
+ Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `BOCHA_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"jina"`, `"serpdive"`, `"kagi"`, `"bocha"`, `"ollama"`, `"anysearch"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. AnySearch, xAI, Bright Data, and SerpBase are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `chromeProfile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `maxInlineContentChars` sets the direct `fetch_content` content slice and the default and maximum `get_search_content` slice. It defaults to `30000`, must be a positive integer, and is capped at `200000`; full fetched content remains stored for later retrieval. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
460
462
 
461
463
  ### All providers
462
464
 
@@ -0,0 +1,221 @@
1
+ import { existsSync, readFileSync } from "node:fs";
2
+ import { activityMonitor } from "./activity.ts";
3
+ import type { SearchOptions, SearchResponse } from "./perplexity.ts";
4
+ import { hasCredentialSource, redactCredential, resolveCredential } from "./credential-source.ts";
5
+ import { getWebSearchConfigPath } from "./utils.ts";
6
+
7
+ const BOCHA_SEARCH_URL = "https://api.bochaai.com/v1/web-search";
8
+ const CONFIG_PATH = getWebSearchConfigPath();
9
+ const SEARCH_TIMEOUT_MS = 60_000;
10
+
11
+ interface WebSearchConfig {
12
+ bochaApiKey?: unknown;
13
+ }
14
+
15
+ let cachedConfig: WebSearchConfig | null = null;
16
+
17
+ function loadConfig(): WebSearchConfig {
18
+ if (cachedConfig) return cachedConfig;
19
+ if (!existsSync(CONFIG_PATH)) {
20
+ cachedConfig = {};
21
+ return cachedConfig;
22
+ }
23
+ const raw = readFileSync(CONFIG_PATH, "utf-8");
24
+ let parsed: unknown;
25
+ try {
26
+ parsed = JSON.parse(raw);
27
+ } catch (err) {
28
+ const message = err instanceof Error ? err.message : String(err);
29
+ throw new Error(`Failed to parse ${CONFIG_PATH}: ${message}`);
30
+ }
31
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
32
+ throw new Error(`Invalid config in ${CONFIG_PATH}: expected a JSON object`);
33
+ }
34
+ cachedConfig = parsed as WebSearchConfig;
35
+ return cachedConfig;
36
+ }
37
+
38
+ async function getApiKey(signal?: AbortSignal): Promise<string | null> {
39
+ return resolveCredential({
40
+ provider: "Bocha",
41
+ configuredValue: loadConfig().bochaApiKey,
42
+ environmentValue: process.env.BOCHA_API_KEY,
43
+ signal,
44
+ });
45
+ }
46
+
47
+ async function requireApiKey(signal?: AbortSignal): Promise<string> {
48
+ const apiKey = await getApiKey(signal);
49
+ if (!apiKey) {
50
+ throw new Error(
51
+ "Bocha API key not found. Either:\n" +
52
+ ` 1. Create ${CONFIG_PATH} with { "bochaApiKey": "your-key" }\n` +
53
+ " 2. Set BOCHA_API_KEY environment variable\n" +
54
+ "Create a key at https://open.bochaai.com/",
55
+ );
56
+ }
57
+ return apiKey;
58
+ }
59
+
60
+ function normalizeCount(value: number | undefined): number {
61
+ if (typeof value !== "number" || !Number.isFinite(value)) return 8;
62
+ return Math.max(1, Math.min(Math.floor(value), 20));
63
+ }
64
+
65
+ function mapFreshness(value: SearchOptions["recencyFilter"]): string {
66
+ switch (value) {
67
+ case "day": return "oneDay";
68
+ case "week": return "oneWeek";
69
+ case "month": return "oneMonth";
70
+ case "year": return "oneYear";
71
+ default: return "noLimit";
72
+ }
73
+ }
74
+
75
+ interface DomainFilters {
76
+ include: string[];
77
+ exclude: string[];
78
+ }
79
+
80
+ function normalizeDomain(value: string): string | null {
81
+ let input = value.trim().toLowerCase();
82
+ if (!input) return null;
83
+ if (input.startsWith("-")) input = input.slice(1).trim();
84
+ if (!input) return null;
85
+ try {
86
+ const parsed = input.includes("://") ? new URL(input) : new URL(`https://${input}`);
87
+ input = parsed.hostname;
88
+ } catch {
89
+ input = input.split("/")[0]?.split(":")[0] ?? "";
90
+ }
91
+ input = input.replace(/^\.+|\.+$/g, "");
92
+ return /^[a-z0-9][a-z0-9.-]*\.[a-z]{2,}$/i.test(input) ? input : null;
93
+ }
94
+
95
+ function parseDomainFilter(domainFilter: string[] | undefined): DomainFilters {
96
+ const filters: DomainFilters = { include: [], exclude: [] };
97
+ for (const raw of domainFilter ?? []) {
98
+ const domain = normalizeDomain(raw);
99
+ if (!domain) continue;
100
+ const target = raw.trim().startsWith("-") ? filters.exclude : filters.include;
101
+ if (!target.includes(domain)) target.push(domain);
102
+ }
103
+ return filters;
104
+ }
105
+
106
+ function passesDomainFilters(url: string, filters: DomainFilters): boolean {
107
+ if (filters.include.length === 0 && filters.exclude.length === 0) return true;
108
+ let hostname: string;
109
+ try {
110
+ hostname = new URL(url).hostname.toLowerCase();
111
+ } catch {
112
+ return false;
113
+ }
114
+ const matches = (domain: string) => hostname === domain || hostname.endsWith(`.${domain}`);
115
+ if (filters.exclude.some(matches)) return false;
116
+ return filters.include.length === 0 || filters.include.some(matches);
117
+ }
118
+
119
+ function errorMessage(err: unknown): string {
120
+ return err instanceof Error ? err.message : String(err);
121
+ }
122
+
123
+ function invalidResponse(message: string): Error {
124
+ return new Error(`Bocha API returned invalid response: ${message}`);
125
+ }
126
+
127
+ function firstString(...values: unknown[]): string | null {
128
+ for (const value of values) {
129
+ if (typeof value === "string" && value.trim()) return value.trim();
130
+ }
131
+ return null;
132
+ }
133
+
134
+ function parseSearchResponse(value: unknown): { results: SearchResponse["results"] } {
135
+ if (!value || typeof value !== "object" || Array.isArray(value)) throw invalidResponse("expected an object envelope");
136
+ const envelope = value as Record<string, unknown>;
137
+ if (envelope.code !== undefined && Number(envelope.code) !== 200) {
138
+ throw invalidResponse(`code ${String(envelope.code)}: ${firstString(envelope.msg) ?? "unknown error"}`);
139
+ }
140
+ const data = envelope.data;
141
+ const pages = (typeof data === "object" && data !== null && !Array.isArray(data))
142
+ ? (data as Record<string, unknown>).webPages
143
+ : undefined;
144
+ const items = (typeof pages === "object" && pages !== null && !Array.isArray(pages))
145
+ ? (pages as Record<string, unknown>).value
146
+ : undefined;
147
+ if (!Array.isArray(items)) throw invalidResponse("missing data.webPages.value array");
148
+ const results: SearchResponse["results"] = [];
149
+ for (const item of items) {
150
+ if (!item || typeof item !== "object" || Array.isArray(item)) continue;
151
+ const entry = item as Record<string, unknown>;
152
+ const url = firstString(entry.url, entry.link, entry.href);
153
+ if (!url) continue;
154
+ const title = firstString(entry.title, entry.name) ?? url;
155
+ const snippet = firstString(entry.summary, entry.snippet, entry.description, entry.content) ?? "";
156
+ results.push({ title, url, snippet });
157
+ }
158
+ return { results };
159
+ }
160
+
161
+ function buildAnswer(results: SearchResponse["results"]): string {
162
+ return results.map((result) => result.snippet
163
+ ? `${result.snippet}\nSource: ${result.title} (${result.url})`
164
+ : `Source: ${result.title} (${result.url})`).join("\n\n");
165
+ }
166
+
167
+ export function isBochaAvailable(): boolean {
168
+ return hasCredentialSource({ provider: "Bocha", configuredValue: loadConfig().bochaApiKey, environmentValue: process.env.BOCHA_API_KEY });
169
+ }
170
+
171
+ export async function searchWithBocha(query: string, options: SearchOptions = {}): Promise<SearchResponse> {
172
+ const apiKey = await requireApiKey(options.signal);
173
+ const numResults = normalizeCount(options.numResults);
174
+ const filters = parseDomainFilter(options.domainFilter);
175
+ const activityId = activityMonitor.logStart({ type: "api", query });
176
+ let response: Response;
177
+ try {
178
+ response = await fetch(BOCHA_SEARCH_URL, {
179
+ method: "POST",
180
+ headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json", Accept: "application/json" },
181
+ body: JSON.stringify({ query, count: numResults, freshness: mapFreshness(options.recencyFilter), summary: true }),
182
+ signal: options.signal ? AbortSignal.any([AbortSignal.timeout(SEARCH_TIMEOUT_MS), options.signal]) : AbortSignal.timeout(SEARCH_TIMEOUT_MS),
183
+ });
184
+ } catch (err) {
185
+ const message = errorMessage(err);
186
+ const redactedMessage = redactCredential(message, apiKey);
187
+ if (redactedMessage.toLowerCase().includes("abort")) activityMonitor.logComplete(activityId, 0);
188
+ else activityMonitor.logError(activityId, redactedMessage);
189
+ if (redactedMessage === message) throw err;
190
+ const redactedError = new Error(redactedMessage);
191
+ if (err instanceof Error) redactedError.name = err.name;
192
+ throw redactedError;
193
+ }
194
+ if (!response.ok) {
195
+ activityMonitor.logComplete(activityId, response.status);
196
+ const errorText = redactCredential(await response.text(), apiKey);
197
+ throw new Error(`Bocha API error ${response.status}: ${errorText.slice(0, 300)}`);
198
+ }
199
+ let rawData: unknown;
200
+ try {
201
+ rawData = await response.json();
202
+ } catch (err) {
203
+ activityMonitor.logComplete(activityId, response.status);
204
+ throw new Error(`Bocha API returned invalid JSON: ${errorMessage(err)}`);
205
+ }
206
+ let parsed: { results: SearchResponse["results"] };
207
+ try {
208
+ parsed = parseSearchResponse(rawData);
209
+ } catch (err) {
210
+ const message = errorMessage(err);
211
+ const redactedMessage = redactCredential(message, apiKey);
212
+ activityMonitor.logError(activityId, redactedMessage);
213
+ if (redactedMessage === message) throw err;
214
+ const redactedError = new Error(redactedMessage);
215
+ if (err instanceof Error) redactedError.name = err.name;
216
+ throw redactedError;
217
+ }
218
+ activityMonitor.logComplete(activityId, response.status);
219
+ const results = parsed.results.filter((result) => passesDomainFilters(result.url, filters)).slice(0, numResults);
220
+ return { answer: buildAnswer(results), results };
221
+ }
@@ -8,7 +8,7 @@ function safeInlineJSON(data: unknown): string {
8
8
  }
9
9
 
10
10
  function buildProviderButtons(
11
- available: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
11
+ available: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
12
12
  selected: string,
13
13
  hasInitialQueries: boolean,
14
14
  ): string {
@@ -26,6 +26,7 @@ function buildProviderButtons(
26
26
  { value: "jina", label: "Jina", available: available.jina },
27
27
  { value: "serpdive", label: "SERPdive", available: available.serpdive },
28
28
  { value: "kagi", label: "Kagi", available: available.kagi },
29
+ { value: "bocha", label: "Bocha", available: available.bocha },
29
30
  { value: "ollama", label: "Ollama", available: available.ollama },
30
31
  { value: "searxng", label: "SearXNG", available: available.searxng },
31
32
  { value: "duckduckgo", label: "DuckDuckGo", available: available.duckduckgo },
@@ -53,7 +54,7 @@ export function generateCuratorPage(
53
54
  queries: string[],
54
55
  sessionToken: string,
55
56
  timeout: number,
56
- availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
57
+ availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
57
58
  defaultProvider: string,
58
59
  searchProvider: string,
59
60
  summaryModels: Array<{ value: string; label: string }>,
@@ -1454,7 +1455,7 @@ const SCRIPT = `(function() {
1454
1455
  var token = DATA.sessionToken;
1455
1456
  var timeoutSec = DATA.timeout;
1456
1457
  var queries = Array.isArray(DATA.queries) ? DATA.queries : [];
1457
- var providers = ["all", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "anysearch", "xai", "brightdata", "serpbase"];
1458
+ var providers = ["all", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "bocha", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "anysearch", "xai", "brightdata", "serpbase"];
1458
1459
  var availProviders = DATA.availableProviders && typeof DATA.availableProviders === "object" ? DATA.availableProviders : {};
1459
1460
  var workflow = "summary-review";
1460
1461
  var initialDefaultProvider = typeof DATA.defaultProvider === "string" ? DATA.defaultProvider : "exa";
@@ -1669,6 +1670,7 @@ const SCRIPT = `(function() {
1669
1670
  if (provider === "jina") return "Jina";
1670
1671
  if (provider === "serpdive") return "SERPdive";
1671
1672
  if (provider === "kagi") return "Kagi";
1673
+ if (provider === "bocha") return "Bocha";
1672
1674
  if (provider === "ollama") return "Ollama";
1673
1675
  if (provider === "searxng") return "SearXNG";
1674
1676
  if (provider === "duckduckgo") return "DuckDuckGo";
@@ -18,7 +18,7 @@ export interface CuratorServerOptions {
18
18
  queries: string[];
19
19
  sessionToken: string;
20
20
  timeout: number;
21
- availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean };
21
+ availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean };
22
22
  defaultProvider: string;
23
23
  searchProvider: string;
24
24
  summaryModels: Array<{ value: string; label: string }>;
@@ -283,6 +283,7 @@ export function startCuratorServer(
283
283
  if (provider === "jina") return availableProviders.jina;
284
284
  if (provider === "serpdive") return availableProviders.serpdive;
285
285
  if (provider === "kagi") return availableProviders.kagi;
286
+ if (provider === "bocha") return availableProviders.bocha;
286
287
  if (provider === "ollama") return availableProviders.ollama;
287
288
  if (provider === "searxng") return availableProviders.searxng;
288
289
  if (provider === "duckduckgo") return availableProviders.duckduckgo;
@@ -17,6 +17,7 @@ import { isTavilyAvailable, searchWithTavily } from "./tavily.ts";
17
17
  import { isJinaSearchAvailable, searchWithJina } from "./jina-search.ts";
18
18
  import { isSerpdiveAvailable, searchWithSerpdive } from "./serpdive.ts";
19
19
  import { isKagiAvailable, searchWithKagi } from "./kagi.ts";
20
+ import { isBochaAvailable, searchWithBocha } from "./bocha.ts";
20
21
  import { isOllamaAvailable, searchWithOllama } from "./ollama.ts";
21
22
  import { isSearXNGAvailable, searchWithSearXNG } from "./searxng.ts";
22
23
  import { isDuckDuckGoAvailable, searchWithDuckDuckGo } from "./duckduckgo.ts";
@@ -26,7 +27,7 @@ import { isBrightDataAvailable, searchWithBrightData } from "./brightdata.ts";
26
27
  import { isSerpBaseAvailable, searchWithSerpBase } from "./serpbase.ts";
27
28
  import { getWebSearchConfigPath } from "./utils.ts";
28
29
 
29
- export const RESOLVED_SEARCH_PROVIDERS = ["openai", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "searxng", "duckduckgo", "perplexity", "gemini", "exa", "serpdive", "kagi", "ollama", "anysearch", "xai", "brightdata", "serpbase"] as const;
30
+ export const RESOLVED_SEARCH_PROVIDERS = ["openai", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "searxng", "duckduckgo", "perplexity", "gemini", "exa", "serpdive", "kagi", "ollama", "anysearch", "xai", "brightdata", "serpbase", "bocha"] as const;
30
31
  export const SEARCH_PROVIDERS = ["auto", "all", ...RESOLVED_SEARCH_PROVIDERS] as const;
31
32
 
32
33
  export type ResolvedSearchProvider = typeof RESOLVED_SEARCH_PROVIDERS[number];
@@ -90,7 +91,7 @@ const CONFIG_PATH = getWebSearchConfigPath();
90
91
  const DEFAULT_SEARCH_MODEL = "gemini-3.6-flash";
91
92
  // Explicit-only providers (DuckDuckGo, AnySearch, xAI, Bright Data, SerpBase) are deliberately absent:
92
93
  // `all` must never fan out to an opt-in or paid provider without the user asking for it.
93
- const ALL_SEARCH_PROVIDERS: ResolvedSearchProvider[] = ["searxng", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "ollama", "perplexity", "gemini"];
94
+ const ALL_SEARCH_PROVIDERS: ResolvedSearchProvider[] = ["searxng", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "ollama", "perplexity", "gemini", "bocha"];
94
95
  const VALID_ROUTING_KINDS = ["transient", "quota", "network", "invalid-response"] as const;
95
96
 
96
97
  type SearchConfig = {
@@ -306,6 +307,7 @@ async function searchWithResolvedProvider(
306
307
  if (provider === "jina") return { ...(await searchWithJina(query, options)), provider };
307
308
  if (provider === "serpdive") return { ...(await searchWithSerpdive(query, options)), provider };
308
309
  if (provider === "kagi") return { ...(await searchWithKagi(query, options)), provider };
310
+ if (provider === "bocha") return { ...(await searchWithBocha(query, options)), provider };
309
311
  if (provider === "ollama") return { ...(await searchWithOllama(query, options)), provider };
310
312
  if (provider === "anysearch") return { ...(await searchWithAnySearch(query, options)), provider };
311
313
  if (provider === "xai") return { ...(await searchWithXai(query, options, options.extensionContext)), provider };
@@ -341,6 +343,7 @@ async function isResolvedProviderAvailable(provider: ResolvedSearchProvider, opt
341
343
  if (provider === "jina") return isJinaSearchAvailable();
342
344
  if (provider === "serpdive") return isSerpdiveAvailable();
343
345
  if (provider === "kagi") return isKagiAvailable();
346
+ if (provider === "bocha") return isBochaAvailable();
344
347
  if (provider === "ollama") return isOllamaAvailable();
345
348
  if (provider === "anysearch") return isAnySearchAvailable();
346
349
  if (provider === "xai") return isXaiSearchAvailable(options.extensionContext);
@@ -363,6 +366,7 @@ function providerLabel(provider: ResolvedSearchProvider): string {
363
366
  if (provider === "searxng") return "SearXNG";
364
367
  if (provider === "duckduckgo") return "DuckDuckGo";
365
368
  if (provider === "kagi") return "Kagi";
369
+ if (provider === "bocha") return "Bocha";
366
370
  if (provider === "ollama") return "Ollama";
367
371
  if (provider === "xai") return "xAI";
368
372
  if (provider === "brightdata") return "Bright Data";
@@ -626,6 +630,16 @@ export async function search(query: string, options: FullSearchOptions = {}): Pr
626
630
  }
627
631
  }
628
632
 
633
+ if (isBochaAvailable()) {
634
+ try {
635
+ const result = await searchWithBocha(query, options);
636
+ return { ...result, provider: "bocha" };
637
+ } catch (err) {
638
+ if (isAbortError(err)) throw err;
639
+ fallbackErrors.push(`Bocha: ${errorMessage(err)}`);
640
+ }
641
+ }
642
+
629
643
  if (isOllamaAvailable()) {
630
644
  try {
631
645
  const result = await searchWithOllama(query, options);
@@ -661,8 +675,8 @@ export async function search(query: string, options: FullSearchOptions = {}): Pr
661
675
  throw new Error(
662
676
  "No search provider available. Either:\n" +
663
677
  " 1. Use /login to sign in with a Codex subscription for OpenAI web search\n" +
664
- ` 2. Set openaiApiKey, braveApiKey, parallelApiKey, tinyfishApiKey, search1apiApiKey, searchinfinityApiKey, queritApiKey, tavilyApiKey, jinaApiKey, serpdiveApiKey, kagiApiKey, ollamaApiKey, searxngBaseUrl, perplexityApiKey, exaApiKey, geminiApiKey, or cloudflareApiKey in ${CONFIG_PATH}\n` +
665
- " 3. Set OPENAI_API_KEY, BRAVE_API_KEY, PARALLEL_API_KEY, TINYFISH_API_KEY, SEARCH1API_KEY, SEARCHINFINITY_API_KEY, QUERIT_API_KEY, TAVILY_API_KEY, JINA_API_KEY, SERPDIVE_API_KEY, KAGI_API_KEY, OLLAMA_API_KEY, SEARXNG_BASE_URL, EXA_API_KEY, PERPLEXITY_API_KEY, GEMINI_API_KEY, or CLOUDFLARE_API_KEY env vars\n" +
678
+ ` 2. Set openaiApiKey, braveApiKey, parallelApiKey, tinyfishApiKey, search1apiApiKey, searchinfinityApiKey, queritApiKey, tavilyApiKey, jinaApiKey, serpdiveApiKey, kagiApiKey, ollamaApiKey, searxngBaseUrl, perplexityApiKey, exaApiKey, geminiApiKey, bochaApiKey, or cloudflareApiKey in ${CONFIG_PATH}\n` +
679
+ " 3. Set OPENAI_API_KEY, BRAVE_API_KEY, PARALLEL_API_KEY, TINYFISH_API_KEY, SEARCH1API_KEY, SEARCHINFINITY_API_KEY, QUERIT_API_KEY, TAVILY_API_KEY, JINA_API_KEY, SERPDIVE_API_KEY, KAGI_API_KEY, BOCHA_API_KEY, OLLAMA_API_KEY, SEARXNG_BASE_URL, EXA_API_KEY, PERPLEXITY_API_KEY, GEMINI_API_KEY, or CLOUDFLARE_API_KEY env vars\n" +
666
680
  " 4. Set GOOGLE_GEMINI_BASE_URL with CLOUDFLARE_API_KEY for Cloudflare AI Gateway routing\n" +
667
681
  " 5. Sign into gemini.google.com in a supported Chromium-based browser\n" +
668
682
  " 6. Explicitly select provider: \"anysearch\" for anonymous AnySearch, \"xai\" for Grok, \"brightdata\" with brightdataSerpZone for paid Bright Data SERP, or \"serpbase\" with serpbaseApiKey for paid Google SERP"
@@ -53,6 +53,7 @@ import { isTavilyAvailable } from "./tavily.ts";
53
53
  import { isJinaSearchAvailable } from "./jina-search.ts";
54
54
  import { isSerpdiveAvailable } from "./serpdive.ts";
55
55
  import { isKagiAvailable } from "./kagi.ts";
56
+ import { isBochaAvailable } from "./bocha.ts";
56
57
  import { isOllamaAvailable } from "./ollama.ts";
57
58
  import { isSearXNGAvailable } from "./searxng.ts";
58
59
  import { isDuckDuckGoAvailable } from "./duckduckgo.ts";
@@ -124,6 +125,7 @@ interface WebSearchConfig {
124
125
  curatorRemote?: unknown;
125
126
  summaryModel?: string;
126
127
  summaryGenerationDeadlineMs?: unknown;
128
+ maxInlineContentChars?: unknown;
127
129
  webSearch?: {
128
130
  enabled?: boolean;
129
131
  };
@@ -160,6 +162,7 @@ interface ProviderAvailability {
160
162
  exa: boolean;
161
163
  gemini: boolean;
162
164
  kagi: boolean;
165
+ bocha: boolean;
163
166
  ollama: boolean;
164
167
  anysearch: boolean;
165
168
  xai: boolean;
@@ -384,6 +387,7 @@ async function getProviderAvailability(ctx: ExtensionContext): Promise<ProviderA
384
387
  jina: isJinaSearchAvailable(),
385
388
  serpdive: isSerpdiveAvailable(),
386
389
  kagi: isKagiAvailable(),
390
+ bocha: isBochaAvailable(),
387
391
  ollama: isOllamaAvailable(),
388
392
  searxng: isSearXNGAvailable(),
389
393
  duckduckgo: isDuckDuckGoAvailable(),
@@ -440,6 +444,7 @@ function firstAvailableProvider(available: ProviderAvailability, preferOpenAI: b
440
444
  if (available.jina) return "jina";
441
445
  if (available.serpdive) return "serpdive";
442
446
  if (available.kagi) return "kagi";
447
+ if (available.bocha) return "bocha";
443
448
  if (available.ollama) return "ollama";
444
449
  if (available.perplexity) return "perplexity";
445
450
  if (available.gemini) return "gemini";
@@ -500,6 +505,9 @@ function resolveProvider(
500
505
  if (provider === "kagi" && !available.kagi) {
501
506
  return firstAvailableProvider(available, preferOpenAI, "kagi");
502
507
  }
508
+ if (provider === "bocha" && !available.bocha) {
509
+ return firstAvailableProvider(available, preferOpenAI, "bocha");
510
+ }
503
511
  if (provider === "ollama" && !available.ollama) {
504
512
  return firstAvailableProvider(available, preferOpenAI, "ollama");
505
513
  }
@@ -555,15 +563,22 @@ interface PendingCurate {
555
563
  }
556
564
 
557
565
 
558
- const MAX_INLINE_CONTENT = 30000; // Content returned directly to agent
559
- const DEFAULT_CONTENT_SLICE_LENGTH = MAX_INLINE_CONTENT;
560
- const MAX_CONTENT_SLICE_LENGTH = MAX_INLINE_CONTENT;
566
+ const DEFAULT_MAX_INLINE_CONTENT_CHARS = 30_000;
567
+ const MAX_INLINE_CONTENT_CHARS = 200_000;
568
+
569
+ function getMaxInlineContentChars(): number {
570
+ const value = loadConfig().maxInlineContentChars;
571
+ if (typeof value !== "number" || !Number.isFinite(value) || !Number.isInteger(value) || value <= 0) {
572
+ return DEFAULT_MAX_INLINE_CONTENT_CHARS;
573
+ }
574
+ return Math.min(value, MAX_INLINE_CONTENT_CHARS);
575
+ }
561
576
 
562
577
  function stripThumbnails(results: ExtractedContent[]): ExtractedContent[] {
563
578
  return results.map(({ thumbnail, frames, ...rest }) => rest);
564
579
  }
565
580
 
566
- function initialContentSlice(content: string): {
581
+ function initialContentSlice(content: string, maxChars: number): {
567
582
  text: string;
568
583
  endOffset: number;
569
584
  totalBytes: number;
@@ -571,10 +586,10 @@ function initialContentSlice(content: string): {
571
586
  shownBytes: number;
572
587
  shownLines: number;
573
588
  } {
574
- let endOffset = Math.min(content.length, MAX_INLINE_CONTENT);
589
+ let endOffset = Math.min(content.length, maxChars);
575
590
  if (endOffset < content.length) {
576
591
  const lineBreak = content.lastIndexOf("\n", endOffset);
577
- if (lineBreak >= Math.floor(MAX_INLINE_CONTENT * 0.8)) endOffset = lineBreak + 1;
592
+ if (lineBreak >= Math.floor(maxChars * 0.8)) endOffset = lineBreak + 1;
578
593
  }
579
594
  const text = content.slice(0, endOffset);
580
595
  return {
@@ -1621,7 +1636,7 @@ export default function (pi: ExtensionAPI) {
1621
1636
  name: toolNames.webSearch,
1622
1637
  label: "Web Search",
1623
1638
  description:
1624
- `Search the web using OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, SearXNG, DuckDuckGo, Exa, Perplexity, Gemini, AnySearch, xAI, Bright Data, or SerpBase. Pass a provider array to search only those providers simultaneously, or use provider "all" to search every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase. Returns an AI-synthesized answer with source citations. OpenAI search uses a Codex subscription or OpenAI API key; xAI search uses a SuperGrok/X Premium subscription or xAI API key. DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are available only when explicitly selected. For comprehensive research, prefer queries (plural) with 2-4 varied angles over a single query — each query gets its own synthesized answer, so varying phrasing and scope gives much broader coverage. When includeContent is true, full page content is fetched in the background. Searches auto-open the interactive browser curator and stream results live; set workflow to "none" to skip curation or "auto-summary" for a model-generated summary without the browser curator. The configured provider is used when provider is omitted or set to auto; omit provider unless explicitly overriding it. Without a configured provider, auto-selects OpenAI when suitable and available, then Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, Perplexity, Gemini API, or Gemini Web. When SearXNG is configured, it is preferred first for local/private search.`,
1639
+ `Search the web using OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, SearXNG, DuckDuckGo, Exa, Perplexity, Gemini, AnySearch, xAI, Bright Data, or SerpBase. Pass a provider array to search only those providers simultaneously, or use provider "all" to search every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase. Returns an AI-synthesized answer with source citations. OpenAI search uses a Codex subscription or OpenAI API key; xAI search uses a SuperGrok/X Premium subscription or xAI API key. DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are available only when explicitly selected. For comprehensive research, prefer queries (plural) with 2-4 varied angles over a single query — each query gets its own synthesized answer, so varying phrasing and scope gives much broader coverage. When includeContent is true, full page content is fetched in the background. Searches auto-open the interactive browser curator and stream results live; set workflow to "none" to skip curation or "auto-summary" for a model-generated summary without the browser curator. The configured provider is used when provider is omitted or set to auto; omit provider unless explicitly overriding it. Without a configured provider, auto-selects OpenAI when suitable and available, then Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, or Gemini Web. When SearXNG is configured, it is preferred first for local/private search.`,
1625
1640
  promptSnippet:
1626
1641
  "Use for web research questions. Prefer {queries:[...]} with 2-4 varied angles over a single query for broader coverage. Omit provider unless explicitly overriding the configured default.",
1627
1642
  parameters: Type.Object({
@@ -2402,7 +2417,7 @@ export default function (pi: ExtensionAPI) {
2402
2417
  }
2403
2418
 
2404
2419
  const fullLength = result.content.length;
2405
- const slice = initialContentSlice(result.content);
2420
+ const slice = initialContentSlice(result.content, getMaxInlineContentChars());
2406
2421
  const truncated = slice.endOffset < fullLength;
2407
2422
  let output = slice.text;
2408
2423
 
@@ -2614,7 +2629,7 @@ export default function (pi: ExtensionAPI) {
2614
2629
  url: Type.Optional(Type.String({ description: "Get content for this URL" })),
2615
2630
  urlIndex: Type.Optional(Type.Number({ description: "Get content for URL at index" })),
2616
2631
  offset: Type.Optional(Type.Number({ description: "Character offset for fetched URL content slices (default 0). Cannot be combined with findText." })),
2617
- limit: Type.Optional(Type.Number({ description: `Maximum characters to return for fetched URL content slices (default/max ${MAX_CONTENT_SLICE_LENGTH}). Cannot be combined with findText.` })),
2632
+ limit: Type.Optional(Type.Number({ description: "Maximum characters to return for fetched URL content slices (default and max are set by maxInlineContentChars). Cannot be combined with findText." })),
2618
2633
  findText: Type.Optional(Type.Union([
2619
2634
  Type.String({ minLength: 1, maxLength: 500 }),
2620
2635
  Type.Array(Type.String({ minLength: 1, maxLength: 500 }), { minItems: 1, maxItems: 10 }),
@@ -2646,13 +2661,14 @@ export default function (pi: ExtensionAPI) {
2646
2661
  };
2647
2662
  }
2648
2663
  const serialized = JSON.stringify(artifact, null, 2);
2664
+ const maxInlineContentChars = getMaxInlineContentChars();
2649
2665
  const offset = params.offset ?? 0;
2650
- const limit = params.limit ?? MAX_CONTENT_SLICE_LENGTH;
2666
+ const limit = params.limit ?? maxInlineContentChars;
2651
2667
  if (!Number.isInteger(offset) || offset < 0) {
2652
2668
  return { content: [{ type: "text", text: "offset must be a non-negative integer" }], details: { error: "Invalid offset", offset } };
2653
2669
  }
2654
- if (!Number.isInteger(limit) || limit <= 0 || limit > MAX_CONTENT_SLICE_LENGTH) {
2655
- return { content: [{ type: "text", text: `limit must be an integer from 1 to ${MAX_CONTENT_SLICE_LENGTH}` }], details: { error: "Invalid limit", limit, maxLimit: MAX_CONTENT_SLICE_LENGTH } };
2670
+ if (!Number.isInteger(limit) || limit <= 0 || limit > maxInlineContentChars) {
2671
+ return { content: [{ type: "text", text: `limit must be an integer from 1 to ${maxInlineContentChars}` }], details: { error: "Invalid limit", limit, maxLimit: maxInlineContentChars } };
2656
2672
  }
2657
2673
  if (offset > serialized.length) {
2658
2674
  return { content: [{ type: "text", text: `offset ${offset} is out of range (0-${serialized.length})` }], details: { error: "Offset out of range", offset, contentLength: serialized.length } };
@@ -2774,18 +2790,19 @@ export default function (pi: ExtensionAPI) {
2774
2790
  }
2775
2791
  }
2776
2792
 
2793
+ const maxInlineContentChars = getMaxInlineContentChars();
2777
2794
  const offset = params.offset ?? 0;
2778
- const limit = params.limit ?? DEFAULT_CONTENT_SLICE_LENGTH;
2795
+ const limit = params.limit ?? maxInlineContentChars;
2779
2796
  if (!Number.isInteger(offset) || offset < 0) {
2780
2797
  return {
2781
2798
  content: [{ type: "text", text: "offset must be a non-negative integer" }],
2782
2799
  details: { error: "Invalid offset", offset },
2783
2800
  };
2784
2801
  }
2785
- if (!Number.isInteger(limit) || limit <= 0 || limit > MAX_CONTENT_SLICE_LENGTH) {
2802
+ if (!Number.isInteger(limit) || limit <= 0 || limit > maxInlineContentChars) {
2786
2803
  return {
2787
- content: [{ type: "text", text: `limit must be an integer from 1 to ${MAX_CONTENT_SLICE_LENGTH}` }],
2788
- details: { error: "Invalid limit", limit, maxLimit: MAX_CONTENT_SLICE_LENGTH },
2804
+ content: [{ type: "text", text: `limit must be an integer from 1 to ${maxInlineContentChars}` }],
2805
+ details: { error: "Invalid limit", limit, maxLimit: maxInlineContentChars },
2789
2806
  };
2790
2807
  }
2791
2808
  if (offset > urlData.content.length) {
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-web-access",
3
- "version": "0.21.0",
3
+ "version": "0.22.0",
4
4
  "description": "Web search, URL fetching, GitHub repo cloning, PDF extraction, YouTube video understanding, and local video analysis for Pi coding agent. Supports OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, AnySearch, Bright Data SERP, SerpBase, SearXNG, Firecrawl extraction, Exa, Perplexity, and Gemini.",
5
5
  "type": "module",
6
6
  "scripts": {
@@ -1,4 +1,5 @@
1
- import { existsSync, mkdirSync, readFileSync, readdirSync, renameSync, statSync, unlinkSync, writeFileSync } from "node:fs";
1
+ import { closeSync, constants, fchmodSync, fstatSync, fsyncSync, lstatSync, mkdirSync, openSync, readFileSync, readdirSync, renameSync, type Stats, unlinkSync, writeFileSync } from "node:fs";
2
+ import { randomBytes } from "node:crypto";
2
3
  import { join } from "node:path";
3
4
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
4
5
  import type { ExtractedContent } from "./extract.ts";
@@ -9,8 +10,25 @@ const CACHE_TTL_MS = 60 * 60 * 1000;
9
10
  const FETCH_CACHE_DIR = "web-search-cache";
10
11
  const FETCH_CACHE_VERSION = 1;
11
12
  const CACHE_KEY_PATTERN = /^[A-Za-z0-9_-]+\.json$/;
13
+ const CACHE_TMP_PATTERN = /^[A-Za-z0-9_-]+\.json\.\d+\.\d+(?:\.[a-f0-9]{32})?\.tmp$/;
12
14
  const CACHE_ID_PATTERN = /^[A-Za-z0-9_-]+$/;
13
15
  const MAX_METADATA_TEXT = 8192;
16
+ const DEFAULT_CACHE_LIMITS = { maxEntries: 128, maxBytes: 128 * 1024 * 1024 };
17
+ const O_DIRECTORY = process.platform === "win32" ? 0 : (constants.O_DIRECTORY ?? 0);
18
+ const O_NOFOLLOW = process.platform === "win32" ? 0 : (constants.O_NOFOLLOW ?? 0);
19
+
20
+ interface FetchCacheLimits {
21
+ maxEntries: number;
22
+ maxBytes: number;
23
+ }
24
+
25
+ interface CacheFile {
26
+ name: string;
27
+ size: number;
28
+ mtimeMs: number;
29
+ dev: number;
30
+ ino: number;
31
+ }
14
32
 
15
33
  export interface QueryResultData {
16
34
  query: string;
@@ -120,17 +138,220 @@ function isInlineFetchData(data: StoredSearchData): data is StoredSearchData & {
120
138
  return data.type === "fetch" && Array.isArray(data.urls) && data.urls.every(isInlineFetchedUrl);
121
139
  }
122
140
 
123
- function writeFetchCache(data: StoredSearchData & { urls: ExtractedContent[] }): FetchCacheRef {
141
+ function cacheLimits(limits?: Partial<FetchCacheLimits>): FetchCacheLimits {
142
+ const resolved = {
143
+ maxEntries: limits?.maxEntries ?? DEFAULT_CACHE_LIMITS.maxEntries,
144
+ maxBytes: limits?.maxBytes ?? DEFAULT_CACHE_LIMITS.maxBytes,
145
+ };
146
+ if (!Number.isFinite(resolved.maxEntries) || !Number.isInteger(resolved.maxEntries) || resolved.maxEntries <= 0 ||
147
+ !Number.isFinite(resolved.maxBytes) || !Number.isInteger(resolved.maxBytes) || resolved.maxBytes <= 0) {
148
+ throw new Error("Fetched content cache limits must be finite positive integers");
149
+ }
150
+ return resolved;
151
+ }
152
+
153
+ function enforceDirectoryMode(fd: number): void {
154
+ try {
155
+ fchmodSync(fd, 0o700);
156
+ } catch (err) {
157
+ if (process.platform !== "win32") throw err;
158
+ }
159
+ }
160
+
161
+ function enforceFileMode(fd: number): void {
162
+ try {
163
+ fchmodSync(fd, 0o600);
164
+ } catch (err) {
165
+ if (process.platform !== "win32") throw err;
166
+ }
167
+ }
168
+
169
+ function safeFetchCacheDir(create: true): string;
170
+ function safeFetchCacheDir(create: false): string | null;
171
+ function safeFetchCacheDir(create: boolean): string | null {
124
172
  const dir = getFetchCacheDir();
125
- mkdirSync(dir, { recursive: true, mode: 0o700 });
173
+ if (create) mkdirSync(dir, { recursive: true, mode: 0o700 });
174
+ let before: Stats;
175
+ try {
176
+ before = lstatSync(dir);
177
+ } catch (err) {
178
+ if (!create && (err as NodeJS.ErrnoException).code === "ENOENT") return null;
179
+ throw err;
180
+ }
181
+ if (before.isSymbolicLink() || !before.isDirectory()) {
182
+ throw new Error("Fetched content cache path is not a safe directory");
183
+ }
184
+ if (process.platform === "win32") {
185
+ const after = lstatSync(dir);
186
+ if (after.isSymbolicLink() || !after.isDirectory() || after.dev !== before.dev || after.ino !== before.ino) {
187
+ throw new Error("Fetched content cache directory changed while securing it");
188
+ }
189
+ return dir;
190
+ }
191
+ let fd: number | null = null;
192
+ try {
193
+ fd = openSync(dir, constants.O_RDONLY | O_DIRECTORY | O_NOFOLLOW);
194
+ const opened = fstatSync(fd);
195
+ if (!opened.isDirectory() || opened.dev !== before.dev || opened.ino !== before.ino) {
196
+ throw new Error("Fetched content cache directory changed while opening");
197
+ }
198
+ enforceDirectoryMode(fd);
199
+ closeSync(fd);
200
+ fd = null;
201
+ const after = lstatSync(dir);
202
+ if (after.isSymbolicLink() || !after.isDirectory() || after.dev !== before.dev || after.ino !== before.ino) {
203
+ throw new Error("Fetched content cache directory changed while securing it");
204
+ }
205
+ return dir;
206
+ } finally {
207
+ if (fd !== null) try { closeSync(fd); } catch {}
208
+ }
209
+ }
210
+
211
+ function openRegularFile(path: string): { fd: number; info: Stats } {
212
+ const before = lstatSync(path);
213
+ if (before.isSymbolicLink() || !before.isFile()) throw new Error("Fetched content cache entry is not a regular file");
214
+ const fd = openSync(path, constants.O_RDONLY | O_NOFOLLOW);
215
+ try {
216
+ const info = fstatSync(fd);
217
+ if (!info.isFile() || info.dev !== before.dev || info.ino !== before.ino) {
218
+ throw new Error("Fetched content cache entry changed while opening");
219
+ }
220
+ return { fd, info };
221
+ } catch (err) {
222
+ closeSync(fd);
223
+ throw err;
224
+ }
225
+ }
226
+
227
+ type CacheUnlinkResult = "removed" | "missing" | "changed" | "error";
228
+
229
+ function unlinkCacheFile(dir: string, file: CacheFile): CacheUnlinkResult {
230
+ try {
231
+ const root = lstatSync(dir);
232
+ if (root.isSymbolicLink() || !root.isDirectory()) return "changed";
233
+ const path = join(dir, file.name);
234
+ let current: Stats;
235
+ try {
236
+ current = lstatSync(path);
237
+ } catch (err) {
238
+ return (err as NodeJS.ErrnoException).code === "ENOENT" ? "missing" : "error";
239
+ }
240
+ if (current.isSymbolicLink() || !current.isFile() || current.dev !== file.dev || current.ino !== file.ino) return "changed";
241
+ try {
242
+ unlinkSync(path);
243
+ return "removed";
244
+ } catch (err) {
245
+ return (err as NodeJS.ErrnoException).code === "ENOENT" ? "missing" : "error";
246
+ }
247
+ } catch {
248
+ return "error";
249
+ }
250
+ }
251
+
252
+ function pruneFetchCache(now: number, limits: FetchCacheLimits, preferredKey?: string, reservation?: { key: string; bytes: number }): boolean {
253
+ let dir: string | null;
254
+ try {
255
+ dir = safeFetchCacheDir(false);
256
+ } catch {
257
+ return false;
258
+ }
259
+ if (!dir) return true;
260
+
261
+ let entries: string[];
262
+ try {
263
+ entries = readdirSync(dir);
264
+ } catch {
265
+ return false;
266
+ }
267
+
268
+ const files: CacheFile[] = [];
269
+ for (const entry of entries) {
270
+ if (!CACHE_KEY_PATTERN.test(entry) && !CACHE_TMP_PATTERN.test(entry)) continue;
271
+ const path = join(dir, entry);
272
+ let opened: ReturnType<typeof openRegularFile>;
273
+ try {
274
+ opened = openRegularFile(path);
275
+ } catch {
276
+ continue;
277
+ }
278
+ try {
279
+ enforceFileMode(opened.fd);
280
+ } catch {
281
+ closeSync(opened.fd);
282
+ continue;
283
+ }
284
+ closeSync(opened.fd);
285
+ const file = { name: entry, size: opened.info.size, mtimeMs: opened.info.mtimeMs, dev: opened.info.dev, ino: opened.info.ino };
286
+ if (now - file.mtimeMs >= CACHE_TTL_MS) {
287
+ const removed = unlinkCacheFile(dir, file);
288
+ if (removed !== "removed" && removed !== "missing") return false;
289
+ continue;
290
+ }
291
+ if (CACHE_KEY_PATTERN.test(entry)) files.push(file);
292
+ }
293
+
294
+ files.sort((a, b) => a.mtimeMs - b.mtimeMs || a.name.localeCompare(b.name));
295
+ const projectedUsage = () => {
296
+ const replaced = reservation ? files.find((file) => file.name === reservation.key) : undefined;
297
+ return {
298
+ entries: files.length + (reservation && !replaced ? 1 : 0),
299
+ bytes: files.reduce((total, file) => total + file.size, 0) + (reservation ? reservation.bytes - (replaced?.size ?? 0) : 0),
300
+ };
301
+ };
302
+ const attempted = new Set<string>();
303
+ let usage = projectedUsage();
304
+ while (usage.entries > limits.maxEntries || usage.bytes > limits.maxBytes) {
305
+ const index = files.findIndex((file) => file.name !== preferredKey && !attempted.has(file.name));
306
+ if (index < 0) break;
307
+ const file = files[index];
308
+ attempted.add(file.name);
309
+ const removed = unlinkCacheFile(dir, file);
310
+ if (removed === "removed" || removed === "missing") files.splice(index, 1);
311
+ usage = projectedUsage();
312
+ }
313
+ return usage.entries <= limits.maxEntries && usage.bytes <= limits.maxBytes;
314
+ }
315
+
316
+ function writeFetchCache(data: StoredSearchData & { urls: ExtractedContent[] }): FetchCacheRef {
317
+ const limits = DEFAULT_CACHE_LIMITS;
318
+ const serialized = JSON.stringify(data);
319
+ const size = Buffer.byteLength(serialized);
320
+ if (size > limits.maxBytes) throw new Error(`Fetched content cache entry exceeds ${limits.maxBytes} bytes`);
321
+
322
+ const dir = safeFetchCacheDir(true);
126
323
  const key = cacheKeyForId(data.id);
324
+ if (!pruneFetchCache(Date.now(), limits, key, { key, bytes: size })) {
325
+ throw new Error("Fetched content cache could not reserve space for a new entry");
326
+ }
127
327
  const finalPath = join(dir, key);
128
- const tmpPath = join(dir, `${key}.${process.pid}.${Date.now()}.tmp`);
328
+ const tmpName = `${key}.${process.pid}.${Date.now()}.${randomBytes(16).toString("hex")}.tmp`;
329
+ const tmpPath = join(dir, tmpName);
330
+ let fd: number | null = null;
331
+ let tmpFile: CacheFile | null = null;
332
+ let renamed = false;
129
333
  try {
130
- writeFileSync(tmpPath, JSON.stringify(data), { mode: 0o600 });
334
+ fd = openSync(tmpPath, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY | O_NOFOLLOW, 0o600);
335
+ const tmpInfo = fstatSync(fd);
336
+ tmpFile = { name: tmpName, size: tmpInfo.size, mtimeMs: tmpInfo.mtimeMs, dev: tmpInfo.dev, ino: tmpInfo.ino };
337
+ enforceFileMode(fd);
338
+ writeFileSync(fd, serialized, "utf8");
339
+ fsyncSync(fd);
340
+ closeSync(fd);
341
+ fd = null;
342
+ safeFetchCacheDir(false);
131
343
  renameSync(tmpPath, finalPath);
344
+ renamed = true;
345
+ const written = lstatSync(finalPath);
346
+ if (!written.isFile() || written.dev !== tmpFile.dev || written.ino !== tmpFile.ino) {
347
+ throw new Error("Fetched content cache entry changed after writing");
348
+ }
349
+ if (!pruneFetchCache(Date.now(), limits, key)) {
350
+ throw new Error("Fetched content cache could not meet its limits after writing");
351
+ }
132
352
  } catch (err) {
133
- try { unlinkSync(tmpPath); } catch {}
353
+ if (fd !== null) try { closeSync(fd); } catch {}
354
+ if (tmpFile) unlinkCacheFile(dir, { ...tmpFile, name: renamed ? key : tmpName });
134
355
  throw err;
135
356
  }
136
357
  return { version: FETCH_CACHE_VERSION, key, storedAt: Date.now() };
@@ -182,37 +403,32 @@ function readCachedFetchData(data: StoredSearchData, now = Date.now()): StoredSe
182
403
  return unavailableFetchData(data, data.fetchCacheError ?? "Cached fetched content is unavailable");
183
404
  }
184
405
  const path = fetchCachePath(data.fetchCache.key);
185
- if (!path || !existsSync(path)) {
186
- return unavailableFetchData(data, "Cached fetched content is missing or expired");
187
- }
406
+ if (!path) return unavailableFetchData(data, "Cached fetched content is missing or expired");
407
+ let fd: number | null = null;
188
408
  try {
189
- const parsed: unknown = JSON.parse(readFileSync(path, "utf-8"));
409
+ if (!safeFetchCacheDir(false)) return unavailableFetchData(data, "Cached fetched content is missing or expired");
410
+ const opened = openRegularFile(path);
411
+ fd = opened.fd;
412
+ enforceFileMode(fd);
413
+ const parsed: unknown = JSON.parse(readFileSync(fd, "utf8"));
190
414
  if (!isValidStoredData(parsed) || parsed.type !== "fetch" || parsed.id !== data.id || !isInlineFetchData(parsed)) {
191
415
  return unavailableFetchData(data, "Cached fetched content is invalid");
192
416
  }
193
417
  return { ...parsed, fetchCache: data.fetchCache, urlMetadata: data.urlMetadata };
194
418
  } catch (err) {
419
+ if ((err as NodeJS.ErrnoException).code === "ENOENT") {
420
+ return unavailableFetchData(data, "Cached fetched content is missing or expired");
421
+ }
195
422
  const message = err instanceof Error ? err.message : String(err);
196
423
  return unavailableFetchData(data, `Cached fetched content could not be read: ${message}`);
424
+ } finally {
425
+ if (fd !== null) try { closeSync(fd); } catch {}
197
426
  }
198
427
  }
199
428
 
200
- export function pruneExpiredFetchCache(now = Date.now()): void {
201
- const dir = getFetchCacheDir();
202
- if (!existsSync(dir)) return;
203
- let entries: string[];
204
- try {
205
- entries = readdirSync(dir);
206
- } catch {
207
- return;
208
- }
209
- for (const entry of entries) {
210
- if (!CACHE_KEY_PATTERN.test(entry)) continue;
211
- const path = join(dir, entry);
212
- try {
213
- if (now - statSync(path).mtimeMs >= CACHE_TTL_MS) unlinkSync(path);
214
- } catch {}
215
- }
429
+ export function pruneExpiredFetchCache(now = Date.now(), requestedLimits?: Partial<FetchCacheLimits>): void {
430
+ const limits = cacheLimits(requestedLimits);
431
+ try { pruneFetchCache(now, limits); } catch {}
216
432
  }
217
433
 
218
434
  export function storeResult(id: string, data: StoredSearchData): void {
@@ -220,7 +436,6 @@ export function storeResult(id: string, data: StoredSearchData): void {
220
436
  }
221
437
 
222
438
  export function storeFetchedContentResult(id: string, data: StoredSearchData & { type: "fetch"; urls: ExtractedContent[] }): StoredSearchData {
223
- pruneExpiredFetchCache();
224
439
  let ref: FetchCacheRef | null = null;
225
440
  let cacheError: string | undefined;
226
441
  try {
@@ -247,8 +462,16 @@ export function getAllResults(): StoredSearchData[] {
247
462
  export function deleteResult(id: string): boolean {
248
463
  const data = storedResults.get(id);
249
464
  if (data?.fetchCache) {
250
- const path = fetchCachePath(data.fetchCache.key);
251
- if (path) try { unlinkSync(path); } catch {}
465
+ try {
466
+ const dir = safeFetchCacheDir(false);
467
+ const path = fetchCachePath(data.fetchCache.key);
468
+ if (dir && path) {
469
+ const info = lstatSync(path);
470
+ if (!info.isSymbolicLink() && info.isFile()) {
471
+ unlinkCacheFile(dir, { name: data.fetchCache.key, size: info.size, mtimeMs: info.mtimeMs, dev: info.dev, ino: info.ino });
472
+ }
473
+ }
474
+ } catch {}
252
475
  }
253
476
  return storedResults.delete(id);
254
477
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "bestony-pi-preset",
3
- "version": "0.0.21",
3
+ "version": "0.0.22",
4
4
  "description": "Bestony's personal Pi coding agent preset — skills, extensions, prompts, and themes.",
5
5
  "license": "MIT",
6
6
  "author": "bestony",
@@ -40,9 +40,9 @@
40
40
  "@tintinweb/pi-subagents": "^0.15.0",
41
41
  "pi-cache-optimizer": "^2.8.2",
42
42
  "pi-init": "^1.0.0",
43
- "pi-mcp-adapter": "^2.22.0",
43
+ "pi-mcp-adapter": "^2.23.0",
44
44
  "pi-session-name": "^0.1.2",
45
- "pi-web-access": "^0.21.0",
45
+ "pi-web-access": "^0.22.0",
46
46
  "pi-xai": "^0.17.1"
47
47
  },
48
48
  "bundledDependencies": [