@herbertgao/pi-extensions 2026.8.13 → 2026.8.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -6
- package/node_modules/@czottmann/pi-automode/CHANGELOG.md +12 -0
- package/node_modules/@czottmann/pi-automode/README.md +1 -1
- package/node_modules/@czottmann/pi-automode/extensions/auto-mode/classifier.ts +55 -4
- package/node_modules/@czottmann/pi-automode/extensions/auto-mode/config.ts +7 -3
- package/node_modules/@czottmann/pi-automode/extensions/auto-mode/constants.ts +2 -0
- package/node_modules/@czottmann/pi-automode/extensions/auto-mode/hard-deny.ts +134 -23
- package/node_modules/@czottmann/pi-automode/package.json +1 -1
- package/node_modules/@herbertgao/pi-cc-extensions/README.en.md +1 -1
- package/node_modules/@herbertgao/pi-cc-extensions/README.md +1 -1
- package/node_modules/@herbertgao/pi-cc-extensions/package.json +3 -3
- package/node_modules/@herbertgao/pi-subagents/CHANGELOG.md +6 -0
- package/node_modules/@herbertgao/pi-subagents/package.json +2 -1
- package/node_modules/@herbertgao/pi-subagents/src/ui/conversation-viewer.ts +8 -1
- package/node_modules/@tifan/pi-preferred-thinking/README.md +1 -1
- package/node_modules/@tifan/pi-preferred-thinking/package.json +1 -1
- package/node_modules/@tifan/pi-preferred-thinking/src/index.ts +15 -2
- package/node_modules/pi-mcp-adapter/CHANGELOG.md +20 -0
- package/node_modules/pi-mcp-adapter/README.md +3 -0
- package/node_modules/pi-mcp-adapter/cli.js +4 -4
- package/node_modules/pi-mcp-adapter/config.ts +1 -0
- package/node_modules/pi-mcp-adapter/dist/config.js +1 -0
- package/node_modules/pi-mcp-adapter/dist/config.js.map +1 -1
- package/node_modules/pi-mcp-adapter/dist/mcp-bearer-store.d.ts +24 -0
- package/node_modules/pi-mcp-adapter/dist/mcp-bearer-store.js +336 -0
- package/node_modules/pi-mcp-adapter/dist/mcp-bearer-store.js.map +1 -0
- package/node_modules/pi-mcp-adapter/dist/types.d.ts +2 -0
- package/node_modules/pi-mcp-adapter/dist/types.js.map +1 -1
- package/node_modules/pi-mcp-adapter/index.ts +90 -6
- package/node_modules/pi-mcp-adapter/mcp-auth-flow.ts +19 -0
- package/node_modules/pi-mcp-adapter/mcp-bearer-store.ts +0 -2
- package/node_modules/pi-mcp-adapter/mcp-oauth-provider.ts +89 -0
- package/node_modules/pi-mcp-adapter/mcp-references.ts +9 -1
- package/node_modules/pi-mcp-adapter/package.json +1 -1
- package/node_modules/pi-mcp-adapter/proxy-modes.ts +40 -10
- package/node_modules/pi-mcp-adapter/request-headers-command.ts +1 -1
- package/node_modules/pi-mcp-adapter/types.ts +2 -0
- package/node_modules/pi-web-access/CHANGELOG.md +22 -0
- package/node_modules/pi-web-access/README.md +11 -8
- package/node_modules/pi-web-access/curator-page.ts +12 -3
- package/node_modules/pi-web-access/curator-server.ts +3 -1
- package/node_modules/pi-web-access/extract.ts +31 -1
- package/node_modules/pi-web-access/gemini-search.ts +10 -5
- package/node_modules/pi-web-access/index.ts +39 -39
- package/node_modules/pi-web-access/package.json +1 -1
- package/node_modules/pi-web-access/xcrawl.ts +264 -0
- package/package.json +7 -10
- package/node_modules/@herbertgao/pi-stash/LICENSE +0 -22
- package/node_modules/@herbertgao/pi-stash/README.md +0 -34
- package/node_modules/@herbertgao/pi-stash/package.json +0 -51
- package/node_modules/@herbertgao/pi-stash/src/index.ts +0 -118
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Pi Web Access
|
|
6
6
|
|
|
7
|
-
**Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, Valyu, xAI/Grok, Bright Data SERP, SerpBase, Serper, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, Kimi Code Plan search, or bring your own API keys.**
|
|
7
|
+
**Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, XCrawl, Valyu, xAI/Grok, Bright Data SERP, SerpBase, Serper, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, Kimi Code Plan search, or bring your own API keys.**
|
|
8
8
|
|
|
9
9
|
[](https://www.npmjs.com/package/pi-web-access)
|
|
10
10
|
[](https://opensource.org/licenses/MIT)
|
|
@@ -110,7 +110,7 @@ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on sc
|
|
|
110
110
|
|
|
111
111
|
### web_search
|
|
112
112
|
|
|
113
|
-
Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, Valyu, xAI, Bright Data SERP, SerpBase, Serper, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, Gemini, or Kimi. Returns a synthesized answer with source citations.
|
|
113
|
+
Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, XCrawl, Valyu, xAI, Bright Data SERP, SerpBase, Serper, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, Gemini, or Kimi. Returns a synthesized answer with source citations.
|
|
114
114
|
|
|
115
115
|
```typescript
|
|
116
116
|
web_search({ query: "rust async programming" })
|
|
@@ -132,7 +132,7 @@ web_search({ queries: ["query 1", "query 2"], workflow: "auto-summary" })
|
|
|
132
132
|
| `numResults` | Results per query (default: 5, max: 20) |
|
|
133
133
|
| `recencyFilter` | `day`, `week`, `month`, or `year` |
|
|
134
134
|
| `domainFilter` | Limit to domains (prefix with `-` to exclude) |
|
|
135
|
-
| `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper simultaneously; otherwise `openai`, `brave`, `parallel`, `parallel-mcp`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `firecrawl`, `jina`, `serpdive`, `kagi`, `bocha`, `ollama`, `anysearch`, `valyu`, `xai`, `brightdata`, `serpbase`, `serper`, `searxng`, `duckduckgo`, `exa`, `perplexity`, `gemini`, or `kimi` (auto-selects when no provider or routing is configured; Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper are explicit-only) |
|
|
135
|
+
| `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper simultaneously; otherwise `openai`, `brave`, `parallel`, `parallel-mcp`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `firecrawl`, `jina`, `serpdive`, `kagi`, `bocha`, `ollama`, `anysearch`, `xcrawl`, `valyu`, `xai`, `brightdata`, `serpbase`, `serper`, `searxng`, `duckduckgo`, `exa`, `perplexity`, `gemini`, or `kimi` (auto-selects when no provider or routing is configured; Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper are explicit-only) |
|
|
136
136
|
| `includeContent` | Fetch full page content from sources in background |
|
|
137
137
|
| `workflow` | `none` (skip curator), `summary-review` (open curator and auto-generate a summary draft, default), or `auto-summary` (generate a summary without opening the curator) |
|
|
138
138
|
|
|
@@ -479,7 +479,7 @@ Config defaults to `~/.pi/web-search.json`, or `web-search.json` under `PI_CODIN
|
|
|
479
479
|
|
|
480
480
|
`summaryModel` accepts an optional thinking-level suffix, such as `anthropic/claude-haiku-4-5:low`. Supported suffixes are `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`.
|
|
481
481
|
|
|
482
|
-
All provider API-key fields (`openaiApiKey`, `braveApiKey`, `parallelApiKey`, `tinyfishApiKey`, `search1apiApiKey`, `searchinfinityApiKey`, `queritApiKey`, `tavilyApiKey`, `jinaApiKey`, `serpdiveApiKey`, `kagiApiKey`, `bochaApiKey`, `ollamaApiKey`, `serpbaseApiKey`, `anysearchApiKey`, `xaiApiKey`, `brightdataApiKey`, `firecrawlApiKey`, `exaApiKey`, `perplexityApiKey`, `geminiApiKey`, `datalabApiKey`, and `cloudflareApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
|
|
482
|
+
All provider API-key fields (`openaiApiKey`, `braveApiKey`, `parallelApiKey`, `tinyfishApiKey`, `search1apiApiKey`, `searchinfinityApiKey`, `queritApiKey`, `tavilyApiKey`, `jinaApiKey`, `serpdiveApiKey`, `kagiApiKey`, `bochaApiKey`, `ollamaApiKey`, `serpbaseApiKey`, `anysearchApiKey`, `xcrawlApiKey`, `xaiApiKey`, `brightdataApiKey`, `firecrawlApiKey`, `exaApiKey`, `perplexityApiKey`, `geminiApiKey`, `datalabApiKey`, and `cloudflareApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
|
|
483
483
|
|
|
484
484
|
```json
|
|
485
485
|
{
|
|
@@ -526,9 +526,7 @@ Bright Data Web Unlocker is a paid `fetch_content` fallback after Parallel and b
|
|
|
526
526
|
|
|
527
527
|
**Parallel MCP.** Select `provider: "parallel-mcp"` to use Parallel Search MCP without an API key, or add it to `searchRouting`. It is explicit-only and is never chosen by `auto` or `provider: "all"`; the existing `parallel` provider remains the key-required REST API. A configured `parallelApiKey` or `PARALLEL_API_KEY` is sent as an optional Bearer token for higher MCP limits. To use MCP `web_fetch`, add `parallel-mcp` to `fetchRouting.providers` and set `fetchRouting.allowRemoteHostedProviders` to `true`; it is not part of the default fetch route.
|
|
528
528
|
|
|
529
|
-
Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `BOCHA_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. `openaiSearchProviders` sets which Pi model providers OpenAI `web_search` resolves login credentials from, in priority order; it defaults to `["openai-codex", "openai"]`, entries that are not registered or not signed in are skipped, and an empty array skips Pi credentials entirely so the `openaiApiKey` / `OPENAI_API_KEY` fallback applies. Useful for choosing between multiple Codex accounts (for example a second account registered by an extension) or forcing API-key billing while signed into Codex. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Alternatively, set `geminiAuth` to `"adc"` to authenticate Gemini generate-content calls with Google Application Default Credentials (ADC) instead of an API key; calls go to the Vertex AI endpoint (`aiplatform.googleapis.com`) with an OAuth bearer token minted from the ADC file (`GOOGLE_APPLICATION_CREDENTIALS` or `~/.config/gcloud/application_default_credentials.json`, i.e. `gcloud auth application-default login`). `geminiProject`/`geminiLocation` set the Vertex project and location and fall back to the `GOOGLE_CLOUD_PROJECT`/`GOOGLE_CLOUD_LOCATION` (or `GCLOUD_PROJECT`) env vars; project and location are required. ADC supports `authorized_user` (OAuth refresh token) and `service_account` (JWT assertion) credential files, and tokens are cached and refreshed from expiry. ADC mode covers search, URL context, and PDF/inline-data extraction; YouTube and local video analysis still go through the Gemini Files API, so they fall back to Gemini Web unless a `GEMINI_API_KEY` is also configured. The access token is treated as a credential and is redacted from errors. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"parallel-mcp"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"firecrawl"`, `"jina"`, `"serpdive"`, `"kagi"`, `"bocha"`, `"ollama"`, `"anysearch"`, `"valyu"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"serper"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. Parallel MCP, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `browserCookies.profile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. macOS discovery supports Helium, Chrome, Brave, and Arc; Linux discovery supports Chromium and Chrome. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `maxInlineContentChars` sets the direct `fetch_content` content slice and the default and maximum `get_search_content` slice. It defaults to `30000`, must be a positive integer, and is capped at `200000`; full fetched content remains stored for later retrieval. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
|
|
530
|
-
`VALYU_API_KEY` and `SERPER_API_KEY` also retain this precedence. `provider` and `searchProvider` also accept `"parallel-mcp"`, `"kimi"`, `"valyu"`, and `"serper"`; all four remain explicit-only.
|
|
531
|
-
|
|
529
|
+
Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `BOCHA_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `SERPER_API_KEY`, `ANYSEARCH_API_KEY`, `XCRAWL_API_KEY`, `VALYU_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. `openaiSearchProviders` sets which Pi model providers OpenAI `web_search` resolves login credentials from, in priority order; it defaults to `["openai-codex", "openai"]`, entries that are not registered or not signed in are skipped, and an empty array skips Pi credentials entirely so the `openaiApiKey` / `OPENAI_API_KEY` fallback applies. Useful for choosing between multiple Codex accounts (for example a second account registered by an extension) or forcing API-key billing while signed into Codex. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Alternatively, set `geminiAuth` to `"adc"` to authenticate Gemini generate-content calls with Google Application Default Credentials (ADC) instead of an API key; calls go to the Vertex AI endpoint (`aiplatform.googleapis.com`) with an OAuth bearer token minted from the ADC file (`GOOGLE_APPLICATION_CREDENTIALS` or `~/.config/gcloud/application_default_credentials.json`, i.e. `gcloud auth application-default login`). `geminiProject`/`geminiLocation` set the Vertex project and location and fall back to the `GOOGLE_CLOUD_PROJECT`/`GOOGLE_CLOUD_LOCATION` (or `GCLOUD_PROJECT`) env vars; project and location are required. ADC supports `authorized_user` (OAuth refresh token) and `service_account` (JWT assertion) credential files, and tokens are cached and refreshed from expiry. ADC mode covers search, URL context, and PDF/inline-data extraction; YouTube and local video analysis still go through the Gemini Files API, so they fall back to Gemini Web unless a `GEMINI_API_KEY` is also configured. The access token is treated as a credential and is redacted from errors. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"parallel-mcp"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"firecrawl"`, `"jina"`, `"serpdive"`, `"kagi"`, `"bocha"`, `"ollama"`, `"anysearch"`, `"xcrawl"`, `"valyu"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"serper"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. Parallel MCP, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `browserCookies.profile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. macOS discovery supports Helium, Chrome, Brave, and Arc; Linux discovery supports Chromium and Chrome. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `maxInlineContentChars` sets the direct `fetch_content` content slice and the default and maximum `get_search_content` slice. It defaults to `30000`, must be a positive integer, and is capped at `200000`; full fetched content remains stored for later retrieval. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
|
|
532
530
|
### Kimi Code Plan
|
|
533
531
|
|
|
534
532
|
Run `/login kimi-coding` in Pi and complete sign-in for an active Kimi Code Plan. Then select `provider: "kimi"`, include `"kimi"` in an explicit provider array, or add it to `searchRouting.providers`. The extension resolves a model with provider `kimi-coding` from Pi's model registry and reuses Pi's refreshed OAuth credential; no Moonshot Open Platform key is configured here.
|
|
@@ -539,7 +537,7 @@ Kimi is explicit-only: it is never chosen by `auto` and never participates in `p
|
|
|
539
537
|
|
|
540
538
|
### All providers
|
|
541
539
|
|
|
542
|
-
Set `provider: "all"` on `web_search` or `source_check`, or configure `"provider": "all"` as the default, to run the same query against every eligible search provider simultaneously. Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper are always excluded because they are explicit-only; Bright Data, SerpBase, and Serper are paid Google SERP providers, while Kimi draws from the user's shared Code Plan quota, so `all` never spends either resource without an explicit request. Exa remains eligible through its zero-config MCP path, OpenAI can use Pi auth, and other API-backed search providers participate when their API key, local endpoint, or gateway makes them available. Browser-cookie access alone does not opt Gemini into `all`; select Gemini explicitly or configure its API/gateway.
|
|
540
|
+
Set `provider: "all"` on `web_search` or `source_check`, or configure `"provider": "all"` as the default, to run the same query against every eligible search provider simultaneously. Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper are always excluded because they are explicit-only; Bright Data, SerpBase, and Serper are paid Google SERP providers, while Kimi draws from the user's shared Code Plan quota, so `all` never spends either resource without an explicit request. Exa remains eligible through its zero-config MCP path, OpenAI can use Pi auth, and other API-backed search providers participate when their API key, local endpoint, or gateway makes them available. Browser-cookie access alone does not opt Gemini into `all`; select Gemini explicitly or configure its API/gateway.
|
|
543
541
|
|
|
544
542
|
Successful provider answers are preserved separately while source URLs and inline content are deduplicated, and one provider failure does not discard the other results. If every participating provider fails, the tool returns per-provider diagnostics. Configured Firecrawl participates in `all` like other eligible providers. In the Curator, **All** can also be selected like the other provider buttons. Each participating provider gets its own result card, including a provider badge and independent selection checkbox; failed providers get their own disabled error card. The final summary is generated from the selected provider cards and is what Pi receives. Outside the Curator, the same provider answers remain available as labeled sections in one tool response.
|
|
545
543
|
|
|
@@ -626,6 +624,10 @@ Search requests follow the official [`querit-python`](https://github.com/querit-
|
|
|
626
624
|
|
|
627
625
|
AnySearch is an explicit-only provider: it is never included in zero-config `auto` fallback or in `provider: "all"`, but it can be selected with `provider: "anysearch"`, configured as the named provider, or placed in `searchRouting`. It supports anonymous requests and optional `anysearchApiKey` / `ANYSEARCH_API_KEY` credentials. Requests intentionally send only `{ query, max_results }`; `recencyFilter`, `domainFilter`, and `includeContent` do not add API request parameters. When `includeContent` is true, returned `content` fields are exposed as inline content.
|
|
628
626
|
|
|
627
|
+
### XCrawl
|
|
628
|
+
|
|
629
|
+
XCrawl is an explicit-only provider: it is never included in zero-config `auto` fallback or in `provider: "all"`, but it can be selected with `provider: "xcrawl"`, configured as the named provider, or placed in `searchRouting`. It requires `xcrawlApiKey` / `XCRAWL_API_KEY` credentials ([dashboard](https://dash.xcrawl.com/)). Requests send `{ engine: "google_search", q }` to the SERP endpoint; `recencyFilter`, `domainFilter`, and `includeContent` do not add API request parameters, and the shared include/exclude `domainFilter` is applied client-side. A provider-side timeout surfaces as a retriable failure rather than caller cancellation. Results with a null title fall back to the result URL; result items expose their SERP snippet as usual.
|
|
630
|
+
|
|
629
631
|
### xAI (Grok)
|
|
630
632
|
|
|
631
633
|
xAI is an explicit-only provider: it is never included in zero-config `auto` fallback or in `provider: "all"`, but it can be selected with `provider: "xai"`, configured as the named provider, or placed in `searchRouting`.
|
|
@@ -905,6 +907,7 @@ Rate limits: Perplexity is capped at 10 requests/minute (client-side). Jina Sear
|
|
|
905
907
|
| `serpbase.ts` | Explicit-only SerpBase Google SERP provider |
|
|
906
908
|
| `serper.ts` | Explicit-only Serper Google SERP provider |
|
|
907
909
|
| `anysearch.ts` | Explicit-only AnySearch search provider |
|
|
910
|
+
| `xcrawl.ts` | Explicit-only XCrawl search provider |
|
|
908
911
|
| `valyu.ts` | Explicit-only Valyu research search provider |
|
|
909
912
|
| `xai-search.ts` | Explicit-only xAI (Grok) hosted web_search provider |
|
|
910
913
|
| `kimi-search.ts` | Explicit-only Kimi Code Plan search provider |
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import type { ProviderAvailability } from "./gemini-search.ts";
|
|
2
|
+
|
|
1
3
|
function safeInlineJSON(data: unknown): string {
|
|
2
4
|
return JSON.stringify(data)
|
|
3
5
|
.replace(/</g, "\\u003c")
|
|
@@ -8,7 +10,7 @@ function safeInlineJSON(data: unknown): string {
|
|
|
8
10
|
}
|
|
9
11
|
|
|
10
12
|
function buildProviderButtons(
|
|
11
|
-
available:
|
|
13
|
+
available: ProviderAvailability,
|
|
12
14
|
selected: string,
|
|
13
15
|
hasInitialQueries: boolean,
|
|
14
16
|
): string {
|
|
@@ -36,6 +38,7 @@ function buildProviderButtons(
|
|
|
36
38
|
{ value: "gemini", label: "Gemini", available: available.gemini },
|
|
37
39
|
{ value: "kimi", label: "Kimi", available: available.kimi },
|
|
38
40
|
{ value: "anysearch", label: "AnySearch", available: available.anysearch },
|
|
41
|
+
{ value: "xcrawl", label: "XCrawl", available: available.xcrawl },
|
|
39
42
|
{ value: "xai", label: "xAI", available: available.xai },
|
|
40
43
|
{ value: "brightdata", label: "Bright Data", available: available.brightdata },
|
|
41
44
|
{ value: "serpbase", label: "SerpBase", available: available.serpbase },
|
|
@@ -59,7 +62,7 @@ export function generateCuratorPage(
|
|
|
59
62
|
queries: string[],
|
|
60
63
|
sessionToken: string,
|
|
61
64
|
timeout: number,
|
|
62
|
-
availableProviders:
|
|
65
|
+
availableProviders: ProviderAvailability,
|
|
63
66
|
defaultProvider: string,
|
|
64
67
|
searchProvider: string,
|
|
65
68
|
summaryModels: Array<{ value: string; label: string }>,
|
|
@@ -677,6 +680,11 @@ main {
|
|
|
677
680
|
background: rgba(249, 199, 79, 0.14);
|
|
678
681
|
border-color: rgba(249, 199, 79, 0.3);
|
|
679
682
|
}
|
|
683
|
+
.provider-tag.provider-xcrawl {
|
|
684
|
+
color: #7dd3ae;
|
|
685
|
+
background: rgba(125, 211, 174, 0.14);
|
|
686
|
+
border-color: rgba(125, 211, 174, 0.3);
|
|
687
|
+
}
|
|
680
688
|
.provider-tag.provider-xai {
|
|
681
689
|
color: #c4b5fd;
|
|
682
690
|
background: rgba(196, 181, 253, 0.14);
|
|
@@ -1460,7 +1468,7 @@ const SCRIPT = `(function() {
|
|
|
1460
1468
|
var token = DATA.sessionToken;
|
|
1461
1469
|
var timeoutSec = DATA.timeout;
|
|
1462
1470
|
var queries = Array.isArray(DATA.queries) ? DATA.queries : [];
|
|
1463
|
-
var providers = ["all", "openai", "exa", "brave", "parallel", "parallel-mcp", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "serpdive", "kagi", "bocha", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "kimi", "anysearch", "xai", "brightdata", "serpbase", "serper", "valyu"];
|
|
1471
|
+
var providers = ["all", "openai", "exa", "brave", "parallel", "parallel-mcp", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "serpdive", "kagi", "bocha", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "kimi", "anysearch", "xcrawl", "xai", "brightdata", "serpbase", "serper", "valyu"];
|
|
1464
1472
|
var availProviders = DATA.availableProviders && typeof DATA.availableProviders === "object" ? DATA.availableProviders : {};
|
|
1465
1473
|
var workflow = "summary-review";
|
|
1466
1474
|
var initialDefaultProvider = typeof DATA.defaultProvider === "string" ? DATA.defaultProvider : "exa";
|
|
@@ -1685,6 +1693,7 @@ const SCRIPT = `(function() {
|
|
|
1685
1693
|
if (provider === "gemini") return "Gemini";
|
|
1686
1694
|
if (provider === "kimi") return "Kimi";
|
|
1687
1695
|
if (provider === "anysearch") return "AnySearch";
|
|
1696
|
+
if (provider === "xcrawl") return "XCrawl";
|
|
1688
1697
|
if (provider === "xai") return "xAI";
|
|
1689
1698
|
if (provider === "brightdata") return "Bright Data";
|
|
1690
1699
|
if (provider === "serpbase") return "SerpBase";
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import http, { type IncomingMessage, type ServerResponse } from "node:http";
|
|
2
2
|
import { generateCuratorPage } from "./curator-page.ts";
|
|
3
|
+
import type { ProviderAvailability } from "./gemini-search.ts";
|
|
3
4
|
import type { SummaryMeta } from "./summary-review.ts";
|
|
4
5
|
import { resolveCuratorNetworkConfig } from "./utils.ts";
|
|
5
6
|
|
|
@@ -18,7 +19,7 @@ export interface CuratorServerOptions {
|
|
|
18
19
|
queries: string[];
|
|
19
20
|
sessionToken: string;
|
|
20
21
|
timeout: number;
|
|
21
|
-
availableProviders:
|
|
22
|
+
availableProviders: ProviderAvailability;
|
|
22
23
|
defaultProvider: string;
|
|
23
24
|
searchProvider: string;
|
|
24
25
|
summaryModels: Array<{ value: string; label: string }>;
|
|
@@ -294,6 +295,7 @@ export function startCuratorServer(
|
|
|
294
295
|
if (provider === "gemini") return availableProviders.gemini;
|
|
295
296
|
if (provider === "kimi") return availableProviders.kimi;
|
|
296
297
|
if (provider === "anysearch") return availableProviders.anysearch;
|
|
298
|
+
if (provider === "xcrawl") return availableProviders.xcrawl;
|
|
297
299
|
if (provider === "xai") return availableProviders.xai;
|
|
298
300
|
if (provider === "brightdata") return availableProviders.brightdata;
|
|
299
301
|
if (provider === "serpbase") return availableProviders.serpbase;
|
|
@@ -43,10 +43,40 @@ type FetchRouting = { providers: FetchProvider[]; allowRemoteHostedProviders: bo
|
|
|
43
43
|
const DEFAULT_FETCH_PROVIDER_ORDER: FetchProvider[] = ["http", "firecrawl", "jina", "tinyfish", "search1api", "querit", "kagi", "ollama", "parallel", "brightdata", "gemini"];
|
|
44
44
|
const REMOTE_HOSTED_FETCH_PROVIDERS = new Set<FetchProvider>(["jina", "tinyfish", "search1api", "querit", "kagi", "ollama", "parallel", "parallel-mcp", "brightdata", "gemini"]);
|
|
45
45
|
|
|
46
|
+
function isDefuddleConsoleError(args: Parameters<typeof console.error>): boolean {
|
|
47
|
+
const prefix = args[0];
|
|
48
|
+
return prefix === "Defuddle" || (typeof prefix === "string" && /^Defuddle(?:\s|:)/.test(prefix));
|
|
49
|
+
}
|
|
50
|
+
|
|
46
51
|
async function extractWithDefuddle(text: string, url: string): Promise<{ title: string; content: string } | null> {
|
|
47
52
|
const { Defuddle } = await import("defuddle/node");
|
|
48
53
|
const { document } = parseHTML(text);
|
|
49
|
-
|
|
54
|
+
let processingError: unknown;
|
|
55
|
+
const originalConsoleError = console.error;
|
|
56
|
+
console.error = (...args) => {
|
|
57
|
+
if (isDefuddleConsoleError(args)) {
|
|
58
|
+
if (args[0] === "Defuddle" && args[1] === "Error processing document:") {
|
|
59
|
+
processingError = args[2];
|
|
60
|
+
}
|
|
61
|
+
return;
|
|
62
|
+
}
|
|
63
|
+
originalConsoleError(...args);
|
|
64
|
+
};
|
|
65
|
+
|
|
66
|
+
let resultPromise: ReturnType<typeof Defuddle>;
|
|
67
|
+
try {
|
|
68
|
+
// With useAsync:false, Defuddle parses synchronously before returning its promise.
|
|
69
|
+
// Keep the console interception limited to that call so unrelated Pi output is
|
|
70
|
+
// never routed through this fallback's handler.
|
|
71
|
+
resultPromise = Defuddle(document as unknown as Document, url, { markdown: true, useAsync: false });
|
|
72
|
+
} finally {
|
|
73
|
+
console.error = originalConsoleError;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const result = await resultPromise;
|
|
77
|
+
if (processingError !== undefined) {
|
|
78
|
+
throw new Error(`Defuddle failed to process document: ${errorMessage(processingError)}`);
|
|
79
|
+
}
|
|
50
80
|
return typeof result.content === "string" ? { title: result.title, content: result.content } : null;
|
|
51
81
|
}
|
|
52
82
|
|
|
@@ -30,6 +30,7 @@ import { isOllamaAvailable, searchWithOllama } from "./ollama.ts";
|
|
|
30
30
|
import { isSearXNGAvailable, searchWithSearXNG } from "./searxng.ts";
|
|
31
31
|
import { isDuckDuckGoAvailable, searchWithDuckDuckGo } from "./duckduckgo.ts";
|
|
32
32
|
import { isAnySearchAvailable, searchWithAnySearch } from "./anysearch.ts";
|
|
33
|
+
import { isXcrawlAvailable, searchWithXCrawl } from "./xcrawl.ts";
|
|
33
34
|
import { isXaiSearchAvailable, searchWithXai } from "./xai-search.ts";
|
|
34
35
|
import { isBrightDataAvailable, searchWithBrightData } from "./brightdata.ts";
|
|
35
36
|
import { isSerpBaseAvailable, searchWithSerpBase } from "./serpbase.ts";
|
|
@@ -38,12 +39,13 @@ import { isValyuAvailable, searchWithValyu } from "./valyu.ts";
|
|
|
38
39
|
import { isKimiSearchAvailable, searchWithKimi } from "./kimi-search.ts";
|
|
39
40
|
import { getWebSearchConfigPath } from "./utils.ts";
|
|
40
41
|
|
|
41
|
-
export const RESOLVED_SEARCH_PROVIDERS = ["openai", "brave", "parallel", "parallel-mcp", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "searxng", "duckduckgo", "perplexity", "gemini", "kimi", "exa", "serpdive", "kagi", "ollama", "anysearch", "xai", "brightdata", "serpbase", "serper", "valyu", "bocha"] as const;
|
|
42
|
+
export const RESOLVED_SEARCH_PROVIDERS = ["openai", "brave", "parallel", "parallel-mcp", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "searxng", "duckduckgo", "perplexity", "gemini", "kimi", "exa", "serpdive", "kagi", "ollama", "anysearch", "xai", "brightdata", "serpbase", "serper", "valyu", "bocha", "xcrawl"] as const;
|
|
42
43
|
export const SEARCH_PROVIDERS = ["auto", "all", ...RESOLVED_SEARCH_PROVIDERS] as const;
|
|
43
44
|
|
|
44
45
|
export type ResolvedSearchProvider = typeof RESOLVED_SEARCH_PROVIDERS[number];
|
|
45
46
|
export type SearchProvider = typeof SEARCH_PROVIDERS[number];
|
|
46
47
|
export type SearchProviderSelection = SearchProvider | ResolvedSearchProvider[];
|
|
48
|
+
export type ProviderAvailability = { all: boolean } & Record<ResolvedSearchProvider, boolean>;
|
|
47
49
|
export type SearchProviderErrorKind =
|
|
48
50
|
| "transient"
|
|
49
51
|
| "quota"
|
|
@@ -102,9 +104,9 @@ export interface AttributedSearchResponse extends SearchResponse {
|
|
|
102
104
|
|
|
103
105
|
const CONFIG_PATH = getWebSearchConfigPath();
|
|
104
106
|
const DEFAULT_SEARCH_MODEL = "gemini-3.6-flash";
|
|
105
|
-
// Explicit-only providers (Parallel MCP, DuckDuckGo, Kimi, AnySearch, xAI, Bright Data, SerpBase, Serper, Valyu) are deliberately absent:
|
|
107
|
+
// Explicit-only providers (Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, xAI, Bright Data, SerpBase, Serper, Valyu) are deliberately absent:
|
|
106
108
|
// `all` must never fan out to an opt-in or paid provider without the user asking for it.
|
|
107
|
-
const ALL_SEARCH_PROVIDERS: ResolvedSearchProvider[] = ["searxng", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "serpdive", "kagi", "ollama", "perplexity", "gemini", "bocha"];
|
|
109
|
+
export const ALL_SEARCH_PROVIDERS: ResolvedSearchProvider[] = ["searxng", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "serpdive", "kagi", "ollama", "perplexity", "gemini", "bocha"];
|
|
108
110
|
const VALID_ROUTING_KINDS = ["transient", "quota", "network", "invalid-response", "unsupported"] as const;
|
|
109
111
|
|
|
110
112
|
type SearchConfig = {
|
|
@@ -362,6 +364,7 @@ async function searchWithResolvedProvider(
|
|
|
362
364
|
if (provider === "serpbase") return { ...(await searchWithSerpBase(query, options)), provider };
|
|
363
365
|
if (provider === "serper") return { ...(await searchWithSerper(query, options)), provider };
|
|
364
366
|
if (provider === "valyu") return { ...(await searchWithValyu(query, options)), provider };
|
|
367
|
+
if (provider === "xcrawl") return { ...(await searchWithXCrawl(query, options)), provider };
|
|
365
368
|
if (provider === "perplexity") return { ...(await searchWithPerplexity(query, options)), provider };
|
|
366
369
|
if (provider === "searxng") return { ...(await searchWithSearXNG(query, options)), provider };
|
|
367
370
|
if (provider === "duckduckgo") return { ...(await searchWithDuckDuckGo(query, options)), provider };
|
|
@@ -408,6 +411,7 @@ async function isResolvedProviderAvailable(provider: ResolvedSearchProvider, opt
|
|
|
408
411
|
if (provider === "serpbase") return isSerpBaseAvailable();
|
|
409
412
|
if (provider === "serper") return isSerperAvailable();
|
|
410
413
|
if (provider === "valyu") return isValyuAvailable();
|
|
414
|
+
if (provider === "xcrawl") return isXcrawlAvailable();
|
|
411
415
|
if (provider === "perplexity") return isPerplexityAvailable();
|
|
412
416
|
if (provider === "searxng") return isSearXNGAvailable();
|
|
413
417
|
if (provider === "duckduckgo") return isDuckDuckGoAvailable();
|
|
@@ -437,6 +441,7 @@ function providerLabel(provider: ResolvedSearchProvider): string {
|
|
|
437
441
|
if (provider === "duckduckgo") return "DuckDuckGo";
|
|
438
442
|
if (provider === "kagi") return "Kagi";
|
|
439
443
|
if (provider === "bocha") return "Bocha";
|
|
444
|
+
if (provider === "xcrawl") return "XCrawl";
|
|
440
445
|
if (provider === "kimi") return "Kimi";
|
|
441
446
|
if (provider === "ollama") return "Ollama";
|
|
442
447
|
if (provider === "xai") return "xAI";
|
|
@@ -470,7 +475,7 @@ async function searchWithProviders(
|
|
|
470
475
|
: await isResolvedProviderAvailable(provider, options),
|
|
471
476
|
})))).filter((entry) => entry.available).map((entry) => entry.provider);
|
|
472
477
|
if (providers.length === 0) {
|
|
473
|
-
throw new Error("No configured search provider available for provider \"all\". Parallel MCP, DuckDuckGo, Kimi, AnySearch, xAI, Bright Data, SerpBase, Serper, and
|
|
478
|
+
throw new Error("No configured search provider available for provider \"all\". Parallel MCP, DuckDuckGo, Kimi, AnySearch, xAI, Bright Data, SerpBase, Serper, Valyu, and XCrawl are excluded.");
|
|
474
479
|
}
|
|
475
480
|
|
|
476
481
|
const settled = await Promise.allSettled(
|
|
@@ -763,7 +768,7 @@ export async function search(query: string, options: FullSearchOptions = {}): Pr
|
|
|
763
768
|
" 3. Set OPENAI_API_KEY, BRAVE_API_KEY, PARALLEL_API_KEY, TINYFISH_API_KEY, SEARCH1API_KEY, SEARCHINFINITY_API_KEY, QUERIT_API_KEY, TAVILY_API_KEY, FIRECRAWL_BASE_URL, JINA_API_KEY, SERPDIVE_API_KEY, KAGI_API_KEY, BOCHA_API_KEY, OLLAMA_API_KEY, SEARXNG_BASE_URL, EXA_API_KEY, PERPLEXITY_API_KEY, GEMINI_API_KEY, or CLOUDFLARE_API_KEY env vars\n" +
|
|
764
769
|
" 4. Set GOOGLE_GEMINI_BASE_URL with CLOUDFLARE_API_KEY for Cloudflare AI Gateway routing\n" +
|
|
765
770
|
" 5. Sign into gemini.google.com in a supported Chromium-based browser\n" +
|
|
766
|
-
" 6. Explicitly select provider: \"anysearch\" for anonymous AnySearch, \"xai\" for Grok, \"brightdata\" with brightdataSerpZone for paid Bright Data SERP, \"serpbase\" or \"serper\" for Google SERP, or \"valyu\" for research search"
|
|
771
|
+
" 6. Explicitly select provider: \"anysearch\" for anonymous AnySearch, \"xcrawl\" for XCrawl, \"xai\" for Grok, \"brightdata\" with brightdataSerpZone for paid Bright Data SERP, \"serpbase\" or \"serper\" for Google SERP, or \"valyu\" for research search"
|
|
767
772
|
);
|
|
768
773
|
}
|
|
769
774
|
|
|
@@ -9,7 +9,8 @@ import { findContent, type FindMode } from "./content-find.ts";
|
|
|
9
9
|
import { answerFromPage } from "./page-query.ts";
|
|
10
10
|
import { rewriteSearchQuery } from "./query-rewrite.ts";
|
|
11
11
|
import { clearCloneCache } from "./github-extract.ts";
|
|
12
|
-
import { getConfiguredSearchRouting, normalizeSearchProviderSelection, RESOLVED_SEARCH_PROVIDERS, SEARCH_PROVIDERS, search, type AttributedSearchResponse, type SearchProvider, type SearchProviderSelection, type ResolvedSearchProvider } from "./gemini-search.ts";
|
|
12
|
+
import { ALL_SEARCH_PROVIDERS, getConfiguredSearchRouting, normalizeSearchProviderSelection, RESOLVED_SEARCH_PROVIDERS, SEARCH_PROVIDERS, search, type AttributedSearchResponse, type ProviderAvailability, type SearchProvider, type SearchProviderSelection, type ResolvedSearchProvider } from "./gemini-search.ts";
|
|
13
|
+
export type { ProviderAvailability } from "./gemini-search.ts";
|
|
13
14
|
import type { SearchResult } from "./perplexity.ts";
|
|
14
15
|
import { formatSeconds, getWebSearchConfigDir, getWebSearchConfigPath, installGlobalProxyFetch, resolveCuratorNetworkConfig, runWithProxy } from "./utils.ts";
|
|
15
16
|
import {
|
|
@@ -68,6 +69,7 @@ import { isBrightDataAvailable } from "./brightdata.ts";
|
|
|
68
69
|
import { isSerpBaseAvailable } from "./serpbase.ts";
|
|
69
70
|
import { isSerperAvailable } from "./serper.ts";
|
|
70
71
|
import { isValyuAvailable } from "./valyu.ts";
|
|
72
|
+
import { isXcrawlAvailable } from "./xcrawl.ts";
|
|
71
73
|
import { buildSearchErrorPlan, type SearchErrorDetails, type SearchErrorPlan } from "./render-search-error.ts";
|
|
72
74
|
import { findModelWithProviderRouting, loadEnabledModelPatterns, modelMatchesEnabledPatterns, splitThinkingSuffix } from "./summary-model-scope.ts";
|
|
73
75
|
import {
|
|
@@ -129,6 +131,7 @@ function renderSearchErrorPlan(plan: SearchErrorPlan, expanded: boolean, theme:
|
|
|
129
131
|
|
|
130
132
|
interface WebSearchConfig {
|
|
131
133
|
anysearchApiKey?: unknown;
|
|
134
|
+
xcrawlApiKey?: unknown;
|
|
132
135
|
brightdataApiKey?: unknown;
|
|
133
136
|
brightdataSerpZone?: unknown;
|
|
134
137
|
kagiApiKey?: unknown;
|
|
@@ -165,37 +168,6 @@ interface WebSearchConfig {
|
|
|
165
168
|
};
|
|
166
169
|
}
|
|
167
170
|
|
|
168
|
-
export interface ProviderAvailability {
|
|
169
|
-
all: boolean;
|
|
170
|
-
openai: boolean;
|
|
171
|
-
brave: boolean;
|
|
172
|
-
parallel: boolean;
|
|
173
|
-
"parallel-mcp": boolean;
|
|
174
|
-
tinyfish: boolean;
|
|
175
|
-
search1api: boolean;
|
|
176
|
-
searchinfinity: boolean;
|
|
177
|
-
querit: boolean;
|
|
178
|
-
tavily: boolean;
|
|
179
|
-
firecrawl: boolean;
|
|
180
|
-
jina: boolean;
|
|
181
|
-
serpdive: boolean;
|
|
182
|
-
searxng: boolean;
|
|
183
|
-
duckduckgo: boolean;
|
|
184
|
-
perplexity: boolean;
|
|
185
|
-
exa: boolean;
|
|
186
|
-
gemini: boolean;
|
|
187
|
-
kimi: boolean;
|
|
188
|
-
kagi: boolean;
|
|
189
|
-
bocha: boolean;
|
|
190
|
-
ollama: boolean;
|
|
191
|
-
anysearch: boolean;
|
|
192
|
-
xai: boolean;
|
|
193
|
-
brightdata: boolean;
|
|
194
|
-
serpbase: boolean;
|
|
195
|
-
serper: boolean;
|
|
196
|
-
valyu: boolean;
|
|
197
|
-
}
|
|
198
|
-
|
|
199
171
|
type WebSearchWorkflow = "none" | "summary-review" | "auto-summary";
|
|
200
172
|
type CuratorWorkflow = "summary-review";
|
|
201
173
|
export type CuratorProvider = Exclude<SearchProvider, "auto">;
|
|
@@ -374,6 +346,33 @@ function normalizeQueryList(queryList: unknown[]): string[] {
|
|
|
374
346
|
return normalized;
|
|
375
347
|
}
|
|
376
348
|
|
|
349
|
+
// Some local models serialize a multi-query list into the single-string `query`
|
|
350
|
+
// field as a JSON array (query: "[\"a\", \"b\"]") instead of using the
|
|
351
|
+
// `queries` parameter. Forwarding that raw string verbatim makes every backend
|
|
352
|
+
// search for the literal array text and return zero results, with no signal to
|
|
353
|
+
// the model that its argument shape was wrong. Expand a string that parses as a
|
|
354
|
+
// JSON array of strings so each element is searched independently.
|
|
355
|
+
function expandQueryString(query: unknown): string[] {
|
|
356
|
+
if (typeof query !== "string") return [];
|
|
357
|
+
const trimmed = query.trim();
|
|
358
|
+
if (trimmed.startsWith("[") && trimmed.endsWith("]")) {
|
|
359
|
+
try {
|
|
360
|
+
const parsed: unknown = JSON.parse(trimmed);
|
|
361
|
+
// Only expand an unambiguously string-only array. Mixed or non-string
|
|
362
|
+
// arrays are kept as the literal query so we never silently drop
|
|
363
|
+
// members or collapse them into an empty search.
|
|
364
|
+
if (Array.isArray(parsed) && parsed.every((entry): entry is string => typeof entry === "string")) {
|
|
365
|
+
return parsed
|
|
366
|
+
.map((entry) => entry.trim())
|
|
367
|
+
.filter((entry) => entry.length > 0);
|
|
368
|
+
}
|
|
369
|
+
} catch {
|
|
370
|
+
// Not JSON — treat as a literal query string.
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
return [query];
|
|
374
|
+
}
|
|
375
|
+
|
|
377
376
|
function getCuratorTimeoutSeconds(): number {
|
|
378
377
|
const source = loadConfig();
|
|
379
378
|
const explicit = normalizeCuratorTimeoutSeconds(source.curatorTimeoutSeconds);
|
|
@@ -422,15 +421,16 @@ async function getProviderAvailability(ctx: ExtensionContext): Promise<ProviderA
|
|
|
422
421
|
gemini: geminiApiAvail || !!geminiWebAvail,
|
|
423
422
|
kimi: await isKimiSearchAvailable(ctx),
|
|
424
423
|
anysearch: isAnySearchAvailable(),
|
|
424
|
+
xcrawl: isXcrawlAvailable(),
|
|
425
425
|
xai: await isXaiSearchAvailable(ctx),
|
|
426
426
|
brightdata: isBrightDataAvailable(),
|
|
427
427
|
serpbase: isSerpBaseAvailable(),
|
|
428
428
|
serper: isSerperAvailable(),
|
|
429
429
|
valyu: isValyuAvailable(),
|
|
430
430
|
};
|
|
431
|
+
const allSearchProviders = new Set<ResolvedSearchProvider>(ALL_SEARCH_PROVIDERS);
|
|
431
432
|
return {
|
|
432
|
-
|
|
433
|
-
all: Object.entries(providers).some(([provider, available]) => provider !== "parallel-mcp" && provider !== "duckduckgo" && provider !== "kimi" && provider !== "anysearch" && provider !== "valyu" && provider !== "xai" && provider !== "brightdata" && provider !== "serpbase" && provider !== "serper" && provider !== "gemini" && available) || geminiApiAvail,
|
|
433
|
+
all: Object.entries(providers).some(([provider, available]) => provider !== "gemini" && allSearchProviders.has(provider as ResolvedSearchProvider) && available) || geminiApiAvail,
|
|
434
434
|
...providers,
|
|
435
435
|
};
|
|
436
436
|
}
|
|
@@ -1750,7 +1750,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1750
1750
|
name: toolNames.webSearch,
|
|
1751
1751
|
label: "Web Search",
|
|
1752
1752
|
description:
|
|
1753
|
-
`Search the web using OpenAI, Brave, Parallel, Parallel MCP, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, SearXNG, DuckDuckGo, Exa, Perplexity, Gemini, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, or Serper. Pass a provider array to search only those providers simultaneously, or use provider "all" to search every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper. Returns an AI-synthesized answer with source citations. OpenAI search uses a Codex subscription or OpenAI API key; Kimi search uses a Kimi Code Plan authenticated through /login kimi-coding; xAI search uses a SuperGrok/X Premium subscription or xAI API key. Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper are available only when explicitly selected. For comprehensive research, prefer queries (plural) with 2-4 varied angles over a single query — each query gets its own synthesized answer, so varying phrasing and scope gives much broader coverage. When includeContent is true, full page content is fetched in the background. Searches auto-open the interactive browser curator and stream results live; set workflow to "none" to skip curation or "auto-summary" for a model-generated summary without the browser curator. The configured provider is used when provider is omitted or set to auto; omit provider unless explicitly overriding it. Without a configured provider, SearXNG is preferred first for local/private search. When the active Pi model is openai-codex, Codex-backed OpenAI search is preferred next. Otherwise Exa is preferred before OpenAI, then Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, or Gemini Web.`,
|
|
1753
|
+
`Search the web using OpenAI, Brave, Parallel, Parallel MCP, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, SearXNG, DuckDuckGo, Exa, Perplexity, Gemini, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, or Serper. Pass a provider array to search only those providers simultaneously, or use provider "all" to search every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper. Returns an AI-synthesized answer with source citations. OpenAI search uses a Codex subscription or OpenAI API key; Kimi search uses a Kimi Code Plan authenticated through /login kimi-coding; xAI search uses a SuperGrok/X Premium subscription or xAI API key. Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper are available only when explicitly selected. For comprehensive research, prefer queries (plural) with 2-4 varied angles over a single query — each query gets its own synthesized answer, so varying phrasing and scope gives much broader coverage. When includeContent is true, full page content is fetched in the background. Searches auto-open the interactive browser curator and stream results live; set workflow to "none" to skip curation or "auto-summary" for a model-generated summary without the browser curator. The configured provider is used when provider is omitted or set to auto; omit provider unless explicitly overriding it. Without a configured provider, SearXNG is preferred first for local/private search. When the active Pi model is openai-codex, Codex-backed OpenAI search is preferred next. Otherwise Exa is preferred before OpenAI, then Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, or Gemini Web.`,
|
|
1754
1754
|
promptSnippet:
|
|
1755
1755
|
"Use for web research questions. Prefer {queries:[...]} with 2-4 varied angles over a single query for broader coverage. Omit provider unless explicitly overriding the configured default.",
|
|
1756
1756
|
parameters: Type.Object({
|
|
@@ -1762,7 +1762,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1762
1762
|
StringEnum(["day", "week", "month", "year"], { description: "Filter by recency" }),
|
|
1763
1763
|
),
|
|
1764
1764
|
domainFilter: Type.Optional(Type.Array(Type.String(), { description: "Limit to domains (prefix with - to exclude)" })),
|
|
1765
|
-
provider: Type.Optional(searchProviderSchema("Search provider or non-empty list of providers to search simultaneously; use all to search every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper, omit this field to use the configured provider, or use auto when none is configured")),
|
|
1765
|
+
provider: Type.Optional(searchProviderSchema("Search provider or non-empty list of providers to search simultaneously; use all to search every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper, omit this field to use the configured provider, or use auto when none is configured")),
|
|
1766
1766
|
workflow: Type.Optional(
|
|
1767
1767
|
StringEnum(["none", "summary-review", "auto-summary"], {
|
|
1768
1768
|
description: "Search workflow mode: none = no curator, summary-review = open curator with auto summary draft (default), auto-summary = generate summary without opening curator",
|
|
@@ -1777,7 +1777,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1777
1777
|
return runWithProxy(typeof params.proxy === "string" ? params.proxy : undefined, async () => {
|
|
1778
1778
|
const rawQueryList: unknown[] = Array.isArray(params.queries)
|
|
1779
1779
|
? params.queries
|
|
1780
|
-
: (params.query !== undefined ?
|
|
1780
|
+
: (params.query !== undefined ? expandQueryString(params.query) : []);
|
|
1781
1781
|
const queryList = normalizeQueryList(rawQueryList);
|
|
1782
1782
|
const configWorkflow = loadConfigForExtensionInit().workflow;
|
|
1783
1783
|
const workflow = resolveWorkflow(params.workflow ?? configWorkflow, ctx?.hasUI !== false);
|
|
@@ -2073,7 +2073,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2073
2073
|
const input = args as { query?: unknown; queries?: unknown };
|
|
2074
2074
|
const rawQueryList: unknown[] = Array.isArray(input.queries)
|
|
2075
2075
|
? input.queries
|
|
2076
|
-
: (input.query !== undefined ?
|
|
2076
|
+
: (input.query !== undefined ? expandQueryString(input.query) : []);
|
|
2077
2077
|
const queryList = normalizeQueryList(rawQueryList);
|
|
2078
2078
|
if (queryList.length === 0) {
|
|
2079
2079
|
return new Text(theme.fg("toolTitle", theme.bold("search ")) + theme.fg("error", "(no query)"), 0, 0);
|
|
@@ -2337,7 +2337,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2337
2337
|
fetchContent: Type.Optional(Type.Boolean({ description: "Fetch up to 5 result pages for exact passage extraction." })),
|
|
2338
2338
|
recencyFilter: Type.Optional(StringEnum(["day", "week", "month", "year"], { description: "Filter by recency." })),
|
|
2339
2339
|
domainFilter: Type.Optional(Type.Array(Type.String(), { description: "Limit to domains; prefix with - to exclude." })),
|
|
2340
|
-
provider: Type.Optional(searchProviderSchema("Search provider or non-empty list of providers to search simultaneously; all searches every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, Valyu, xAI, Bright Data, SerpBase, and Serper")),
|
|
2340
|
+
provider: Type.Optional(searchProviderSchema("Search provider or non-empty list of providers to search simultaneously; all searches every eligible provider except Parallel MCP, DuckDuckGo, Kimi, AnySearch, XCrawl, Valyu, xAI, Bright Data, SerpBase, and Serper")),
|
|
2341
2341
|
proxy: Type.Optional(Type.String({
|
|
2342
2342
|
description: "http(s) proxy URL (e.g. http://host:port) used for every outbound request in this call (search APIs and result-page fetches). Empty string forces direct access.",
|
|
2343
2343
|
})),
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-web-access",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.26.0",
|
|
4
4
|
"description": "Web search, URL fetching, GitHub repo cloning, PDF extraction, YouTube video understanding, and local video analysis for Pi coding agent. Supports OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Ollama, AnySearch, Bright Data SERP, SerpBase, SearXNG, Exa, Perplexity, Gemini, and Kimi.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"scripts": {
|