@herbertgao/pi-extensions 2026.8.5 → 2026.8.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/node_modules/@herbertgao/pi-cc-extensions/README.en.md +1 -1
- package/node_modules/@herbertgao/pi-cc-extensions/README.md +1 -1
- package/node_modules/@herbertgao/pi-cc-extensions/extensions/feature/context.ts +74 -5
- package/node_modules/@herbertgao/pi-cc-extensions/extensions/renderer/markdown-enhance.ts +48 -6
- package/node_modules/@herbertgao/pi-cc-extensions/package.json +3 -3
- package/node_modules/@herbertgao/pi-subagents/CHANGELOG.md +10 -0
- package/node_modules/@herbertgao/pi-subagents/package.json +3 -3
- package/node_modules/@herbertgao/pi-subagents/src/agent-runner.ts +5 -2
- package/node_modules/@herbertgao/pi-subagents/src/ui/fleet-list.ts +15 -6
- package/node_modules/@herbertgao/pi-subagents/src/worktree.ts +9 -5
- package/node_modules/@juicesharp/rpiv-ask-user-question/README.md +2 -0
- package/node_modules/@juicesharp/rpiv-ask-user-question/ask-user-question.ts +20 -0
- package/node_modules/@juicesharp/rpiv-ask-user-question/docs/hosts.md +6 -0
- package/node_modules/@juicesharp/rpiv-ask-user-question/package.json +2 -2
- package/node_modules/@narumitw/pi-btw/README.md +24 -17
- package/node_modules/@narumitw/pi-btw/package.json +5 -5
- package/node_modules/@narumitw/pi-btw/src/btw.ts +4 -2
- package/node_modules/@narumitw/pi-btw/src/menu.ts +33 -13
- package/node_modules/@narumitw/pi-btw/src/settings.ts +22 -2
- package/node_modules/pi-lens/CHANGELOG.md +95 -0
- package/node_modules/pi-lens/dist/clients/advisory-provenance.js +314 -0
- package/node_modules/pi-lens/dist/clients/agent-nudge.js +14 -7
- package/node_modules/pi-lens/dist/clients/biome-client.js +121 -13
- package/node_modules/pi-lens/dist/clients/bus-events-logger.js +62 -6
- package/node_modules/pi-lens/dist/clients/bus-publish.js +11 -3
- package/node_modules/pi-lens/dist/clients/cascade-format.js +57 -2
- package/node_modules/pi-lens/dist/clients/console-guard-install.js +16 -4
- package/node_modules/pi-lens/dist/clients/dead-code-client.js +135 -30
- package/node_modules/pi-lens/dist/clients/dependency-checker.js +19 -7
- package/node_modules/pi-lens/dist/clients/diagnostic-dispositions.js +6 -4
- package/node_modules/pi-lens/dist/clients/diagnostics-publish.js +10 -3
- package/node_modules/pi-lens/dist/clients/dispatch/dispatcher.js +114 -10
- package/node_modules/pi-lens/dist/clients/dispatch/integration.js +101 -16
- package/node_modules/pi-lens/dist/clients/dispatch/runners/lsp.js +37 -4
- package/node_modules/pi-lens/dist/clients/dispatch/runners/utils/availability-policy.js +226 -0
- package/node_modules/pi-lens/dist/clients/dispatch/runners/utils/candidate-probe.js +69 -0
- package/node_modules/pi-lens/dist/clients/dispatch/runners/utils/runner-helpers.js +230 -50
- package/node_modules/pi-lens/dist/clients/dispatch/runners/utils/toolchain-availability.js +97 -0
- package/node_modules/pi-lens/dist/clients/disposition-publish.js +10 -3
- package/node_modules/pi-lens/dist/clients/eval-timestamp.js +17 -0
- package/node_modules/pi-lens/dist/clients/extension-log.js +296 -3
- package/node_modules/pi-lens/dist/clients/fix-worklog.js +5 -1
- package/node_modules/pi-lens/dist/clients/format-events-publish.js +39 -8
- package/node_modules/pi-lens/dist/clients/git-guard.js +18 -21
- package/node_modules/pi-lens/dist/clients/go-client.js +21 -39
- package/node_modules/pi-lens/dist/clients/govulncheck-client.js +116 -7
- package/node_modules/pi-lens/dist/clients/host-ports.js +1 -1
- package/node_modules/pi-lens/dist/clients/installer/index.js +5 -1
- package/node_modules/pi-lens/dist/clients/jscpd-client.js +4 -9
- package/node_modules/pi-lens/dist/clients/knip-client.js +51 -14
- package/node_modules/pi-lens/dist/clients/latency-logger.js +50 -1
- package/node_modules/pi-lens/dist/clients/lens-events.js +56 -25
- package/node_modules/pi-lens/dist/clients/lens-flag-registry.js +8 -0
- package/node_modules/pi-lens/dist/clients/live-bus-emitter.js +45 -2
- package/node_modules/pi-lens/dist/clients/lsp/aggregation.js +30 -4
- package/node_modules/pi-lens/dist/clients/lsp/cascade-tier.js +58 -13
- package/node_modules/pi-lens/dist/clients/lsp/client.js +169 -8
- package/node_modules/pi-lens/dist/clients/lsp/diagnostic-binding.js +29 -0
- package/node_modules/pi-lens/dist/clients/lsp/index.js +264 -66
- package/node_modules/pi-lens/dist/clients/lsp/server.js +247 -60
- package/node_modules/pi-lens/dist/clients/lsp/tsserver-sync.js +96 -0
- package/node_modules/pi-lens/dist/clients/lsp/wait-policy/classification.js +21 -5
- package/node_modules/pi-lens/dist/clients/lsp/wait-policy/strategies.js +18 -3
- package/node_modules/pi-lens/dist/clients/mcp/analyze.js +4 -0
- package/node_modules/pi-lens/dist/clients/mcp/session.js +30 -17
- package/node_modules/pi-lens/dist/clients/model-provider.js +53 -0
- package/node_modules/pi-lens/dist/clients/pipeline.js +29 -2
- package/node_modules/pi-lens/dist/clients/project-diagnostics/fresh-fetch.js +1 -1
- package/node_modules/pi-lens/dist/clients/project-diagnostics/runner-adapters/runner-findings.js +23 -2
- package/node_modules/pi-lens/dist/clients/review-graph/query.js +24 -0
- package/node_modules/pi-lens/dist/clients/run-duration.js +55 -0
- package/node_modules/pi-lens/dist/clients/runtime-agent-end.js +153 -7
- package/node_modules/pi-lens/dist/clients/runtime-context.js +104 -11
- package/node_modules/pi-lens/dist/clients/runtime-coordinator.js +170 -22
- package/node_modules/pi-lens/dist/clients/runtime-session.js +28 -5
- package/node_modules/pi-lens/dist/clients/runtime-tool-result.js +124 -10
- package/node_modules/pi-lens/dist/clients/runtime-turn.js +388 -35
- package/node_modules/pi-lens/dist/clients/rust-client.js +21 -37
- package/node_modules/pi-lens/dist/clients/security-scan-client.js +88 -5
- package/node_modules/pi-lens/dist/clients/sg-runner.js +141 -23
- package/node_modules/pi-lens/dist/clients/smells-rollup.js +18 -11
- package/node_modules/pi-lens/dist/clients/startup-timing.js +7 -1
- package/node_modules/pi-lens/dist/clients/test-runner-client.js +431 -24
- package/node_modules/pi-lens/dist/clients/tool-policy.js +2 -0
- package/node_modules/pi-lens/dist/clients/tool-set-policy.js +76 -0
- package/node_modules/pi-lens/dist/clients/warm-attach.js +17 -0
- package/node_modules/pi-lens/dist/clients/word-index.js +305 -33
- package/node_modules/pi-lens/dist/index.js +4827 -1633
- package/node_modules/pi-lens/dist/mcp/server.js +8 -2
- package/node_modules/pi-lens/dist/tools/activate-tools.js +17 -5
- package/node_modules/pi-lens/dist/tools/ast-grep-replace.js +9 -4
- package/node_modules/pi-lens/dist/tools/lens-diagnostic-mark.js +5 -2
- package/node_modules/pi-lens/dist/tools/lens-diagnostics.js +3 -2
- package/node_modules/pi-lens/dist/tools/lsp-diagnostics.js +62 -7
- package/node_modules/pi-lens/dist/tools/symbol-search.js +1 -1
- package/node_modules/pi-lens/docs/agent-guide.md +38 -13
- package/node_modules/pi-lens/docs/ast-grep_rules_catalog.md +11 -3
- package/node_modules/pi-lens/docs/features.md +14 -1
- package/node_modules/pi-lens/docs/globalconfig.md +11 -0
- package/node_modules/pi-lens/docs/servercapabilities.md +1 -1
- package/node_modules/pi-lens/docs/settings.md +6 -0
- package/node_modules/pi-lens/docs/usage.md +23 -5
- package/node_modules/pi-lens/docs/word-index.md +35 -0
- package/node_modules/pi-lens/package.json +1 -1
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-chained-type-assertions-test.yml +8 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-conditional-empty-object-spread-js-test.yml +9 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-conditional-empty-object-spread-test.yml +9 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-reflect-apply-js-test.yml +7 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-reflect-apply-test.yml +7 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-reflect-get-js-test.yml +8 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-reflect-get-test.yml +8 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rule-tests/no-unknown-laundering-test.yml +11 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-chained-type-assertions.yml +21 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-conditional-empty-object-spread-js.yml +21 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-conditional-empty-object-spread.yml +29 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-reflect-apply-js.yml +9 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-reflect-apply.yml +9 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-reflect-get-js.yml +13 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-reflect-get.yml +16 -0
- package/node_modules/pi-lens/rules/ast-grep-rules/rules/no-unknown-laundering.yml +27 -0
- package/node_modules/pi-lens/scripts/analyze-pi-lens-logs.mjs +79 -0
- package/node_modules/pi-lens/skills/pi-lens-ast-grep/SKILL.md +10 -11
- package/node_modules/pi-lens/skills/pi-lens-lsp-navigation/SKILL.md +22 -22
- package/node_modules/pi-lens/skills/pi-lens-write-ast-grep-rule/SKILL.md +8 -114
- package/node_modules/pi-lens/skills/pi-lens-write-ast-grep-rule/reference.md +129 -0
- package/node_modules/pi-lens/skills/pi-lens-write-tree-sitter-rule/SKILL.md +3 -1
- package/node_modules/pi-mcp-adapter/CHANGELOG.md +13 -0
- package/node_modules/pi-mcp-adapter/README.md +4 -1
- package/node_modules/pi-mcp-adapter/agent-dir.ts +12 -4
- package/node_modules/pi-mcp-adapter/cli.js +25 -4
- package/node_modules/pi-mcp-adapter/config.ts +4 -4
- package/node_modules/pi-mcp-adapter/direct-tools.ts +18 -15
- package/node_modules/pi-mcp-adapter/mcp-setup-panel.ts +2 -1
- package/node_modules/pi-mcp-adapter/metadata-cache.ts +18 -8
- package/node_modules/pi-mcp-adapter/package.json +2 -1
- package/node_modules/pi-mcp-adapter/request-headers-command.ts +336 -0
- package/node_modules/pi-mcp-adapter/server-manager.ts +5 -0
- package/node_modules/pi-mcp-adapter/tool-metadata.ts +38 -18
- package/node_modules/pi-mcp-adapter/types.ts +93 -10
- package/node_modules/pi-web-access/CHANGELOG.md +14 -0
- package/node_modules/pi-web-access/README.md +18 -12
- package/node_modules/pi-web-access/auth-fetch.ts +148 -0
- package/node_modules/pi-web-access/chrome-cookies.ts +110 -23
- package/node_modules/pi-web-access/curator-page.ts +5 -3
- package/node_modules/pi-web-access/curator-server.ts +2 -1
- package/node_modules/pi-web-access/extract.ts +106 -34
- package/node_modules/pi-web-access/fetch-params.ts +17 -3
- package/node_modules/pi-web-access/firecrawl.ts +172 -12
- package/node_modules/pi-web-access/gemini-search.ts +18 -4
- package/node_modules/pi-web-access/index.ts +120 -48
- package/node_modules/pi-web-access/package.json +2 -2
- package/node_modules/pi-web-access/summary-review.ts +11 -5
- package/node_modules/pi-web-access/youtube-extract.ts +2 -2
- package/package.json +9 -9
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Pi Web Access
|
|
6
6
|
|
|
7
|
-
**Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI/Grok, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, or bring your own API keys.**
|
|
7
|
+
**Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI/Grok, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, or bring your own API keys.**
|
|
8
8
|
|
|
9
9
|
[](https://www.npmjs.com/package/pi-web-access)
|
|
10
10
|
[](https://opensource.org/licenses/MIT)
|
|
@@ -14,11 +14,11 @@
|
|
|
14
14
|
|
|
15
15
|
## Why Pi Web Access
|
|
16
16
|
|
|
17
|
-
**Zero Config** — Works out of the box with Exa MCP (no API key needed). If you're signed into Pi with a Codex subscription, OpenAI web search can reuse that auth. Add API keys for OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, SerpBase, Exa, Perplexity, or Gemini API for more control; configure a self-hosted SearXNG endpoint for private search; or opt into browser-cookie access for Gemini Web.
|
|
17
|
+
**Zero Config** — Works out of the box with Exa MCP (no API key needed). If you're signed into Pi with a Codex subscription, OpenAI web search can reuse that auth. Add API keys or endpoints for OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, SerpBase, Exa, Perplexity, or Gemini API for more control; configure a self-hosted SearXNG endpoint for private search; or opt into browser-cookie access for Gemini Web.
|
|
18
18
|
|
|
19
19
|
**Video Understanding** — Point it at a YouTube video or local screen recording and ask questions about what's on screen. Full transcripts, visual descriptions, and frame extraction at exact timestamps.
|
|
20
20
|
|
|
21
|
-
**Smart Fallbacks** — Every capability has a fallback chain. Search tries configured SearXNG first for local/private search, then OpenAI when suitable and available, Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, and Gemini Web when browser cookies are enabled. With no SearXNG configured, the existing zero-config order is unchanged. YouTube tries Gemini Web when enabled, then API, then Perplexity. Blocked pages try configured self-hosted Firecrawl first. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
|
|
21
|
+
**Smart Fallbacks** — Every capability has a fallback chain. Search tries configured SearXNG first for local/private search, then OpenAI when suitable and available, Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, and Gemini Web when browser cookies are enabled. With no SearXNG configured, the existing zero-config order is unchanged. YouTube tries Gemini Web when enabled, then API, then Perplexity. Blocked pages try configured self-hosted Firecrawl first. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
|
|
22
22
|
|
|
23
23
|
**GitHub Cloning** — GitHub URLs are cloned locally instead of scraped. The agent gets real file contents and a local path to explore, not rendered HTML.
|
|
24
24
|
|
|
@@ -46,7 +46,7 @@ Works immediately with no API keys — Exa MCP provides zero-config search. If P
|
|
|
46
46
|
}
|
|
47
47
|
```
|
|
48
48
|
|
|
49
|
-
In `auto` mode (default), `web_search` tries a configured SearXNG endpoint first for local/private search, then OpenAI when suitable and available, Exa (direct API if keyed, MCP if not), Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Perplexity, Gemini API, and Gemini Web when browser-cookie access is enabled. With no SearXNG configured, the existing zero-config order is unchanged. Exa handles search; curator summary drafts are generated separately by the configured Pi summary model. Slow summary drafts fall back to a deterministic result summary after a bounded deadline.
|
|
49
|
+
In `auto` mode (default), `web_search` tries a configured SearXNG endpoint first for local/private search, then OpenAI when suitable and available, Exa (direct API if keyed, MCP if not), Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Perplexity, Gemini API, and Gemini Web when browser-cookie access is enabled. With no SearXNG configured, the existing zero-config order is unchanged. Exa handles search; curator summary drafts are generated separately by the configured Pi summary model. Slow summary drafts fall back to a deterministic result summary after a bounded deadline.
|
|
50
50
|
|
|
51
51
|
If your OpenAI key belongs to a third-party Responses-compatible gateway, set `openaiResponsesUrl` to that gateway's full Responses endpoint. The default remains `https://api.openai.com/v1/responses`.
|
|
52
52
|
|
|
@@ -96,7 +96,7 @@ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on sc
|
|
|
96
96
|
|
|
97
97
|
### web_search
|
|
98
98
|
|
|
99
|
-
Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, or Gemini. Returns a synthesized answer with source citations.
|
|
99
|
+
Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Firecrawl, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, or Gemini. Returns a synthesized answer with source citations.
|
|
100
100
|
|
|
101
101
|
```typescript
|
|
102
102
|
web_search({ query: "rust async programming" })
|
|
@@ -117,7 +117,7 @@ web_search({ queries: ["query 1", "query 2"], workflow: "auto-summary" })
|
|
|
117
117
|
| `numResults` | Results per query (default: 5, max: 20) |
|
|
118
118
|
| `recencyFilter` | `day`, `week`, `month`, or `year` |
|
|
119
119
|
| `domainFilter` | Limit to domains (prefix with `-` to exclude) |
|
|
120
|
-
| `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase simultaneously; otherwise `openai`, `brave`, `parallel`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `jina`, `serpdive`, `kagi`, `bocha`, `ollama`, `anysearch`, `xai`, `brightdata`, `serpbase`, `searxng`, `duckduckgo`, `exa`, `perplexity`, or `gemini` (auto-selects when no provider or routing is configured; DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are explicit-only) |
|
|
120
|
+
| `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase simultaneously; otherwise `openai`, `brave`, `parallel`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `firecrawl`, `jina`, `serpdive`, `kagi`, `bocha`, `ollama`, `anysearch`, `xai`, `brightdata`, `serpbase`, `searxng`, `duckduckgo`, `exa`, `perplexity`, or `gemini` (auto-selects when no provider or routing is configured; DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are explicit-only) |
|
|
121
121
|
| `includeContent` | Fetch full page content from sources in background |
|
|
122
122
|
| `workflow` | `none` (skip curator), `summary-review` (open curator and auto-generate a summary draft, default), or `auto-summary` (generate a summary without opening the curator) |
|
|
123
123
|
|
|
@@ -134,6 +134,7 @@ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on sc
|
|
|
134
134
|
fetch_content({ url: "https://youtube.com/watch?v=abc", timestamp: "23:41-25:00", frames: 4 })
|
|
135
135
|
fetch_content({ url: "https://example.com/api", mode: "raw" })
|
|
136
136
|
fetch_content({ url: "https://example.com/guide", mode: "answer", prompt: "What are the installation steps?" })
|
|
137
|
+
fetch_content({ url: "https://example.com/account", auth: "work", mode: "raw" })
|
|
137
138
|
fetch_content({ url: "https://example.com/diagram.png" })
|
|
138
139
|
```
|
|
139
140
|
|
|
@@ -247,13 +248,15 @@ Env vars: `DATALAB_API_KEY` (or `datalabApiKey` in config), `DATALAB_PROCESSING_
|
|
|
247
248
|
|
|
248
249
|
Raw and direct-image HTTP requests use the same SSRF validation, hostname domain policy, redirect checks, timeout, and 5MB streamed response bound as normal extraction. Raw mode returns textual bodies even for non-2xx responses and exposes the HTTP status in tool details; it does not run readability or hosted extraction fallbacks.
|
|
249
250
|
|
|
251
|
+
`fetch_content` can opt into local browser-cookie auth with `auth: "profile"`, or `auth: true` when exactly one `authFetch` profile exists. Configure profiles in `~/.pi/web-search.json`, for example `{ "authFetch": { "social": ["x.com", "instagram.com"], "work": { "hosts": ["docs.company.com"], "chromeProfile": "Profile 2", "cache": "off" } } }`. Auth fetch uses only the local direct HTTP path, requires HTTPS, allows only configured hosts and their subdomains, refuses cross-origin redirects, and never sends cookies or authenticated content to hosted extraction providers. Browser cookie extraction remains opt-in through `allowBrowserCookies: true` or `PI_ALLOW_BROWSER_COOKIES=1`.
|
|
252
|
+
|
|
250
253
|
When Readability fails or returns only a cookie notice, the extension can retry configured Firecrawl extraction, Jina Reader (handles JS rendering server-side, no API key needed), TinyFish, Search1API, Querit, Kagi Extract, Ollama Web Fetch, Parallel, Bright Data Web Unlocker, Gemini URL Context API, and Gemini Web extraction when browser cookies are enabled. Configure `fetchRouting.providers` to change the order or set of `fetch_content` providers. Supported values are `http`, `firecrawl`, `jina`, `tinyfish`, `search1api`, `querit`, `kagi`, `ollama`, `parallel`, `brightdata`, and `gemini`; when absent, the default order is unchanged. For remote HTTP(S) targets, third-party hosted providers are disabled unless `fetchRouting.allowRemoteHostedProviders` is `true`, because hosted services perform their own fetch and can see a different redirect chain than the local safety gate. Firecrawl stays available as a configured extraction service. Firecrawl requests are cache-only by default and require an explicit fresh-scrape opt-in before the Firecrawl server can fetch target URLs. Bright Data Web Unlocker runs last of the remote scraping providers, ahead of only the Gemini fallbacks, because it is billed per request against a paid account; it is skipped unless both a key and an `unblocker` zone are configured. It applies no minimum-length check, so any non-empty body it returns — including a short consent or paywall stub — is the final answer for that URL and the Gemini fallbacks are not tried. Handles SPAs, JS-heavy pages, and anti-bot protections transparently. Also parses Next.js RSC flight data when present. HTML extraction also surfaces registered discovery relations (`service-desc`, `service-doc`, `service-meta`, `api-catalog`, `describedby`) from the HTTP `Link` header and matching `link`/`a[rel]` markup. Readable or rendered content remains primary; on an empty shell, the normal extraction fallbacks run before declared links are returned on their own.
|
|
251
254
|
|
|
252
255
|
## How It Works
|
|
253
256
|
|
|
254
257
|
```
|
|
255
258
|
web_search(query)
|
|
256
|
-
→ SearXNG (if configured) → OpenAI (when suitable) → Exa → Brave → Parallel → TinyFish → Search1API → Searchinfinity → Querit → Tavily → Jina → SERPdive → Perplexity → Gemini
|
|
259
|
+
→ SearXNG (if configured) → OpenAI (when suitable) → Exa → Brave → Parallel → TinyFish → Search1API → Searchinfinity → Querit → Tavily → Firecrawl → Jina → SERPdive → Perplexity → Gemini
|
|
257
260
|
|
|
258
261
|
fetch_content(url)
|
|
259
262
|
→ Video file? Gemini API (Files API) → Gemini Web (if browser cookies enabled)
|
|
@@ -438,13 +441,15 @@ This syntax applies to provider credentials only; other configuration fields are
|
|
|
438
441
|
|
|
439
442
|
A command source is not run while the extension loads or registers tools. Each selected provider request runs it again with a five-second timeout, a 16 KiB output limit, a minimized environment, and a one-line non-empty stdout requirement. Command text and stderr are omitted from errors. These commands are trusted local configuration, not a same-user process isolation boundary; use absolute executable paths and protect the config file. `OP_SESSION_*` variables are forwarded to trusted resolver commands so shell-local 1Password sessions can be reused without storing them in config. An explicit source overrides legacy provider environment variables and fails that provider locally rather than falling back with a stale credential. Direct Google Gemini API requests send the resolved key only in the `x-goog-api-key` header, never in the URL.
|
|
440
443
|
|
|
444
|
+
`authFetch` configures named local browser-cookie auth profiles for explicit `fetch_content` calls. A profile can be a host array (`"work": ["docs.company.com"]`) or an object with `hosts`, optional `chromeProfile`, `redirects: "same-origin"`, and `cache: "session" | "off"`.
|
|
445
|
+
|
|
441
446
|
`fetchContent.domainPolicy` is an optional hostname allow/deny policy for `fetch_content` target URLs. It is off when omitted. Each bare hostname matches itself and its subdomains; `deny` wins when a hostname matches both lists. The policy is checked before HTTP(S) target handling and before each redirect followed by this extension's own fetch path. Local file paths and non-HTTP sources are not subject to this policy. It is an additional restriction: the existing SSRF guard still blocks private and internal destinations. Remote extraction services can still perform their own DNS, redirects, and egress after this extension preflights the submitted target URL, so third-party hosted HTTP(S) fallbacks stay disabled unless `fetchRouting.allowRemoteHostedProviders` is enabled for separately isolated provider deployments.
|
|
442
447
|
|
|
443
448
|
Set `searxngBaseUrl` or `SEARXNG_BASE_URL` to use a self-hosted SearXNG JSON API. A configured endpoint is preferred first in `auto` mode for local/private search. Its base URL and redirects remain subject to the SSRF guard; add only the narrowest self-hosted range to `ssrf.allowRanges` when it resolves to a private or synthetic range. Optional `searxngHeaders` merges extra HTTP headers into each SearXNG request (string values only; invalid header names are ignored), which is useful for reverse-proxy or Zero Trust auth such as Cloudflare Access service tokens (`CF-Access-Client-Id` / `CF-Access-Client-Secret`). Configured headers override the default `Accept: application/json` when the same name is supplied. Thanks to Marcos A. Núñez (@marnunez) for PR #107 and Avinash Kanaujiya (@avinashkanaujiya) for issue #105.
|
|
444
449
|
|
|
445
450
|
**DuckDuckGo.** DuckDuckGo HTML search is keyless and explicit-only. Select it with `provider: "duckduckgo"` or place it in `searchRouting`; it is never chosen by `auto` and never participates in `provider: "all"`. Domain filters are enforced locally after DuckDuckGo redirect URLs are decoded. `recencyFilter` is not guaranteed because the HTML endpoint has no documented stable time parameter. A 200 page with no parseable results is reported as an invalid response, so routing can continue when `fallbackOn` includes `"invalid-response"`.
|
|
446
451
|
|
|
447
|
-
Set `firecrawlBaseUrl` or `FIRECRAWL_BASE_URL` to use Firecrawl as an extraction
|
|
452
|
+
Set `firecrawlBaseUrl` or `FIRECRAWL_BASE_URL` to use Firecrawl for `web_search` and as an extraction fallback for `fetch_content`. Search calls `/v2/search` by default, requests `sources: ["web"]`, maps `numResults` to `limit`, maps `recencyFilter` to Firecrawl's `tbs`, and maps domain filters to `includeDomains` or `excludeDomains` when possible. With `includeContent: true`, Firecrawl search adds Markdown scrape options and returns successful result Markdown as inline content. Fetch extraction calls `/v2/scrape` by default. Set `firecrawlApiVersion` or `FIRECRAWL_API_VERSION` to `v1` for older self-hosted images. Firecrawl page scraping is cache-only by default (`lockdown: true`), so the Firecrawl server does not make fresh outbound target requests unless you explicitly set `firecrawlFreshScrape: true` or `FIRECRAWL_FRESH_SCRAPE=1`. Enable fresh scraping only for a Firecrawl deployment whose own egress, redirects, DNS rebinding behavior, and internal-network access are isolated or allowlisted; this extension can preflight submitted fetch URLs but cannot control network requests made by the Firecrawl server. The configured Firecrawl API base URL and redirects are still validated by the same SSRF guard as other remote requests, and Firecrawl credentials are stripped from cross-origin API redirects.
|
|
448
453
|
|
|
449
454
|
**Bright Data.** Set `brightdataApiKey` or `BRIGHTDATA_API_KEY` to use Bright Data-backed features. The SERP search provider also requires `brightdataSerpZone` or `BRIGHTDATA_SERP_ZONE`, and the Web Unlocker extraction fallback also requires `brightdataUnlockerZone` or `BRIGHTDATA_UNLOCKER_ZONE`. These zone settings are separate and are never substituted for each other: SERP requires a Bright Data zone of type `serp`, while Web Unlocker requires a zone of type `unblocker`. Leaving either zone unset keeps that product unavailable, so enabling one Bright Data feature does not opt into the other.
|
|
450
455
|
|
|
@@ -458,13 +463,13 @@ Bright Data Web Unlocker is a paid `fetch_content` fallback after Parallel and b
|
|
|
458
463
|
|
|
459
464
|
**SerpBase.** Set `serpbaseApiKey` or `SERPBASE_API_KEY` and select `provider: "serpbase"` to query SerpBase's Google Search Results API. SerpBase is explicit-only: it is never chosen by `auto` and never participates in `provider: "all"`, because each request can consume paid Google SERP credits. Domain filters are sent as Google `site:` clauses and reapplied locally; recency maps to Google's `tbs` time filter.
|
|
460
465
|
|
|
461
|
-
Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `BOCHA_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"jina"`, `"serpdive"`, `"kagi"`, `"bocha"`, `"ollama"`, `"anysearch"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. AnySearch, xAI, Bright Data, and SerpBase are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `chromeProfile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `maxInlineContentChars` sets the direct `fetch_content` content slice and the default and maximum `get_search_content` slice. It defaults to `30000`, must be a positive integer, and is capped at `200000`; full fetched content remains stored for later retrieval. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
|
|
466
|
+
Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `BOCHA_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"firecrawl"`, `"jina"`, `"serpdive"`, `"kagi"`, `"bocha"`, `"ollama"`, `"anysearch"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. AnySearch, xAI, Bright Data, and SerpBase are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `chromeProfile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. macOS discovery supports Helium, Chrome, Brave, and Arc; Linux discovery supports Chromium and Chrome. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `maxInlineContentChars` sets the direct `fetch_content` content slice and the default and maximum `get_search_content` slice. It defaults to `30000`, must be a positive integer, and is capped at `200000`; full fetched content remains stored for later retrieval. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
|
|
462
467
|
|
|
463
468
|
### All providers
|
|
464
469
|
|
|
465
|
-
Set `provider: "all"` on `web_search` or `source_check`, or configure `"provider": "all"` as the default, to run the same query against every eligible search provider simultaneously. DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are always excluded because they are explicit-only; Bright Data and SerpBase are paid Google SERP providers, so `all` never spends on them. Exa remains eligible through its zero-config MCP path, OpenAI can use Pi auth, and other API-backed search providers participate when their API key, local endpoint, or gateway makes them available. Browser-cookie access alone does not opt Gemini into `all`; select Gemini explicitly or configure its API/gateway.
|
|
470
|
+
Set `provider: "all"` on `web_search` or `source_check`, or configure `"provider": "all"` as the default, to run the same query against every eligible search provider simultaneously. DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are always excluded because they are explicit-only; Bright Data and SerpBase are paid Google SERP providers, so `all` never spends on them. Exa remains eligible through its zero-config MCP path, OpenAI can use Pi auth, and other API-backed search providers participate when their API key, local endpoint, or gateway makes them available. Browser-cookie access alone does not opt Gemini into `all`; select Gemini explicitly or configure its API/gateway.
|
|
466
471
|
|
|
467
|
-
Successful provider answers are preserved separately while source URLs and inline content are deduplicated, and one provider failure does not discard the other results. If every participating provider fails, the tool returns per-provider diagnostics. In the Curator, **All** can also be selected like the other provider buttons. Each participating provider gets its own result card, including a provider badge and independent selection checkbox; failed providers get their own disabled error card. The final summary is generated from the selected provider cards and is what Pi receives. Outside the Curator, the same provider answers remain available as labeled sections in one tool response.
|
|
472
|
+
Successful provider answers are preserved separately while source URLs and inline content are deduplicated, and one provider failure does not discard the other results. If every participating provider fails, the tool returns per-provider diagnostics. Configured Firecrawl participates in `all` like other eligible providers. In the Curator, **All** can also be selected like the other provider buttons. Each participating provider gets its own result card, including a provider badge and independent selection checkbox; failed providers get their own disabled error card. The final summary is generated from the selected provider cards and is what Pi receives. Outside the Curator, the same provider answers remain available as labeled sections in one tool response.
|
|
468
473
|
|
|
469
474
|
### Jina Search
|
|
470
475
|
|
|
@@ -477,7 +482,7 @@ Successful provider answers are preserved separately while source URLs and inlin
|
|
|
477
482
|
}
|
|
478
483
|
```
|
|
479
484
|
|
|
480
|
-
Setting `provider` is optional. In `auto` mode, Jina is tried after
|
|
485
|
+
Setting `provider` is optional. In `auto` mode, Jina is tried after Firecrawl and before SERPdive. It can also be selected per request with `provider: "jina"`, included in provider arrays or `provider: "all"`, or placed in `searchRouting.providers`.
|
|
481
486
|
|
|
482
487
|
Jina Search maps `numResults` to its bounded `count` parameter, sends included domains as `site` filters, and adds excluded domains and recency constraints to the search query. Without `includeContent`, it requests SERP metadata only. With `includeContent: true`, Jina visits matching pages and returns their Markdown inline, so requests can take longer and consume more Jina tokens. The fixed hosted endpoint is `https://s.jina.ai`; no custom endpoint is configured by this extension.
|
|
483
488
|
|
|
@@ -819,6 +824,7 @@ Rate limits: Perplexity is capped at 10 requests/minute (client-side). Jina Sear
|
|
|
819
824
|
| `searchinfinity.ts` | Byteplus Searchinfinity search provider |
|
|
820
825
|
| `querit.ts` | Querit Search and Contents API provider |
|
|
821
826
|
| `tavily.ts` | Tavily Search API provider |
|
|
827
|
+
| `firecrawl.ts` | Firecrawl search provider and extraction fallback |
|
|
822
828
|
| `jina-search.ts` | Jina Search API provider |
|
|
823
829
|
| `serpdive.ts` | SERPdive Search API provider |
|
|
824
830
|
| `kagi.ts` | Kagi Search API provider and Extract API fallback |
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
2
|
+
import { getWebSearchConfigPath } from "./utils.ts";
|
|
3
|
+
|
|
4
|
+
const WEB_SEARCH_CONFIG_PATH = getWebSearchConfigPath();
|
|
5
|
+
const AUTH_PROFILE_NAME_PATTERN = /^[A-Za-z][A-Za-z0-9_-]{0,63}$/;
|
|
6
|
+
|
|
7
|
+
export type AuthFetchRequest = true | string;
|
|
8
|
+
export type AuthFetchCache = "session" | "off";
|
|
9
|
+
|
|
10
|
+
export interface AuthFetchProfile {
|
|
11
|
+
name: string;
|
|
12
|
+
hosts: string[];
|
|
13
|
+
chromeProfile?: string;
|
|
14
|
+
redirects: "same-origin";
|
|
15
|
+
cache: AuthFetchCache;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
interface AuthFetchConfigRoot {
|
|
19
|
+
authFetch?: unknown;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function resolveAuthFetchProfile(request: AuthFetchRequest): AuthFetchProfile {
|
|
23
|
+
const profiles = loadAuthFetchProfiles();
|
|
24
|
+
if (profiles.length === 0) {
|
|
25
|
+
throw new Error(`auth requires at least one authFetch profile in ${WEB_SEARCH_CONFIG_PATH}`);
|
|
26
|
+
}
|
|
27
|
+
if (request === true) {
|
|
28
|
+
if (profiles.length !== 1) {
|
|
29
|
+
throw new Error("auth: true requires exactly one authFetch profile; use a profile name instead");
|
|
30
|
+
}
|
|
31
|
+
return profiles[0];
|
|
32
|
+
}
|
|
33
|
+
const name = request.trim();
|
|
34
|
+
const profile = profiles.find(candidate => candidate.name === name);
|
|
35
|
+
if (!profile) throw new Error(`Unknown authFetch profile: ${name}`);
|
|
36
|
+
return profile;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function assertAuthFetchUrl(profile: AuthFetchProfile, rawUrl: string): URL {
|
|
40
|
+
let url: URL;
|
|
41
|
+
try {
|
|
42
|
+
url = new URL(rawUrl);
|
|
43
|
+
} catch (err) {
|
|
44
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
45
|
+
throw new Error(`Authenticated fetch requires an HTTPS URL: ${message}`);
|
|
46
|
+
}
|
|
47
|
+
if (url.protocol !== "https:") throw new Error("Authenticated fetch requires an HTTPS URL");
|
|
48
|
+
const hostname = normalizeHostname(url.hostname);
|
|
49
|
+
if (!profile.hosts.some(host => hostMatches(hostname, host))) {
|
|
50
|
+
throw new Error(`URL host ${hostname} is not allowed by authFetch profile ${profile.name}`);
|
|
51
|
+
}
|
|
52
|
+
return url;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function authFetchRedirectGuard(profile: AuthFetchProfile, from: URL, to: URL): void {
|
|
56
|
+
if (profile.redirects === "same-origin" && to.origin !== from.origin) {
|
|
57
|
+
throw new Error(`Authenticated fetch refused cross-origin redirect: ${from.origin} -> ${to.origin}`);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function loadAuthFetchProfiles(): AuthFetchProfile[] {
|
|
62
|
+
if (!existsSync(WEB_SEARCH_CONFIG_PATH)) return [];
|
|
63
|
+
const raw = readFileSync(WEB_SEARCH_CONFIG_PATH, "utf-8");
|
|
64
|
+
let parsed: AuthFetchConfigRoot;
|
|
65
|
+
try {
|
|
66
|
+
const value: unknown = JSON.parse(raw);
|
|
67
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("expected a JSON object");
|
|
68
|
+
parsed = value as AuthFetchConfigRoot;
|
|
69
|
+
} catch (err) {
|
|
70
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
71
|
+
throw new Error(`Failed to parse ${WEB_SEARCH_CONFIG_PATH}: ${message}`);
|
|
72
|
+
}
|
|
73
|
+
if (parsed.authFetch === undefined || parsed.authFetch === null) return [];
|
|
74
|
+
if (typeof parsed.authFetch !== "object" || Array.isArray(parsed.authFetch)) {
|
|
75
|
+
throw new Error(`authFetch in ${WEB_SEARCH_CONFIG_PATH} must be an object`);
|
|
76
|
+
}
|
|
77
|
+
return Object.entries(parsed.authFetch as Record<string, unknown>).map(([name, value]) => parseProfile(name, value));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function parseProfile(name: string, value: unknown): AuthFetchProfile {
|
|
81
|
+
if (!AUTH_PROFILE_NAME_PATTERN.test(name)) {
|
|
82
|
+
throw new Error(`authFetch profile name ${JSON.stringify(name)} must start with a letter and contain only letters, numbers, underscores, or hyphens`);
|
|
83
|
+
}
|
|
84
|
+
if (Array.isArray(value)) {
|
|
85
|
+
return { name, hosts: parseHosts(value, `authFetch.${name}`), redirects: "same-origin", cache: "session" };
|
|
86
|
+
}
|
|
87
|
+
if (!value || typeof value !== "object") {
|
|
88
|
+
throw new Error(`authFetch.${name} in ${WEB_SEARCH_CONFIG_PATH} must be an array of hosts or an object`);
|
|
89
|
+
}
|
|
90
|
+
const config = value as Record<string, unknown>;
|
|
91
|
+
if (!Array.isArray(config.hosts)) {
|
|
92
|
+
throw new Error(`authFetch.${name}.hosts in ${WEB_SEARCH_CONFIG_PATH} must be a non-empty array of hostnames`);
|
|
93
|
+
}
|
|
94
|
+
const redirects = config.redirects ?? "same-origin";
|
|
95
|
+
if (redirects !== "same-origin") {
|
|
96
|
+
throw new Error(`authFetch.${name}.redirects in ${WEB_SEARCH_CONFIG_PATH} must be "same-origin"`);
|
|
97
|
+
}
|
|
98
|
+
const cache = config.cache ?? "session";
|
|
99
|
+
if (cache !== "session" && cache !== "off") {
|
|
100
|
+
throw new Error(`authFetch.${name}.cache in ${WEB_SEARCH_CONFIG_PATH} must be "session" or "off"`);
|
|
101
|
+
}
|
|
102
|
+
const chromeProfile = parseChromeProfile(config.chromeProfile, `authFetch.${name}.chromeProfile`);
|
|
103
|
+
return {
|
|
104
|
+
name,
|
|
105
|
+
hosts: parseHosts(config.hosts, `authFetch.${name}.hosts`),
|
|
106
|
+
...(chromeProfile ? { chromeProfile } : {}),
|
|
107
|
+
redirects,
|
|
108
|
+
cache,
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function parseHosts(value: unknown[], label: string): string[] {
|
|
113
|
+
if (value.length === 0) throw new Error(`${label} in ${WEB_SEARCH_CONFIG_PATH} must be a non-empty array of hostnames`);
|
|
114
|
+
const hosts = value.map((entry) => {
|
|
115
|
+
if (typeof entry !== "string") throw new Error(`${label} in ${WEB_SEARCH_CONFIG_PATH} must contain only hostnames`);
|
|
116
|
+
return parseHost(entry, label);
|
|
117
|
+
});
|
|
118
|
+
return [...new Set(hosts)];
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function parseHost(value: string, label: string): string {
|
|
122
|
+
const host = normalizeHostname(value.trim());
|
|
123
|
+
if (!host || host.startsWith(".") || host.endsWith(".") || /\s|[\\/?:#@*]/.test(host)) {
|
|
124
|
+
throw new Error(`${label} in ${WEB_SEARCH_CONFIG_PATH} contains an invalid hostname: ${JSON.stringify(value)}`);
|
|
125
|
+
}
|
|
126
|
+
if (host.length > 253 || !/^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)*[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/i.test(host)) {
|
|
127
|
+
throw new Error(`${label} in ${WEB_SEARCH_CONFIG_PATH} contains an invalid hostname: ${JSON.stringify(value)}`);
|
|
128
|
+
}
|
|
129
|
+
return host;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
function parseChromeProfile(value: unknown, label: string): string | undefined {
|
|
133
|
+
if (value === undefined || value === null) return undefined;
|
|
134
|
+
if (typeof value !== "string") throw new Error(`${label} in ${WEB_SEARCH_CONFIG_PATH} must be a string`);
|
|
135
|
+
const normalized = value.trim();
|
|
136
|
+
if (!normalized || normalized === "." || normalized === ".." || normalized.includes("/") || normalized.includes("\\")) {
|
|
137
|
+
throw new Error(`${label} in ${WEB_SEARCH_CONFIG_PATH} must be a profile directory name, not a path`);
|
|
138
|
+
}
|
|
139
|
+
return normalized;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function normalizeHostname(hostname: string): string {
|
|
143
|
+
return hostname.toLowerCase().replace(/^\[|\]$/g, "").replace(/\.$/, "");
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
function hostMatches(hostname: string, allowedHost: string): boolean {
|
|
147
|
+
return hostname === allowedHost || hostname.endsWith(`.${allowedHost}`);
|
|
148
|
+
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { execFile } from "node:child_process";
|
|
2
2
|
import { pbkdf2Sync, createDecipheriv } from "node:crypto";
|
|
3
3
|
import { copyFileSync, existsSync, mkdtempSync, readdirSync, realpathSync, rmSync } from "node:fs";
|
|
4
|
-
import { tmpdir, homedir
|
|
4
|
+
import { tmpdir, homedir } from "node:os";
|
|
5
5
|
import { isAbsolute, join, sep } from "node:path";
|
|
6
6
|
import { isBrowserCookieAccessAllowed } from "./gemini-web-config.ts";
|
|
7
7
|
|
|
@@ -18,6 +18,12 @@ interface BrowserConfig {
|
|
|
18
18
|
type SqliteRow = Record<string, unknown>;
|
|
19
19
|
type SqliteFailure = "unavailable" | "query";
|
|
20
20
|
|
|
21
|
+
interface BrowserCookieEntry {
|
|
22
|
+
name: string;
|
|
23
|
+
value: string;
|
|
24
|
+
path: string;
|
|
25
|
+
}
|
|
26
|
+
|
|
21
27
|
const GOOGLE_ORIGINS = [
|
|
22
28
|
"https://gemini.google.com",
|
|
23
29
|
"https://accounts.google.com",
|
|
@@ -33,6 +39,7 @@ const ALL_COOKIE_NAMES = new Set([
|
|
|
33
39
|
const MACOS_BROWSER_CONFIGS: BrowserConfig[] = [
|
|
34
40
|
{ name: "Helium", baseDir: "Library/Application Support/net.imput.helium", keychainService: "Helium Storage Key", keychainAccount: "Helium" },
|
|
35
41
|
{ name: "Chrome", baseDir: "Library/Application Support/Google/Chrome", keychainService: "Chrome Safe Storage", keychainAccount: "Chrome" },
|
|
42
|
+
{ name: "Brave", baseDir: "Library/Application Support/BraveSoftware/Brave-Browser", keychainService: "Brave Safe Storage", keychainAccount: "Brave" },
|
|
36
43
|
{ name: "Arc", baseDir: "Library/Application Support/Arc/User Data", keychainService: "Arc Safe Storage", keychainAccount: "Arc" },
|
|
37
44
|
];
|
|
38
45
|
|
|
@@ -50,16 +57,32 @@ export function getLastGoogleCookieDiagnostic(): string | null {
|
|
|
50
57
|
return lastCookieDiagnostic;
|
|
51
58
|
}
|
|
52
59
|
|
|
60
|
+
export function getLastBrowserCookieDiagnostic(): string | null {
|
|
61
|
+
return lastCookieDiagnostic;
|
|
62
|
+
}
|
|
63
|
+
|
|
53
64
|
export async function getGoogleCookies(
|
|
54
65
|
options?: { profile?: string; requiredCookies?: string[] },
|
|
55
66
|
): Promise<{ cookies: CookieMap; warnings: string[] } | null> {
|
|
67
|
+
return getBrowserCookiesForHosts({
|
|
68
|
+
hosts: GOOGLE_ORIGINS.map((origin) => new URL(origin).hostname),
|
|
69
|
+
profile: options?.profile,
|
|
70
|
+
requiredCookies: options?.requiredCookies,
|
|
71
|
+
cookieNames: ALL_COOKIE_NAMES,
|
|
72
|
+
requiredLabel: "Gemini",
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export async function getBrowserCookiesForHosts(
|
|
77
|
+
options: { hosts: string[]; profile?: string; requiredCookies?: string[]; cookieNames?: Iterable<string>; requiredLabel?: string; requestUrl?: URL },
|
|
78
|
+
): Promise<{ cookies: CookieMap; warnings: string[]; cookieHeader?: string } | null> {
|
|
56
79
|
lastCookieDiagnostic = null;
|
|
57
80
|
if (!isBrowserCookieAccessAllowed()) {
|
|
58
|
-
lastCookieDiagnostic = "Browser cookie access is disabled; enable allowBrowserCookies to use
|
|
81
|
+
lastCookieDiagnostic = "Browser cookie access is disabled; enable allowBrowserCookies to use browser cookies.";
|
|
59
82
|
return null;
|
|
60
83
|
}
|
|
61
84
|
|
|
62
|
-
const currentPlatform = platform
|
|
85
|
+
const currentPlatform = process.platform;
|
|
63
86
|
const configs = currentPlatform === "darwin" ? MACOS_BROWSER_CONFIGS : currentPlatform === "linux" ? LINUX_BROWSER_CONFIGS : [];
|
|
64
87
|
if (configs.length === 0) {
|
|
65
88
|
lastCookieDiagnostic = "Chromium cookie extraction is unsupported on this platform.";
|
|
@@ -67,17 +90,23 @@ export async function getGoogleCookies(
|
|
|
67
90
|
}
|
|
68
91
|
|
|
69
92
|
const warningSet = new Set<string>();
|
|
70
|
-
const rawProfile = typeof options
|
|
71
|
-
const requestedProfile = normalizeProfileName(options
|
|
93
|
+
const rawProfile = typeof options.profile === "string" ? options.profile.trim() : "";
|
|
94
|
+
const requestedProfile = normalizeProfileName(options.profile);
|
|
72
95
|
if (rawProfile && !requestedProfile) {
|
|
73
96
|
lastCookieDiagnostic = "Configured Chromium profile must be a profile directory name, not a path.";
|
|
74
97
|
return null;
|
|
75
98
|
}
|
|
76
|
-
const requiredCookies = normalizeCookieNames(options
|
|
77
|
-
const
|
|
99
|
+
const requiredCookies = normalizeCookieNames(options.requiredCookies);
|
|
100
|
+
const cookieNames = normalizeCookieNames(options.cookieNames ? [...options.cookieNames] : undefined);
|
|
101
|
+
const hosts = normalizeHosts(options.hosts);
|
|
102
|
+
if (hosts.length === 0) {
|
|
103
|
+
lastCookieDiagnostic = "No valid cookie hosts were requested.";
|
|
104
|
+
return null;
|
|
105
|
+
}
|
|
78
106
|
const home = homedir();
|
|
79
107
|
let sawCookieDatabase = false;
|
|
80
108
|
let sawRequiredCookies = false;
|
|
109
|
+
let sawAnyHostCookie = false;
|
|
81
110
|
let sawBackendFailure: SqliteFailure | undefined;
|
|
82
111
|
let sawUnsafeProfilePath = false;
|
|
83
112
|
|
|
@@ -117,25 +146,36 @@ export async function getGoogleCookies(
|
|
|
117
146
|
const metaVersion = await readMetaVersion(tempDb);
|
|
118
147
|
if (metaVersion.failure) sawBackendFailure = metaVersion.failure;
|
|
119
148
|
if (metaVersion.value === null) continue;
|
|
120
|
-
const rowsResult = await queryCookieRows(tempDb, hosts,
|
|
149
|
+
const rowsResult = await queryCookieRows(tempDb, hosts, cookieNames ?? null, Boolean(options.requestUrl));
|
|
121
150
|
if (rowsResult.status === "failure") {
|
|
122
151
|
sawBackendFailure = rowsResult.failure;
|
|
123
152
|
continue;
|
|
124
153
|
}
|
|
125
154
|
|
|
155
|
+
const entries: BrowserCookieEntry[] = [];
|
|
126
156
|
const cookies: CookieMap = {};
|
|
127
157
|
for (const row of rowsResult.rows) {
|
|
128
158
|
const name = typeof row.name === "string" ? row.name : "";
|
|
129
|
-
if (!
|
|
159
|
+
if (!name) continue;
|
|
130
160
|
let value = typeof row.value === "string" && row.value.length > 0 ? row.value : null;
|
|
131
161
|
if (!value && typeof row.encrypted_value_hex === "string" && /^[0-9a-f]*$/i.test(row.encrypted_value_hex)) {
|
|
132
162
|
value = decryptCookieValue(Buffer.from(row.encrypted_value_hex, "hex"), key, metaVersion.value >= 24);
|
|
133
163
|
}
|
|
134
|
-
if (value)
|
|
164
|
+
if (!value) continue;
|
|
165
|
+
const path = typeof row.path === "string" && row.path.startsWith("/") ? row.path : "/";
|
|
166
|
+
if (options.requestUrl && !pathMatches(options.requestUrl.pathname || "/", path)) continue;
|
|
167
|
+
entries.push({ name, value, path });
|
|
168
|
+
if (!cookies[name]) cookies[name] = value;
|
|
135
169
|
}
|
|
136
170
|
|
|
171
|
+
if (entries.length > 0) sawAnyHostCookie = true;
|
|
137
172
|
if (requiredCookies?.length && !requiredCookies.every((name) => Boolean(cookies[name]))) continue;
|
|
138
|
-
|
|
173
|
+
if (entries.length === 0) continue;
|
|
174
|
+
return {
|
|
175
|
+
cookies,
|
|
176
|
+
warnings: [...warningSet],
|
|
177
|
+
...(options.requestUrl ? { cookieHeader: buildCookieHeader(entries) } : {}),
|
|
178
|
+
};
|
|
139
179
|
} finally {
|
|
140
180
|
rmSync(tempDir, { recursive: true, force: true });
|
|
141
181
|
}
|
|
@@ -153,7 +193,11 @@ export async function getGoogleCookies(
|
|
|
153
193
|
? `Chromium profile '${requestedProfile}' does not contain a cookie database.`
|
|
154
194
|
: "No detected Chromium profile contains a cookie database.";
|
|
155
195
|
} else if (requiredCookies?.length && !sawRequiredCookies) {
|
|
156
|
-
lastCookieDiagnostic =
|
|
196
|
+
lastCookieDiagnostic = `No detected Chromium profile contains the required ${options.requiredLabel ?? "browser"} cookies.`;
|
|
197
|
+
} else if (!sawAnyHostCookie) {
|
|
198
|
+
lastCookieDiagnostic = options.requestUrl
|
|
199
|
+
? "No detected Chromium profile contains cookies for the requested URL."
|
|
200
|
+
: "No detected Chromium profile contains cookies for the requested host.";
|
|
157
201
|
} else if (warningSet.size > 0) {
|
|
158
202
|
lastCookieDiagnostic = [...warningSet][0];
|
|
159
203
|
} else {
|
|
@@ -193,6 +237,10 @@ function normalizeCookieNames(names: string[] | undefined): string[] | undefined
|
|
|
193
237
|
return normalized.length > 0 ? [...new Set(normalized)] : undefined;
|
|
194
238
|
}
|
|
195
239
|
|
|
240
|
+
function normalizeHosts(hosts: string[]): string[] {
|
|
241
|
+
return [...new Set(hosts.map(host => host.trim().toLowerCase().replace(/^\[|\]$/g, "").replace(/\.$/, "")).filter(Boolean))];
|
|
242
|
+
}
|
|
243
|
+
|
|
196
244
|
function listBrowserProfiles(home: string, config: BrowserConfig): string[] {
|
|
197
245
|
const basePath = join(home, config.baseDir);
|
|
198
246
|
if (!existsSync(basePath)) return ["Default"];
|
|
@@ -246,7 +294,7 @@ function removePkcs7Padding(buf: Buffer): Buffer {
|
|
|
246
294
|
return !padding || padding > 16 ? buf : buf.subarray(0, buf.length - padding);
|
|
247
295
|
}
|
|
248
296
|
|
|
249
|
-
function readBrowserPassword(config: BrowserConfig, currentPlatform:
|
|
297
|
+
function readBrowserPassword(config: BrowserConfig, currentPlatform: typeof process.platform): Promise<string | null> {
|
|
250
298
|
const cacheKey = `${currentPlatform}:${config.name}`;
|
|
251
299
|
const cached = browserPasswordCache.get(cacheKey);
|
|
252
300
|
if (cached) return cached;
|
|
@@ -386,19 +434,58 @@ async function hasCookieNames(dbPath: string, hosts: string[], names: string[]):
|
|
|
386
434
|
return { present: names.every((name) => present.has(name)) };
|
|
387
435
|
}
|
|
388
436
|
|
|
389
|
-
async function queryCookieRows(dbPath: string, hosts: string[], names: Iterable<string>): Promise<QueryResult> {
|
|
390
|
-
|
|
437
|
+
async function queryCookieRows(dbPath: string, hosts: string[], names: Iterable<string> | null, filterExpired: boolean): Promise<QueryResult> {
|
|
438
|
+
const columns = await readCookieColumns(dbPath);
|
|
439
|
+
if (columns.status === "failure") return columns;
|
|
440
|
+
const pathExpr = columns.columns.has("path") ? "path" : "'/' AS path";
|
|
441
|
+
const expiresExpr = columns.columns.has("expires_utc") ? "expires_utc" : "0 AS expires_utc";
|
|
442
|
+
const expiryFilter = filterExpired && columns.columns.has("expires_utc") ? ` AND (expires_utc = 0 OR expires_utc > ${chromeExpiryNowMicros()})` : "";
|
|
443
|
+
const partitionFilter = filterExpired ? unpartitionedCookieFilter(columns.columns) : "";
|
|
444
|
+
return runSqliteQuery(dbPath, `SELECT name, value, host_key, ${pathExpr}, ${expiresExpr}, hex(encrypted_value) AS encrypted_value_hex FROM cookies WHERE ${buildCookieWhere(hosts, names ?? undefined)}${expiryFilter}${partitionFilter} ORDER BY length(path) DESC, expires_utc ASC`);
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
async function readCookieColumns(dbPath: string): Promise<{ status: "success"; columns: Set<string> } | { status: "failure"; failure: SqliteFailure }> {
|
|
448
|
+
const result = await runSqliteQuery(dbPath, "PRAGMA table_info(cookies)");
|
|
449
|
+
if (result.status === "failure") return result;
|
|
450
|
+
return { status: "success", columns: new Set(result.rows.map(row => typeof row.name === "string" ? row.name : "")) };
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
function chromeExpiryNowMicros(): number {
|
|
454
|
+
return (Date.now() + 11644473600000) * 1000;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function unpartitionedCookieFilter(columns: Set<string>): string {
|
|
458
|
+
const clauses: string[] = [];
|
|
459
|
+
if (columns.has("top_frame_site_key")) clauses.push("(top_frame_site_key IS NULL OR top_frame_site_key = '')");
|
|
460
|
+
if (columns.has("partition_key")) clauses.push("(partition_key IS NULL OR partition_key = '')");
|
|
461
|
+
if (columns.has("is_partitioned")) clauses.push("(is_partitioned IS NULL OR is_partitioned = 0)");
|
|
462
|
+
return clauses.length > 0 ? ` AND ${clauses.join(" AND ")}` : "";
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
function buildCookieHeader(entries: BrowserCookieEntry[]): string {
|
|
466
|
+
return entries
|
|
467
|
+
.sort((a, b) => b.path.length - a.path.length)
|
|
468
|
+
.map(({ name, value }) => `${name}=${value}`)
|
|
469
|
+
.join("; ");
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
function pathMatches(requestPath: string, cookiePath: string): boolean {
|
|
473
|
+
if (requestPath === cookiePath) return true;
|
|
474
|
+
if (!requestPath.startsWith(cookiePath)) return false;
|
|
475
|
+
if (cookiePath.endsWith("/")) return true;
|
|
476
|
+
return requestPath[cookiePath.length] === "/";
|
|
391
477
|
}
|
|
392
478
|
|
|
393
479
|
function buildCookieWhere(hosts: string[], cookieNames?: Iterable<string>): string {
|
|
394
480
|
const hostClauses: string[] = [];
|
|
395
481
|
for (const host of hosts) {
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
482
|
+
const escapedHost = escapeSqlString(host);
|
|
483
|
+
hostClauses.push(`host_key = '${escapedHost}'`);
|
|
484
|
+
for (const candidate of domainCookieHosts(host)) {
|
|
485
|
+
hostClauses.push(`host_key = '.${escapeSqlString(candidate)}'`);
|
|
399
486
|
}
|
|
400
487
|
}
|
|
401
|
-
let where = `(${hostClauses.join(" OR ")})`;
|
|
488
|
+
let where = `(${[...new Set(hostClauses)].join(" OR ")})`;
|
|
402
489
|
const names = cookieNames ? [...cookieNames].filter(Boolean) : [];
|
|
403
490
|
if (names.length) where += ` AND name IN (${names.map((name) => `'${escapeSqlString(name)}'`).join(", ")})`;
|
|
404
491
|
return where;
|
|
@@ -408,11 +495,11 @@ function escapeSqlString(value: string): string {
|
|
|
408
495
|
return value.replaceAll("'", "''");
|
|
409
496
|
}
|
|
410
497
|
|
|
411
|
-
function
|
|
498
|
+
function domainCookieHosts(host: string): string[] {
|
|
412
499
|
const parts = host.split(".").filter(Boolean);
|
|
413
|
-
if (parts.length <= 1) return [
|
|
414
|
-
const candidates = new Set(
|
|
415
|
-
for (let i =
|
|
500
|
+
if (parts.length <= 1) return [];
|
|
501
|
+
const candidates = new Set<string>();
|
|
502
|
+
for (let i = 0; i <= parts.length - 2; i++) candidates.add(parts.slice(i).join("."));
|
|
416
503
|
return [...candidates];
|
|
417
504
|
}
|
|
418
505
|
|
|
@@ -8,7 +8,7 @@ function safeInlineJSON(data: unknown): string {
|
|
|
8
8
|
}
|
|
9
9
|
|
|
10
10
|
function buildProviderButtons(
|
|
11
|
-
available: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
11
|
+
available: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; firecrawl: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
12
12
|
selected: string,
|
|
13
13
|
hasInitialQueries: boolean,
|
|
14
14
|
): string {
|
|
@@ -23,6 +23,7 @@ function buildProviderButtons(
|
|
|
23
23
|
{ value: "searchinfinity", label: "Searchinfinity", available: available.searchinfinity },
|
|
24
24
|
{ value: "querit", label: "Querit", available: available.querit },
|
|
25
25
|
{ value: "tavily", label: "Tavily", available: available.tavily },
|
|
26
|
+
{ value: "firecrawl", label: "Firecrawl", available: available.firecrawl },
|
|
26
27
|
{ value: "jina", label: "Jina", available: available.jina },
|
|
27
28
|
{ value: "serpdive", label: "SERPdive", available: available.serpdive },
|
|
28
29
|
{ value: "kagi", label: "Kagi", available: available.kagi },
|
|
@@ -54,7 +55,7 @@ export function generateCuratorPage(
|
|
|
54
55
|
queries: string[],
|
|
55
56
|
sessionToken: string,
|
|
56
57
|
timeout: number,
|
|
57
|
-
availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
58
|
+
availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; firecrawl: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
58
59
|
defaultProvider: string,
|
|
59
60
|
searchProvider: string,
|
|
60
61
|
summaryModels: Array<{ value: string; label: string }>,
|
|
@@ -1455,7 +1456,7 @@ const SCRIPT = `(function() {
|
|
|
1455
1456
|
var token = DATA.sessionToken;
|
|
1456
1457
|
var timeoutSec = DATA.timeout;
|
|
1457
1458
|
var queries = Array.isArray(DATA.queries) ? DATA.queries : [];
|
|
1458
|
-
var providers = ["all", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "bocha", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "anysearch", "xai", "brightdata", "serpbase"];
|
|
1459
|
+
var providers = ["all", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "firecrawl", "jina", "serpdive", "kagi", "bocha", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "anysearch", "xai", "brightdata", "serpbase"];
|
|
1459
1460
|
var availProviders = DATA.availableProviders && typeof DATA.availableProviders === "object" ? DATA.availableProviders : {};
|
|
1460
1461
|
var workflow = "summary-review";
|
|
1461
1462
|
var initialDefaultProvider = typeof DATA.defaultProvider === "string" ? DATA.defaultProvider : "exa";
|
|
@@ -1667,6 +1668,7 @@ const SCRIPT = `(function() {
|
|
|
1667
1668
|
if (provider === "searchinfinity") return "Searchinfinity";
|
|
1668
1669
|
if (provider === "querit") return "Querit";
|
|
1669
1670
|
if (provider === "tavily") return "Tavily";
|
|
1671
|
+
if (provider === "firecrawl") return "Firecrawl";
|
|
1670
1672
|
if (provider === "jina") return "Jina";
|
|
1671
1673
|
if (provider === "serpdive") return "SERPdive";
|
|
1672
1674
|
if (provider === "kagi") return "Kagi";
|