bestony-pi-preset 0.0.21 → 0.0.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/node_modules/pi-mcp-adapter/CHANGELOG.md +9 -0
- package/node_modules/pi-mcp-adapter/README.md +3 -1
- package/node_modules/pi-mcp-adapter/commands.ts +12 -2
- package/node_modules/pi-mcp-adapter/index.ts +13 -3
- package/node_modules/pi-mcp-adapter/mcp-auth-flow.ts +64 -3
- package/node_modules/pi-mcp-adapter/package.json +1 -1
- package/node_modules/pi-web-access/CHANGELOG.md +9 -0
- package/node_modules/pi-web-access/README.md +12 -10
- package/node_modules/pi-web-access/bocha.ts +221 -0
- package/node_modules/pi-web-access/curator-page.ts +5 -3
- package/node_modules/pi-web-access/curator-server.ts +2 -1
- package/node_modules/pi-web-access/gemini-search.ts +18 -4
- package/node_modules/pi-web-access/index.ts +33 -16
- package/node_modules/pi-web-access/package.json +1 -1
- package/node_modules/pi-web-access/storage.ts +252 -29
- package/package.json +3 -3
|
@@ -7,6 +7,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [2.23.0] - 2026-08-11
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
- Added interactive callback URL pasting to `/mcp-auth` for OAuth flows running on remote or headless machines. Thanks @trevorleibert-mixpanel for PR #330.
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
- Stopped load-time MCP initialization from printing a TUI startup error when Pi action methods are not bound yet. Thanks @21307369 for issue #327.
|
|
17
|
+
- Kept interactive OAuth authorization URLs clickable as a single terminal hyperlink. Thanks @rfccg for PR #329.
|
|
18
|
+
|
|
10
19
|
## [2.22.0] - 2026-08-11
|
|
11
20
|
|
|
12
21
|
### Added
|
|
@@ -265,7 +265,9 @@ The adapter owns only its client socket and closes that connection when the Pi r
|
|
|
265
265
|
|
|
266
266
|
### Remote/headless OAuth
|
|
267
267
|
|
|
268
|
-
If Pi is running on a remote server and
|
|
268
|
+
If Pi is running on a remote server, `/mcp-auth <server>` prints the authorization URL and opens a callback input. Open the URL in your local browser. After approval, the browser may fail to load the localhost callback page because localhost refers to your workstation; copy the full URL from its address bar and paste it into Pi. The input closes automatically instead when the browser can reach Pi's callback directly.
|
|
269
|
+
|
|
270
|
+
The same flow is available through the proxy tool for non-interactive clients. Persistent OAuth still requires an available OS credential store; on headless Linux that usually means an unlocked Secret Service/libsecret keyring. The adapter fails closed instead of falling back to plaintext credentials when the secure store is unavailable.
|
|
269
271
|
|
|
270
272
|
On Linux, if credential access fails because Pi inherited a revoked session keyring, the adapter uses a best-effort recovery path through `keyctl session - node <packaged helper>` so explicit re-authentication can write fresh credentials without killing a long-lived tmux server. This path requires `keyctl` and `node` on `PATH`; missing, locked, or otherwise unavailable credential stores still fail closed.
|
|
271
273
|
|
|
@@ -25,6 +25,10 @@ import { loadOnboardingState, markSetupCompleted as persistSetupCompleted, markS
|
|
|
25
25
|
import { openPath, resolveServerUrl, sanitizeTerminalText } from "./utils.ts";
|
|
26
26
|
import { isAbortError } from "./runtime-owner.ts";
|
|
27
27
|
|
|
28
|
+
function terminalHyperlink(label: string, url: string): string {
|
|
29
|
+
return `\u001B]8;;${sanitizeTerminalText(url)}\u001B\\${sanitizeTerminalText(label)}\u001B]8;;\u001B\\`;
|
|
30
|
+
}
|
|
31
|
+
|
|
28
32
|
export async function showStatus(state: McpExtensionState, ctx: ExtensionContext): Promise<void> {
|
|
29
33
|
if (!ctx.hasUI) return;
|
|
30
34
|
|
|
@@ -273,11 +277,17 @@ export async function authenticateServer(
|
|
|
273
277
|
...(authStorageOptions.baseDir ? { authStorageOptions } : {}),
|
|
274
278
|
onAuthorizationUrl: (authorizationUrl) => {
|
|
275
279
|
ui.notify(
|
|
276
|
-
`Open this URL to authenticate ${serverName}:\n\n${authorizationUrl}\n\n` +
|
|
277
|
-
"After approving,
|
|
280
|
+
`Open this URL to authenticate ${serverName}:\n\n${terminalHyperlink(authorizationUrl, authorizationUrl)}\n\n` +
|
|
281
|
+
"After approving, Pi will complete automatically if the browser can reach its localhost callback. " +
|
|
282
|
+
"On a remote machine, copy the full localhost URL from the browser address bar and paste it into Pi.",
|
|
278
283
|
"info"
|
|
279
284
|
);
|
|
280
285
|
},
|
|
286
|
+
onAuthorizationInput: (_authorizationUrl, inputSignal) => ui.input(
|
|
287
|
+
`Complete ${serverName} OAuth`,
|
|
288
|
+
"Paste the full callback URL, or wait for automatic completion",
|
|
289
|
+
{ signal: inputSignal },
|
|
290
|
+
),
|
|
281
291
|
...(signal ? { signal } : {}),
|
|
282
292
|
...(runtime ? { runtime } : {}),
|
|
283
293
|
});
|
|
@@ -169,13 +169,23 @@ function installMcpAdapter(pi: ExtensionAPI, options: McpAdapterOptions) {
|
|
|
169
169
|
return resolveDirectTools(config, cache, prefix, envDirectToolOverride);
|
|
170
170
|
}
|
|
171
171
|
|
|
172
|
+
function getActiveToolsIfReady(): string[] | undefined {
|
|
173
|
+
try {
|
|
174
|
+
return pi.getActiveTools?.();
|
|
175
|
+
} catch (error) {
|
|
176
|
+
if (error instanceof Error
|
|
177
|
+
&& error.message.includes("Action methods cannot be called during extension loading")) return undefined;
|
|
178
|
+
throw error;
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
|
|
172
182
|
function deactivateTools(toolNames: string[]): string[] {
|
|
173
183
|
if (toolNames.length === 0) return [];
|
|
174
184
|
const unregisterTool = (pi as ExtensionAPI & { unregisterTool?: (name: string) => boolean }).unregisterTool;
|
|
175
185
|
const unregistered = toolNames.filter((toolName) => unregisterTool?.(toolName) === true);
|
|
176
186
|
const fallbackNames = toolNames.filter((toolName) => !unregistered.includes(toolName));
|
|
177
187
|
const remove = new Set(toolNames);
|
|
178
|
-
const activeTools =
|
|
188
|
+
const activeTools = getActiveToolsIfReady();
|
|
179
189
|
if (!activeTools || activeTools.length === 0) {
|
|
180
190
|
for (const toolName of fallbackNames) fallbackDeactivatedTools.add(toolName);
|
|
181
191
|
return unregistered;
|
|
@@ -207,7 +217,7 @@ function installMcpAdapter(pi: ExtensionAPI, options: McpAdapterOptions) {
|
|
|
207
217
|
registerDirectTool(spec);
|
|
208
218
|
registeredDirectTools.set(spec.prefixedName, fingerprint);
|
|
209
219
|
if (fallbackDeactivatedTools.delete(spec.prefixedName)) {
|
|
210
|
-
const activeTools =
|
|
220
|
+
const activeTools = getActiveToolsIfReady();
|
|
211
221
|
if (activeTools && !activeTools.includes(spec.prefixedName)) {
|
|
212
222
|
pi.setActiveTools([...activeTools, spec.prefixedName]);
|
|
213
223
|
}
|
|
@@ -849,7 +859,7 @@ function installMcpAdapter(pi: ExtensionAPI, options: McpAdapterOptions) {
|
|
|
849
859
|
registerProxyTool(description);
|
|
850
860
|
return;
|
|
851
861
|
}
|
|
852
|
-
const activeTools =
|
|
862
|
+
const activeTools = getActiveToolsIfReady();
|
|
853
863
|
if (activeTools && !activeTools.includes("mcp")) {
|
|
854
864
|
pi.setActiveTools([...activeTools, "mcp"]);
|
|
855
865
|
}
|
|
@@ -49,6 +49,10 @@ export interface McpOAuthRuntime {
|
|
|
49
49
|
|
|
50
50
|
export interface AuthenticateOptions {
|
|
51
51
|
onAuthorizationUrl?: (authorizationUrl: string) => void | Promise<void>
|
|
52
|
+
onAuthorizationInput?: (
|
|
53
|
+
authorizationUrl: string,
|
|
54
|
+
signal: AbortSignal,
|
|
55
|
+
) => Promise<string | undefined>
|
|
52
56
|
authStorageOptions?: AuthStorageOptions
|
|
53
57
|
signal?: AbortSignal
|
|
54
58
|
runtime?: McpOAuthRuntime
|
|
@@ -568,6 +572,53 @@ export function parseAuthorizationCodeInput(input: string, expectedState?: strin
|
|
|
568
572
|
return parseAuthorizationRedirectInput(input, expectedState).code
|
|
569
573
|
}
|
|
570
574
|
|
|
575
|
+
type AuthorizationResponse = {
|
|
576
|
+
input: AuthorizationCodeInput
|
|
577
|
+
source: "callback" | "manual"
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
/**
|
|
581
|
+
* Wait for either the localhost callback or a manually pasted redirect URL.
|
|
582
|
+
* The manual input prompt is dismissed as soon as either path finishes.
|
|
583
|
+
*/
|
|
584
|
+
export async function waitForAuthorizationResponse(
|
|
585
|
+
callbackPromise: Promise<AuthorizationCodeInput>,
|
|
586
|
+
authorizationUrl: string,
|
|
587
|
+
expectedState: string,
|
|
588
|
+
onAuthorizationInput?: AuthenticateOptions["onAuthorizationInput"],
|
|
589
|
+
signal?: AbortSignal,
|
|
590
|
+
): Promise<AuthorizationResponse> {
|
|
591
|
+
if (!onAuthorizationInput) {
|
|
592
|
+
return {
|
|
593
|
+
input: await abortable(callbackPromise, signal),
|
|
594
|
+
source: "callback",
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
const inputController = new AbortController()
|
|
599
|
+
try {
|
|
600
|
+
const response = await abortable(Promise.race([
|
|
601
|
+
callbackPromise.then((input) => ({ input, source: "callback" as const })),
|
|
602
|
+
onAuthorizationInput(authorizationUrl, inputController.signal).then((input) => ({
|
|
603
|
+
input,
|
|
604
|
+
source: "manual" as const,
|
|
605
|
+
})),
|
|
606
|
+
]), signal)
|
|
607
|
+
|
|
608
|
+
if (response.source === "callback") return response
|
|
609
|
+
if (!response.input?.trim()) throw new Error("OAuth authentication cancelled")
|
|
610
|
+
if (!getSearchParamsFromInput(response.input.trim())) {
|
|
611
|
+
throw new Error("Paste the full OAuth callback URL, including its code and state parameters")
|
|
612
|
+
}
|
|
613
|
+
return {
|
|
614
|
+
input: parseAuthorizationRedirectInput(response.input, expectedState),
|
|
615
|
+
source: "manual",
|
|
616
|
+
}
|
|
617
|
+
} finally {
|
|
618
|
+
inputController.abort()
|
|
619
|
+
}
|
|
620
|
+
}
|
|
621
|
+
|
|
571
622
|
/**
|
|
572
623
|
* Complete OAuth authentication from manual user input.
|
|
573
624
|
*/
|
|
@@ -728,12 +779,22 @@ export async function authenticate(
|
|
|
728
779
|
console.warn(`MCP Auth: Failed to open browser for ${serverName}; waiting for manual callback`, { error })
|
|
729
780
|
}
|
|
730
781
|
|
|
731
|
-
const
|
|
782
|
+
const authorizationResponse = await waitForAuthorizationResponse(
|
|
783
|
+
callbackPromise,
|
|
784
|
+
authorizationUrl,
|
|
785
|
+
oauthState,
|
|
786
|
+
options.onAuthorizationInput,
|
|
787
|
+
signal,
|
|
788
|
+
)
|
|
789
|
+
if (authorizationResponse.source === "manual") {
|
|
790
|
+
cancelPendingCallback(oauthState)
|
|
791
|
+
}
|
|
732
792
|
|
|
733
|
-
// The callback server accepted only the flow-local reserved state.
|
|
793
|
+
// The callback server accepted only the flow-local reserved state. Manual
|
|
794
|
+
// input is checked against the same state before token exchange.
|
|
734
795
|
throwIfAborted(signal)
|
|
735
796
|
|
|
736
|
-
return await completeAuth(serverName,
|
|
797
|
+
return await completeAuth(serverName, authorizationResponse.input, {
|
|
737
798
|
...options,
|
|
738
799
|
...(signal ? { signal } : {}),
|
|
739
800
|
runtime,
|
|
@@ -4,6 +4,15 @@ All notable changes to this project will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.22.0] - 2026-08-11
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
- Added Bocha web search provider support. Thanks @jingyulong for PR #243.
|
|
11
|
+
- Added `maxInlineContentChars` to configure the direct content and stored-content slice limit, with a 200,000-character maximum. Thanks @be4zad for issue #244.
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
- Hardened the fetched-content cache against symlink traversal and unsafe permissions, and bounded it to 128 entries and 128 MiB with oldest-entry eviction. Thanks `@HerbertGao` for issue #240 and PR #241.
|
|
15
|
+
|
|
7
16
|
## [0.21.0] - 2026-08-10
|
|
8
17
|
|
|
9
18
|
### Added
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Pi Web Access
|
|
6
6
|
|
|
7
|
-
**Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, AnySearch, xAI/Grok, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, or bring your own API keys.**
|
|
7
|
+
**Web search, content extraction, and video understanding for Pi agent. OpenAI/Codex search, zero-config Exa search, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI/Grok, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, optional browser-cookie Gemini Web, or bring your own API keys.**
|
|
8
8
|
|
|
9
9
|
[](https://www.npmjs.com/package/pi-web-access)
|
|
10
10
|
[](https://opensource.org/licenses/MIT)
|
|
@@ -14,11 +14,11 @@
|
|
|
14
14
|
|
|
15
15
|
## Why Pi Web Access
|
|
16
16
|
|
|
17
|
-
**Zero Config** — Works out of the box with Exa MCP (no API key needed). If you're signed into Pi with a Codex subscription, OpenAI web search can reuse that auth. Add API keys for OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, SerpBase, Exa, Perplexity, or Gemini API for more control; configure a self-hosted SearXNG endpoint for private search; or opt into browser-cookie access for Gemini Web.
|
|
17
|
+
**Zero Config** — Works out of the box with Exa MCP (no API key needed). If you're signed into Pi with a Codex subscription, OpenAI web search can reuse that auth. Add API keys for OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, SerpBase, Exa, Perplexity, or Gemini API for more control; configure a self-hosted SearXNG endpoint for private search; or opt into browser-cookie access for Gemini Web.
|
|
18
18
|
|
|
19
19
|
**Video Understanding** — Point it at a YouTube video or local screen recording and ask questions about what's on screen. Full transcripts, visual descriptions, and frame extraction at exact timestamps.
|
|
20
20
|
|
|
21
|
-
**Smart Fallbacks** — Every capability has a fallback chain. Search tries configured SearXNG first for local/private search, then OpenAI when suitable and available, Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, Perplexity, Gemini API, and Gemini Web when browser cookies are enabled. With no SearXNG configured, the existing zero-config order is unchanged. YouTube tries Gemini Web when enabled, then API, then Perplexity. Blocked pages try configured self-hosted Firecrawl first. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
|
|
21
|
+
**Smart Fallbacks** — Every capability has a fallback chain. Search tries configured SearXNG first for local/private search, then OpenAI when suitable and available, Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, and Gemini Web when browser cookies are enabled. With no SearXNG configured, the existing zero-config order is unchanged. YouTube tries Gemini Web when enabled, then API, then Perplexity. Blocked pages try configured self-hosted Firecrawl first. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
|
|
22
22
|
|
|
23
23
|
**GitHub Cloning** — GitHub URLs are cloned locally instead of scraped. The agent gets real file contents and a local path to explore, not rendered HTML.
|
|
24
24
|
|
|
@@ -40,6 +40,7 @@ Works immediately with no API keys — Exa MCP provides zero-config search. If P
|
|
|
40
40
|
"searchinfinityApiKey": "...",
|
|
41
41
|
"queritApiKey": "...",
|
|
42
42
|
"jinaApiKey": "jina_...",
|
|
43
|
+
"bochaApiKey": "sk-...",
|
|
43
44
|
"perplexityApiKey": "pplx-...",
|
|
44
45
|
"geminiApiKey": "AIza..."
|
|
45
46
|
}
|
|
@@ -95,7 +96,7 @@ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on sc
|
|
|
95
96
|
|
|
96
97
|
### web_search
|
|
97
98
|
|
|
98
|
-
Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, AnySearch, xAI, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, or Gemini. Returns a synthesized answer with source citations.
|
|
99
|
+
Search the web via OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, AnySearch, xAI, Bright Data SERP, SerpBase, self-hosted SearXNG, keyless DuckDuckGo, Exa, Perplexity AI, or Gemini. Returns a synthesized answer with source citations.
|
|
99
100
|
|
|
100
101
|
```typescript
|
|
101
102
|
web_search({ query: "rust async programming" })
|
|
@@ -116,7 +117,7 @@ web_search({ queries: ["query 1", "query 2"], workflow: "auto-summary" })
|
|
|
116
117
|
| `numResults` | Results per query (default: 5, max: 20) |
|
|
117
118
|
| `recencyFilter` | `day`, `week`, `month`, or `year` |
|
|
118
119
|
| `domainFilter` | Limit to domains (prefix with `-` to exclude) |
|
|
119
|
-
| `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase simultaneously; otherwise `openai`, `brave`, `parallel`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `jina`, `serpdive`, `kagi`, `ollama`, `anysearch`, `xai`, `brightdata`, `serpbase`, `searxng`, `duckduckgo`, `exa`, `perplexity`, or `gemini` (auto-selects when no provider or routing is configured; DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are explicit-only) |
|
|
120
|
+
| `provider` | Configured provider when omitted or set to `auto`; `all` searches every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase simultaneously; otherwise `openai`, `brave`, `parallel`, `tinyfish`, `search1api`, `searchinfinity`, `querit`, `tavily`, `jina`, `serpdive`, `kagi`, `bocha`, `ollama`, `anysearch`, `xai`, `brightdata`, `serpbase`, `searxng`, `duckduckgo`, `exa`, `perplexity`, or `gemini` (auto-selects when no provider or routing is configured; DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are explicit-only) |
|
|
120
121
|
| `includeContent` | Fetch full page content from sources in background |
|
|
121
122
|
| `workflow` | `none` (skip curator), `summary-review` (open curator and auto-generate a summary draft, default), or `auto-summary` (generate a summary without opening the curator) |
|
|
122
123
|
|
|
@@ -148,7 +149,7 @@ fetch_content({ url: "https://example.com/diagram.png" })
|
|
|
148
149
|
|
|
149
150
|
### get_search_content
|
|
150
151
|
|
|
151
|
-
Retrieve stored content from previous searches or fetches. Fetched URL content is stored in full in a private `web-search-cache` directory under the Pi config directory, not in the session JSONL. This includes `fetch_content` answer mode, which stores the original page content.
|
|
152
|
+
Retrieve stored content from previous searches or fetches. Fetched URL content is stored in full in a private `web-search-cache` directory under the Pi config directory, not in the session JSONL. This includes `fetch_content` answer mode, which stores the original page content. The cache has a one-hour lifetime and fixed limits of 128 entries and 128 MiB; when either limit is reached, the oldest entries are removed first. On macOS and Linux the cache directory and files are kept at permissions `0700` and `0600`, respectively. Use `findText` to locate bounded matching passages without paging through a large page, or use `offset` and `limit` to retrieve slices intentionally.
|
|
152
153
|
|
|
153
154
|
```typescript
|
|
154
155
|
get_search_content({ responseId: "abc123", urlIndex: 0 })
|
|
@@ -158,11 +159,11 @@ get_search_content({ responseId: "abc123", urlIndex: 0, findText: "installation"
|
|
|
158
159
|
get_search_content({ responseId: "abc123", urlIndex: 0, findText: ["timeout", "retry"], findMode: "fuzzy" })
|
|
159
160
|
```
|
|
160
161
|
|
|
161
|
-
`findMode` supports `exact`, `case-insensitive` (default), and `fuzzy`. Finder output is capped at 20,000 characters with match counts and nearby context. `findText` cannot be combined with `offset` or `limit`.
|
|
162
|
+
`findMode` supports `exact`, `case-insensitive` (default), and `fuzzy`. Finder output is capped at 20,000 characters with match counts and nearby context. `findText` cannot be combined with `offset` or `limit`. The default `limit` and maximum permitted `limit` use `maxInlineContentChars`.
|
|
162
163
|
|
|
163
164
|
### source_check
|
|
164
165
|
|
|
165
|
-
Check a claim and return a machine-readable artifact with exact passage citations. Search results are deduplicated and capped at 20 sources; `fetchContent` fetches at most 5 pages, while stored and retrieved content remains subject to the
|
|
166
|
+
Check a claim and return a machine-readable artifact with exact passage citations. Search results are deduplicated and capped at 20 sources; `fetchContent` fetches at most 5 pages, while stored and retrieved content remains subject to the configured `maxInlineContentChars` `offset`/`limit` bounds.
|
|
166
167
|
|
|
167
168
|
```typescript
|
|
168
169
|
source_check({ claim: "The API supports streaming responses" })
|
|
@@ -377,6 +378,7 @@ Config defaults to `~/.pi/web-search.json`, or `web-search.json` under `PI_CODIN
|
|
|
377
378
|
"searchModel": "gemini-3.6-flash",
|
|
378
379
|
"summaryModel": "anthropic/claude-haiku-4-5",
|
|
379
380
|
"summaryGenerationDeadlineMs": 30000,
|
|
381
|
+
"maxInlineContentChars": 30000,
|
|
380
382
|
"workflow": "summary-review",
|
|
381
383
|
"curatorTimeoutSeconds": 20,
|
|
382
384
|
"curatorRemote": {
|
|
@@ -421,7 +423,7 @@ Config defaults to `~/.pi/web-search.json`, or `web-search.json` under `PI_CODIN
|
|
|
421
423
|
}
|
|
422
424
|
```
|
|
423
425
|
|
|
424
|
-
All provider API-key fields (`openaiApiKey`, `braveApiKey`, `parallelApiKey`, `tinyfishApiKey`, `search1apiApiKey`, `searchinfinityApiKey`, `queritApiKey`, `tavilyApiKey`, `jinaApiKey`, `serpdiveApiKey`, `kagiApiKey`, `ollamaApiKey`, `serpbaseApiKey`, `anysearchApiKey`, `xaiApiKey`, `brightdataApiKey`, `firecrawlApiKey`, `exaApiKey`, `perplexityApiKey`, `geminiApiKey`, `datalabApiKey`, and `cloudflareApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
|
|
426
|
+
All provider API-key fields (`openaiApiKey`, `braveApiKey`, `parallelApiKey`, `tinyfishApiKey`, `search1apiApiKey`, `searchinfinityApiKey`, `queritApiKey`, `tavilyApiKey`, `jinaApiKey`, `serpdiveApiKey`, `kagiApiKey`, `bochaApiKey`, `ollamaApiKey`, `serpbaseApiKey`, `anysearchApiKey`, `xaiApiKey`, `brightdataApiKey`, `firecrawlApiKey`, `exaApiKey`, `perplexityApiKey`, `geminiApiKey`, `datalabApiKey`, and `cloudflareApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
|
|
425
427
|
|
|
426
428
|
```json
|
|
427
429
|
{
|
|
@@ -456,7 +458,7 @@ Bright Data Web Unlocker is a paid `fetch_content` fallback after Parallel and b
|
|
|
456
458
|
|
|
457
459
|
**SerpBase.** Set `serpbaseApiKey` or `SERPBASE_API_KEY` and select `provider: "serpbase"` to query SerpBase's Google Search Results API. SerpBase is explicit-only: it is never chosen by `auto` and never participates in `provider: "all"`, because each request can consume paid Google SERP credits. Domain filters are sent as Google `site:` clauses and reapplied locally; recency maps to Google's `tbs` time filter.
|
|
458
460
|
|
|
459
|
-
Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"jina"`, `"serpdive"`, `"kagi"`, `"ollama"`, `"anysearch"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. AnySearch, xAI, Bright Data, and SerpBase are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `chromeProfile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
|
|
461
|
+
Without an explicit `$` or `!` source, `OPENAI_API_KEY`, `BRAVE_API_KEY`, `PARALLEL_API_KEY`, `TINYFISH_API_KEY`, `SEARCH1API_KEY`, `SEARCHINFINITY_API_KEY`, `QUERIT_API_KEY`, `TAVILY_API_KEY`, `JINA_API_KEY`, `SERPDIVE_API_KEY`, `KAGI_API_KEY`, `BOCHA_API_KEY`, `OLLAMA_API_KEY`, `SERPBASE_API_KEY`, `ANYSEARCH_API_KEY`, `XAI_API_KEY`, `BRIGHTDATA_API_KEY`, `FIRECRAWL_API_KEY`, `EXA_API_KEY`, `GEMINI_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, `DATALAB_API_BASE`, `PERPLEXITY_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, and `CLOUDFLARE_API_KEY` env vars retain their existing precedence over literal config file values. `openaiResponsesUrl` can point OpenAI `web_search` and `source_check` at a third-party gateway that supports the OpenAI Responses API and web search tool; it is an explicit endpoint override, not derived from Pi model provider settings, and defaults to `https://api.openai.com/v1/responses`. `openaiSearchModel` pins the model id used for OpenAI `web_search`, bypassing automatic selection (newest terra-tier model); the id is sent verbatim with whichever OpenAI auth resolves, so gateway-only model ids work too. `xaiSearchModel` similarly pins the xAI search model. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. `provider` or `searchProvider` sets the default search provider and is used when a tool call omits `provider` or sends `"auto"`: `"all"`, `"openai"`, `"brave"`, `"parallel"`, `"tinyfish"`, `"search1api"`, `"searchinfinity"`, `"querit"`, `"tavily"`, `"jina"`, `"serpdive"`, `"kagi"`, `"bocha"`, `"ollama"`, `"anysearch"`, `"xai"`, `"brightdata"`, `"serpbase"`, `"searxng"`, `"exa"`, `"perplexity"`, or `"gemini"`. AnySearch, xAI, Bright Data, and SerpBase are never selected by `auto`; choose them explicitly or place them in `searchRouting`. If either single-provider field is configured, it takes precedence over `searchRouting`. Otherwise, `searchRouting` can opt into an ordered `providers` list and an explicit `fallbackOn` list containing `"transient"`, `"quota"`, `"network"`, and/or `"invalid-response"`; only those typed failures continue to the next available candidate. `"all"` is not valid inside `searchRouting.providers`, because that list defines sequential fallback rather than multi-provider aggregation. Named providers remain strict, and exhausted routes return per-provider diagnostics. `provider` can also be a non-empty array of named providers such as `["brave", "exa"]`; those providers run concurrently using the same aggregation path as `"all"`, while `"auto"` and `"all"` are invalid inside arrays. Random, weighted, sticky, and cooldown routing are not enabled. This is also updated automatically when you change the provider in the curator UI. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `workflow` sets the default search workflow: `"summary-review"` (default, opens curator with auto-generated summary draft), `"auto-summary"` (returns a model-generated summary without opening the curator), or `"none"` (raw results, no curator). Overridden per-call via the `workflow` parameter on the configured search tool, or toggled at runtime with `/curator`. `chromeProfile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `searchModel` overrides the Gemini API model used by the configured search tool without changing URL, YouTube, or video extraction defaults. Gemini API grounded search uses `gemini-3.6-flash` by default; set `searchModel` to choose another model. Gemini Web browser-cookie fallback uses its separate `gemini-3.1-pro` default because Gemini Web relies on private header values; explicitly configured unsupported Web models fail instead of silently falling back to 2.5 Flash. `summaryModel` sets the default model used for generating summary drafts in the curator UI and `auto-summary` mode (e.g. `"anthropic/claude-haiku-4-5"`, `"openai-codex/gpt-5.3-codex-spark"`, or `"openrouter/nvidia/nemotron-3-super-120b-a12b:free"`). Preferred summary and query-rewrite models also resolve through routed provider registrations such as OpenRouter when the native provider is unavailable. When Pi `enabledModels` is configured, summaries are limited to that allowlist; if no enabled summary model is available, the tool returns a deterministic summary instead of calling an unrelated model. `summaryGenerationDeadlineMs` sets the maximum time for one summary model attempt in the curator UI and `auto-summary` mode. It defaults to `30000`, must be a positive integer, and is capped at `600000`. `maxInlineContentChars` sets the direct `fetch_content` content slice and the default and maximum `get_search_content` slice. It defaults to `30000`, must be a positive integer, and is capped at `200000`; full fetched content remains stored for later retrieval. `curatorTimeoutSeconds` controls the initial curator idle timeout (default `20`, max `600`); users can still adjust the timer in the curator UI. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
|
|
460
462
|
|
|
461
463
|
### All providers
|
|
462
464
|
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
2
|
+
import { activityMonitor } from "./activity.ts";
|
|
3
|
+
import type { SearchOptions, SearchResponse } from "./perplexity.ts";
|
|
4
|
+
import { hasCredentialSource, redactCredential, resolveCredential } from "./credential-source.ts";
|
|
5
|
+
import { getWebSearchConfigPath } from "./utils.ts";
|
|
6
|
+
|
|
7
|
+
const BOCHA_SEARCH_URL = "https://api.bochaai.com/v1/web-search";
|
|
8
|
+
const CONFIG_PATH = getWebSearchConfigPath();
|
|
9
|
+
const SEARCH_TIMEOUT_MS = 60_000;
|
|
10
|
+
|
|
11
|
+
interface WebSearchConfig {
|
|
12
|
+
bochaApiKey?: unknown;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
let cachedConfig: WebSearchConfig | null = null;
|
|
16
|
+
|
|
17
|
+
function loadConfig(): WebSearchConfig {
|
|
18
|
+
if (cachedConfig) return cachedConfig;
|
|
19
|
+
if (!existsSync(CONFIG_PATH)) {
|
|
20
|
+
cachedConfig = {};
|
|
21
|
+
return cachedConfig;
|
|
22
|
+
}
|
|
23
|
+
const raw = readFileSync(CONFIG_PATH, "utf-8");
|
|
24
|
+
let parsed: unknown;
|
|
25
|
+
try {
|
|
26
|
+
parsed = JSON.parse(raw);
|
|
27
|
+
} catch (err) {
|
|
28
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
29
|
+
throw new Error(`Failed to parse ${CONFIG_PATH}: ${message}`);
|
|
30
|
+
}
|
|
31
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
|
|
32
|
+
throw new Error(`Invalid config in ${CONFIG_PATH}: expected a JSON object`);
|
|
33
|
+
}
|
|
34
|
+
cachedConfig = parsed as WebSearchConfig;
|
|
35
|
+
return cachedConfig;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
async function getApiKey(signal?: AbortSignal): Promise<string | null> {
|
|
39
|
+
return resolveCredential({
|
|
40
|
+
provider: "Bocha",
|
|
41
|
+
configuredValue: loadConfig().bochaApiKey,
|
|
42
|
+
environmentValue: process.env.BOCHA_API_KEY,
|
|
43
|
+
signal,
|
|
44
|
+
});
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
async function requireApiKey(signal?: AbortSignal): Promise<string> {
|
|
48
|
+
const apiKey = await getApiKey(signal);
|
|
49
|
+
if (!apiKey) {
|
|
50
|
+
throw new Error(
|
|
51
|
+
"Bocha API key not found. Either:\n" +
|
|
52
|
+
` 1. Create ${CONFIG_PATH} with { "bochaApiKey": "your-key" }\n` +
|
|
53
|
+
" 2. Set BOCHA_API_KEY environment variable\n" +
|
|
54
|
+
"Create a key at https://open.bochaai.com/",
|
|
55
|
+
);
|
|
56
|
+
}
|
|
57
|
+
return apiKey;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
function normalizeCount(value: number | undefined): number {
|
|
61
|
+
if (typeof value !== "number" || !Number.isFinite(value)) return 8;
|
|
62
|
+
return Math.max(1, Math.min(Math.floor(value), 20));
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function mapFreshness(value: SearchOptions["recencyFilter"]): string {
|
|
66
|
+
switch (value) {
|
|
67
|
+
case "day": return "oneDay";
|
|
68
|
+
case "week": return "oneWeek";
|
|
69
|
+
case "month": return "oneMonth";
|
|
70
|
+
case "year": return "oneYear";
|
|
71
|
+
default: return "noLimit";
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
interface DomainFilters {
|
|
76
|
+
include: string[];
|
|
77
|
+
exclude: string[];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function normalizeDomain(value: string): string | null {
|
|
81
|
+
let input = value.trim().toLowerCase();
|
|
82
|
+
if (!input) return null;
|
|
83
|
+
if (input.startsWith("-")) input = input.slice(1).trim();
|
|
84
|
+
if (!input) return null;
|
|
85
|
+
try {
|
|
86
|
+
const parsed = input.includes("://") ? new URL(input) : new URL(`https://${input}`);
|
|
87
|
+
input = parsed.hostname;
|
|
88
|
+
} catch {
|
|
89
|
+
input = input.split("/")[0]?.split(":")[0] ?? "";
|
|
90
|
+
}
|
|
91
|
+
input = input.replace(/^\.+|\.+$/g, "");
|
|
92
|
+
return /^[a-z0-9][a-z0-9.-]*\.[a-z]{2,}$/i.test(input) ? input : null;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
function parseDomainFilter(domainFilter: string[] | undefined): DomainFilters {
|
|
96
|
+
const filters: DomainFilters = { include: [], exclude: [] };
|
|
97
|
+
for (const raw of domainFilter ?? []) {
|
|
98
|
+
const domain = normalizeDomain(raw);
|
|
99
|
+
if (!domain) continue;
|
|
100
|
+
const target = raw.trim().startsWith("-") ? filters.exclude : filters.include;
|
|
101
|
+
if (!target.includes(domain)) target.push(domain);
|
|
102
|
+
}
|
|
103
|
+
return filters;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function passesDomainFilters(url: string, filters: DomainFilters): boolean {
|
|
107
|
+
if (filters.include.length === 0 && filters.exclude.length === 0) return true;
|
|
108
|
+
let hostname: string;
|
|
109
|
+
try {
|
|
110
|
+
hostname = new URL(url).hostname.toLowerCase();
|
|
111
|
+
} catch {
|
|
112
|
+
return false;
|
|
113
|
+
}
|
|
114
|
+
const matches = (domain: string) => hostname === domain || hostname.endsWith(`.${domain}`);
|
|
115
|
+
if (filters.exclude.some(matches)) return false;
|
|
116
|
+
return filters.include.length === 0 || filters.include.some(matches);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
function errorMessage(err: unknown): string {
|
|
120
|
+
return err instanceof Error ? err.message : String(err);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function invalidResponse(message: string): Error {
|
|
124
|
+
return new Error(`Bocha API returned invalid response: ${message}`);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function firstString(...values: unknown[]): string | null {
|
|
128
|
+
for (const value of values) {
|
|
129
|
+
if (typeof value === "string" && value.trim()) return value.trim();
|
|
130
|
+
}
|
|
131
|
+
return null;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function parseSearchResponse(value: unknown): { results: SearchResponse["results"] } {
|
|
135
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) throw invalidResponse("expected an object envelope");
|
|
136
|
+
const envelope = value as Record<string, unknown>;
|
|
137
|
+
if (envelope.code !== undefined && Number(envelope.code) !== 200) {
|
|
138
|
+
throw invalidResponse(`code ${String(envelope.code)}: ${firstString(envelope.msg) ?? "unknown error"}`);
|
|
139
|
+
}
|
|
140
|
+
const data = envelope.data;
|
|
141
|
+
const pages = (typeof data === "object" && data !== null && !Array.isArray(data))
|
|
142
|
+
? (data as Record<string, unknown>).webPages
|
|
143
|
+
: undefined;
|
|
144
|
+
const items = (typeof pages === "object" && pages !== null && !Array.isArray(pages))
|
|
145
|
+
? (pages as Record<string, unknown>).value
|
|
146
|
+
: undefined;
|
|
147
|
+
if (!Array.isArray(items)) throw invalidResponse("missing data.webPages.value array");
|
|
148
|
+
const results: SearchResponse["results"] = [];
|
|
149
|
+
for (const item of items) {
|
|
150
|
+
if (!item || typeof item !== "object" || Array.isArray(item)) continue;
|
|
151
|
+
const entry = item as Record<string, unknown>;
|
|
152
|
+
const url = firstString(entry.url, entry.link, entry.href);
|
|
153
|
+
if (!url) continue;
|
|
154
|
+
const title = firstString(entry.title, entry.name) ?? url;
|
|
155
|
+
const snippet = firstString(entry.summary, entry.snippet, entry.description, entry.content) ?? "";
|
|
156
|
+
results.push({ title, url, snippet });
|
|
157
|
+
}
|
|
158
|
+
return { results };
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function buildAnswer(results: SearchResponse["results"]): string {
|
|
162
|
+
return results.map((result) => result.snippet
|
|
163
|
+
? `${result.snippet}\nSource: ${result.title} (${result.url})`
|
|
164
|
+
: `Source: ${result.title} (${result.url})`).join("\n\n");
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
export function isBochaAvailable(): boolean {
|
|
168
|
+
return hasCredentialSource({ provider: "Bocha", configuredValue: loadConfig().bochaApiKey, environmentValue: process.env.BOCHA_API_KEY });
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
export async function searchWithBocha(query: string, options: SearchOptions = {}): Promise<SearchResponse> {
|
|
172
|
+
const apiKey = await requireApiKey(options.signal);
|
|
173
|
+
const numResults = normalizeCount(options.numResults);
|
|
174
|
+
const filters = parseDomainFilter(options.domainFilter);
|
|
175
|
+
const activityId = activityMonitor.logStart({ type: "api", query });
|
|
176
|
+
let response: Response;
|
|
177
|
+
try {
|
|
178
|
+
response = await fetch(BOCHA_SEARCH_URL, {
|
|
179
|
+
method: "POST",
|
|
180
|
+
headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json", Accept: "application/json" },
|
|
181
|
+
body: JSON.stringify({ query, count: numResults, freshness: mapFreshness(options.recencyFilter), summary: true }),
|
|
182
|
+
signal: options.signal ? AbortSignal.any([AbortSignal.timeout(SEARCH_TIMEOUT_MS), options.signal]) : AbortSignal.timeout(SEARCH_TIMEOUT_MS),
|
|
183
|
+
});
|
|
184
|
+
} catch (err) {
|
|
185
|
+
const message = errorMessage(err);
|
|
186
|
+
const redactedMessage = redactCredential(message, apiKey);
|
|
187
|
+
if (redactedMessage.toLowerCase().includes("abort")) activityMonitor.logComplete(activityId, 0);
|
|
188
|
+
else activityMonitor.logError(activityId, redactedMessage);
|
|
189
|
+
if (redactedMessage === message) throw err;
|
|
190
|
+
const redactedError = new Error(redactedMessage);
|
|
191
|
+
if (err instanceof Error) redactedError.name = err.name;
|
|
192
|
+
throw redactedError;
|
|
193
|
+
}
|
|
194
|
+
if (!response.ok) {
|
|
195
|
+
activityMonitor.logComplete(activityId, response.status);
|
|
196
|
+
const errorText = redactCredential(await response.text(), apiKey);
|
|
197
|
+
throw new Error(`Bocha API error ${response.status}: ${errorText.slice(0, 300)}`);
|
|
198
|
+
}
|
|
199
|
+
let rawData: unknown;
|
|
200
|
+
try {
|
|
201
|
+
rawData = await response.json();
|
|
202
|
+
} catch (err) {
|
|
203
|
+
activityMonitor.logComplete(activityId, response.status);
|
|
204
|
+
throw new Error(`Bocha API returned invalid JSON: ${errorMessage(err)}`);
|
|
205
|
+
}
|
|
206
|
+
let parsed: { results: SearchResponse["results"] };
|
|
207
|
+
try {
|
|
208
|
+
parsed = parseSearchResponse(rawData);
|
|
209
|
+
} catch (err) {
|
|
210
|
+
const message = errorMessage(err);
|
|
211
|
+
const redactedMessage = redactCredential(message, apiKey);
|
|
212
|
+
activityMonitor.logError(activityId, redactedMessage);
|
|
213
|
+
if (redactedMessage === message) throw err;
|
|
214
|
+
const redactedError = new Error(redactedMessage);
|
|
215
|
+
if (err instanceof Error) redactedError.name = err.name;
|
|
216
|
+
throw redactedError;
|
|
217
|
+
}
|
|
218
|
+
activityMonitor.logComplete(activityId, response.status);
|
|
219
|
+
const results = parsed.results.filter((result) => passesDomainFilters(result.url, filters)).slice(0, numResults);
|
|
220
|
+
return { answer: buildAnswer(results), results };
|
|
221
|
+
}
|
|
@@ -8,7 +8,7 @@ function safeInlineJSON(data: unknown): string {
|
|
|
8
8
|
}
|
|
9
9
|
|
|
10
10
|
function buildProviderButtons(
|
|
11
|
-
available: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
11
|
+
available: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
12
12
|
selected: string,
|
|
13
13
|
hasInitialQueries: boolean,
|
|
14
14
|
): string {
|
|
@@ -26,6 +26,7 @@ function buildProviderButtons(
|
|
|
26
26
|
{ value: "jina", label: "Jina", available: available.jina },
|
|
27
27
|
{ value: "serpdive", label: "SERPdive", available: available.serpdive },
|
|
28
28
|
{ value: "kagi", label: "Kagi", available: available.kagi },
|
|
29
|
+
{ value: "bocha", label: "Bocha", available: available.bocha },
|
|
29
30
|
{ value: "ollama", label: "Ollama", available: available.ollama },
|
|
30
31
|
{ value: "searxng", label: "SearXNG", available: available.searxng },
|
|
31
32
|
{ value: "duckduckgo", label: "DuckDuckGo", available: available.duckduckgo },
|
|
@@ -53,7 +54,7 @@ export function generateCuratorPage(
|
|
|
53
54
|
queries: string[],
|
|
54
55
|
sessionToken: string,
|
|
55
56
|
timeout: number,
|
|
56
|
-
availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
57
|
+
availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean },
|
|
57
58
|
defaultProvider: string,
|
|
58
59
|
searchProvider: string,
|
|
59
60
|
summaryModels: Array<{ value: string; label: string }>,
|
|
@@ -1454,7 +1455,7 @@ const SCRIPT = `(function() {
|
|
|
1454
1455
|
var token = DATA.sessionToken;
|
|
1455
1456
|
var timeoutSec = DATA.timeout;
|
|
1456
1457
|
var queries = Array.isArray(DATA.queries) ? DATA.queries : [];
|
|
1457
|
-
var providers = ["all", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "anysearch", "xai", "brightdata", "serpbase"];
|
|
1458
|
+
var providers = ["all", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "bocha", "ollama", "searxng", "duckduckgo", "perplexity", "gemini", "anysearch", "xai", "brightdata", "serpbase"];
|
|
1458
1459
|
var availProviders = DATA.availableProviders && typeof DATA.availableProviders === "object" ? DATA.availableProviders : {};
|
|
1459
1460
|
var workflow = "summary-review";
|
|
1460
1461
|
var initialDefaultProvider = typeof DATA.defaultProvider === "string" ? DATA.defaultProvider : "exa";
|
|
@@ -1669,6 +1670,7 @@ const SCRIPT = `(function() {
|
|
|
1669
1670
|
if (provider === "jina") return "Jina";
|
|
1670
1671
|
if (provider === "serpdive") return "SERPdive";
|
|
1671
1672
|
if (provider === "kagi") return "Kagi";
|
|
1673
|
+
if (provider === "bocha") return "Bocha";
|
|
1672
1674
|
if (provider === "ollama") return "Ollama";
|
|
1673
1675
|
if (provider === "searxng") return "SearXNG";
|
|
1674
1676
|
if (provider === "duckduckgo") return "DuckDuckGo";
|
|
@@ -18,7 +18,7 @@ export interface CuratorServerOptions {
|
|
|
18
18
|
queries: string[];
|
|
19
19
|
sessionToken: string;
|
|
20
20
|
timeout: number;
|
|
21
|
-
availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean };
|
|
21
|
+
availableProviders: { all: boolean; openai: boolean; brave: boolean; parallel: boolean; tinyfish: boolean; search1api: boolean; searchinfinity: boolean; querit: boolean; tavily: boolean; jina: boolean; serpdive: boolean; kagi: boolean; bocha: boolean; ollama: boolean; searxng: boolean; duckduckgo: boolean; perplexity: boolean; exa: boolean; gemini: boolean; anysearch: boolean; xai: boolean; brightdata: boolean; serpbase: boolean };
|
|
22
22
|
defaultProvider: string;
|
|
23
23
|
searchProvider: string;
|
|
24
24
|
summaryModels: Array<{ value: string; label: string }>;
|
|
@@ -283,6 +283,7 @@ export function startCuratorServer(
|
|
|
283
283
|
if (provider === "jina") return availableProviders.jina;
|
|
284
284
|
if (provider === "serpdive") return availableProviders.serpdive;
|
|
285
285
|
if (provider === "kagi") return availableProviders.kagi;
|
|
286
|
+
if (provider === "bocha") return availableProviders.bocha;
|
|
286
287
|
if (provider === "ollama") return availableProviders.ollama;
|
|
287
288
|
if (provider === "searxng") return availableProviders.searxng;
|
|
288
289
|
if (provider === "duckduckgo") return availableProviders.duckduckgo;
|
|
@@ -17,6 +17,7 @@ import { isTavilyAvailable, searchWithTavily } from "./tavily.ts";
|
|
|
17
17
|
import { isJinaSearchAvailable, searchWithJina } from "./jina-search.ts";
|
|
18
18
|
import { isSerpdiveAvailable, searchWithSerpdive } from "./serpdive.ts";
|
|
19
19
|
import { isKagiAvailable, searchWithKagi } from "./kagi.ts";
|
|
20
|
+
import { isBochaAvailable, searchWithBocha } from "./bocha.ts";
|
|
20
21
|
import { isOllamaAvailable, searchWithOllama } from "./ollama.ts";
|
|
21
22
|
import { isSearXNGAvailable, searchWithSearXNG } from "./searxng.ts";
|
|
22
23
|
import { isDuckDuckGoAvailable, searchWithDuckDuckGo } from "./duckduckgo.ts";
|
|
@@ -26,7 +27,7 @@ import { isBrightDataAvailable, searchWithBrightData } from "./brightdata.ts";
|
|
|
26
27
|
import { isSerpBaseAvailable, searchWithSerpBase } from "./serpbase.ts";
|
|
27
28
|
import { getWebSearchConfigPath } from "./utils.ts";
|
|
28
29
|
|
|
29
|
-
export const RESOLVED_SEARCH_PROVIDERS = ["openai", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "searxng", "duckduckgo", "perplexity", "gemini", "exa", "serpdive", "kagi", "ollama", "anysearch", "xai", "brightdata", "serpbase"] as const;
|
|
30
|
+
export const RESOLVED_SEARCH_PROVIDERS = ["openai", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "searxng", "duckduckgo", "perplexity", "gemini", "exa", "serpdive", "kagi", "ollama", "anysearch", "xai", "brightdata", "serpbase", "bocha"] as const;
|
|
30
31
|
export const SEARCH_PROVIDERS = ["auto", "all", ...RESOLVED_SEARCH_PROVIDERS] as const;
|
|
31
32
|
|
|
32
33
|
export type ResolvedSearchProvider = typeof RESOLVED_SEARCH_PROVIDERS[number];
|
|
@@ -90,7 +91,7 @@ const CONFIG_PATH = getWebSearchConfigPath();
|
|
|
90
91
|
const DEFAULT_SEARCH_MODEL = "gemini-3.6-flash";
|
|
91
92
|
// Explicit-only providers (DuckDuckGo, AnySearch, xAI, Bright Data, SerpBase) are deliberately absent:
|
|
92
93
|
// `all` must never fan out to an opt-in or paid provider without the user asking for it.
|
|
93
|
-
const ALL_SEARCH_PROVIDERS: ResolvedSearchProvider[] = ["searxng", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "ollama", "perplexity", "gemini"];
|
|
94
|
+
const ALL_SEARCH_PROVIDERS: ResolvedSearchProvider[] = ["searxng", "openai", "exa", "brave", "parallel", "tinyfish", "search1api", "searchinfinity", "querit", "tavily", "jina", "serpdive", "kagi", "ollama", "perplexity", "gemini", "bocha"];
|
|
94
95
|
const VALID_ROUTING_KINDS = ["transient", "quota", "network", "invalid-response"] as const;
|
|
95
96
|
|
|
96
97
|
type SearchConfig = {
|
|
@@ -306,6 +307,7 @@ async function searchWithResolvedProvider(
|
|
|
306
307
|
if (provider === "jina") return { ...(await searchWithJina(query, options)), provider };
|
|
307
308
|
if (provider === "serpdive") return { ...(await searchWithSerpdive(query, options)), provider };
|
|
308
309
|
if (provider === "kagi") return { ...(await searchWithKagi(query, options)), provider };
|
|
310
|
+
if (provider === "bocha") return { ...(await searchWithBocha(query, options)), provider };
|
|
309
311
|
if (provider === "ollama") return { ...(await searchWithOllama(query, options)), provider };
|
|
310
312
|
if (provider === "anysearch") return { ...(await searchWithAnySearch(query, options)), provider };
|
|
311
313
|
if (provider === "xai") return { ...(await searchWithXai(query, options, options.extensionContext)), provider };
|
|
@@ -341,6 +343,7 @@ async function isResolvedProviderAvailable(provider: ResolvedSearchProvider, opt
|
|
|
341
343
|
if (provider === "jina") return isJinaSearchAvailable();
|
|
342
344
|
if (provider === "serpdive") return isSerpdiveAvailable();
|
|
343
345
|
if (provider === "kagi") return isKagiAvailable();
|
|
346
|
+
if (provider === "bocha") return isBochaAvailable();
|
|
344
347
|
if (provider === "ollama") return isOllamaAvailable();
|
|
345
348
|
if (provider === "anysearch") return isAnySearchAvailable();
|
|
346
349
|
if (provider === "xai") return isXaiSearchAvailable(options.extensionContext);
|
|
@@ -363,6 +366,7 @@ function providerLabel(provider: ResolvedSearchProvider): string {
|
|
|
363
366
|
if (provider === "searxng") return "SearXNG";
|
|
364
367
|
if (provider === "duckduckgo") return "DuckDuckGo";
|
|
365
368
|
if (provider === "kagi") return "Kagi";
|
|
369
|
+
if (provider === "bocha") return "Bocha";
|
|
366
370
|
if (provider === "ollama") return "Ollama";
|
|
367
371
|
if (provider === "xai") return "xAI";
|
|
368
372
|
if (provider === "brightdata") return "Bright Data";
|
|
@@ -626,6 +630,16 @@ export async function search(query: string, options: FullSearchOptions = {}): Pr
|
|
|
626
630
|
}
|
|
627
631
|
}
|
|
628
632
|
|
|
633
|
+
if (isBochaAvailable()) {
|
|
634
|
+
try {
|
|
635
|
+
const result = await searchWithBocha(query, options);
|
|
636
|
+
return { ...result, provider: "bocha" };
|
|
637
|
+
} catch (err) {
|
|
638
|
+
if (isAbortError(err)) throw err;
|
|
639
|
+
fallbackErrors.push(`Bocha: ${errorMessage(err)}`);
|
|
640
|
+
}
|
|
641
|
+
}
|
|
642
|
+
|
|
629
643
|
if (isOllamaAvailable()) {
|
|
630
644
|
try {
|
|
631
645
|
const result = await searchWithOllama(query, options);
|
|
@@ -661,8 +675,8 @@ export async function search(query: string, options: FullSearchOptions = {}): Pr
|
|
|
661
675
|
throw new Error(
|
|
662
676
|
"No search provider available. Either:\n" +
|
|
663
677
|
" 1. Use /login to sign in with a Codex subscription for OpenAI web search\n" +
|
|
664
|
-
` 2. Set openaiApiKey, braveApiKey, parallelApiKey, tinyfishApiKey, search1apiApiKey, searchinfinityApiKey, queritApiKey, tavilyApiKey, jinaApiKey, serpdiveApiKey, kagiApiKey, ollamaApiKey, searxngBaseUrl, perplexityApiKey, exaApiKey, geminiApiKey, or cloudflareApiKey in ${CONFIG_PATH}\n` +
|
|
665
|
-
" 3. Set OPENAI_API_KEY, BRAVE_API_KEY, PARALLEL_API_KEY, TINYFISH_API_KEY, SEARCH1API_KEY, SEARCHINFINITY_API_KEY, QUERIT_API_KEY, TAVILY_API_KEY, JINA_API_KEY, SERPDIVE_API_KEY, KAGI_API_KEY, OLLAMA_API_KEY, SEARXNG_BASE_URL, EXA_API_KEY, PERPLEXITY_API_KEY, GEMINI_API_KEY, or CLOUDFLARE_API_KEY env vars\n" +
|
|
678
|
+
` 2. Set openaiApiKey, braveApiKey, parallelApiKey, tinyfishApiKey, search1apiApiKey, searchinfinityApiKey, queritApiKey, tavilyApiKey, jinaApiKey, serpdiveApiKey, kagiApiKey, ollamaApiKey, searxngBaseUrl, perplexityApiKey, exaApiKey, geminiApiKey, bochaApiKey, or cloudflareApiKey in ${CONFIG_PATH}\n` +
|
|
679
|
+
" 3. Set OPENAI_API_KEY, BRAVE_API_KEY, PARALLEL_API_KEY, TINYFISH_API_KEY, SEARCH1API_KEY, SEARCHINFINITY_API_KEY, QUERIT_API_KEY, TAVILY_API_KEY, JINA_API_KEY, SERPDIVE_API_KEY, KAGI_API_KEY, BOCHA_API_KEY, OLLAMA_API_KEY, SEARXNG_BASE_URL, EXA_API_KEY, PERPLEXITY_API_KEY, GEMINI_API_KEY, or CLOUDFLARE_API_KEY env vars\n" +
|
|
666
680
|
" 4. Set GOOGLE_GEMINI_BASE_URL with CLOUDFLARE_API_KEY for Cloudflare AI Gateway routing\n" +
|
|
667
681
|
" 5. Sign into gemini.google.com in a supported Chromium-based browser\n" +
|
|
668
682
|
" 6. Explicitly select provider: \"anysearch\" for anonymous AnySearch, \"xai\" for Grok, \"brightdata\" with brightdataSerpZone for paid Bright Data SERP, or \"serpbase\" with serpbaseApiKey for paid Google SERP"
|
|
@@ -53,6 +53,7 @@ import { isTavilyAvailable } from "./tavily.ts";
|
|
|
53
53
|
import { isJinaSearchAvailable } from "./jina-search.ts";
|
|
54
54
|
import { isSerpdiveAvailable } from "./serpdive.ts";
|
|
55
55
|
import { isKagiAvailable } from "./kagi.ts";
|
|
56
|
+
import { isBochaAvailable } from "./bocha.ts";
|
|
56
57
|
import { isOllamaAvailable } from "./ollama.ts";
|
|
57
58
|
import { isSearXNGAvailable } from "./searxng.ts";
|
|
58
59
|
import { isDuckDuckGoAvailable } from "./duckduckgo.ts";
|
|
@@ -124,6 +125,7 @@ interface WebSearchConfig {
|
|
|
124
125
|
curatorRemote?: unknown;
|
|
125
126
|
summaryModel?: string;
|
|
126
127
|
summaryGenerationDeadlineMs?: unknown;
|
|
128
|
+
maxInlineContentChars?: unknown;
|
|
127
129
|
webSearch?: {
|
|
128
130
|
enabled?: boolean;
|
|
129
131
|
};
|
|
@@ -160,6 +162,7 @@ interface ProviderAvailability {
|
|
|
160
162
|
exa: boolean;
|
|
161
163
|
gemini: boolean;
|
|
162
164
|
kagi: boolean;
|
|
165
|
+
bocha: boolean;
|
|
163
166
|
ollama: boolean;
|
|
164
167
|
anysearch: boolean;
|
|
165
168
|
xai: boolean;
|
|
@@ -384,6 +387,7 @@ async function getProviderAvailability(ctx: ExtensionContext): Promise<ProviderA
|
|
|
384
387
|
jina: isJinaSearchAvailable(),
|
|
385
388
|
serpdive: isSerpdiveAvailable(),
|
|
386
389
|
kagi: isKagiAvailable(),
|
|
390
|
+
bocha: isBochaAvailable(),
|
|
387
391
|
ollama: isOllamaAvailable(),
|
|
388
392
|
searxng: isSearXNGAvailable(),
|
|
389
393
|
duckduckgo: isDuckDuckGoAvailable(),
|
|
@@ -440,6 +444,7 @@ function firstAvailableProvider(available: ProviderAvailability, preferOpenAI: b
|
|
|
440
444
|
if (available.jina) return "jina";
|
|
441
445
|
if (available.serpdive) return "serpdive";
|
|
442
446
|
if (available.kagi) return "kagi";
|
|
447
|
+
if (available.bocha) return "bocha";
|
|
443
448
|
if (available.ollama) return "ollama";
|
|
444
449
|
if (available.perplexity) return "perplexity";
|
|
445
450
|
if (available.gemini) return "gemini";
|
|
@@ -500,6 +505,9 @@ function resolveProvider(
|
|
|
500
505
|
if (provider === "kagi" && !available.kagi) {
|
|
501
506
|
return firstAvailableProvider(available, preferOpenAI, "kagi");
|
|
502
507
|
}
|
|
508
|
+
if (provider === "bocha" && !available.bocha) {
|
|
509
|
+
return firstAvailableProvider(available, preferOpenAI, "bocha");
|
|
510
|
+
}
|
|
503
511
|
if (provider === "ollama" && !available.ollama) {
|
|
504
512
|
return firstAvailableProvider(available, preferOpenAI, "ollama");
|
|
505
513
|
}
|
|
@@ -555,15 +563,22 @@ interface PendingCurate {
|
|
|
555
563
|
}
|
|
556
564
|
|
|
557
565
|
|
|
558
|
-
const
|
|
559
|
-
const
|
|
560
|
-
|
|
566
|
+
const DEFAULT_MAX_INLINE_CONTENT_CHARS = 30_000;
|
|
567
|
+
const MAX_INLINE_CONTENT_CHARS = 200_000;
|
|
568
|
+
|
|
569
|
+
function getMaxInlineContentChars(): number {
|
|
570
|
+
const value = loadConfig().maxInlineContentChars;
|
|
571
|
+
if (typeof value !== "number" || !Number.isFinite(value) || !Number.isInteger(value) || value <= 0) {
|
|
572
|
+
return DEFAULT_MAX_INLINE_CONTENT_CHARS;
|
|
573
|
+
}
|
|
574
|
+
return Math.min(value, MAX_INLINE_CONTENT_CHARS);
|
|
575
|
+
}
|
|
561
576
|
|
|
562
577
|
function stripThumbnails(results: ExtractedContent[]): ExtractedContent[] {
|
|
563
578
|
return results.map(({ thumbnail, frames, ...rest }) => rest);
|
|
564
579
|
}
|
|
565
580
|
|
|
566
|
-
function initialContentSlice(content: string): {
|
|
581
|
+
function initialContentSlice(content: string, maxChars: number): {
|
|
567
582
|
text: string;
|
|
568
583
|
endOffset: number;
|
|
569
584
|
totalBytes: number;
|
|
@@ -571,10 +586,10 @@ function initialContentSlice(content: string): {
|
|
|
571
586
|
shownBytes: number;
|
|
572
587
|
shownLines: number;
|
|
573
588
|
} {
|
|
574
|
-
let endOffset = Math.min(content.length,
|
|
589
|
+
let endOffset = Math.min(content.length, maxChars);
|
|
575
590
|
if (endOffset < content.length) {
|
|
576
591
|
const lineBreak = content.lastIndexOf("\n", endOffset);
|
|
577
|
-
if (lineBreak >= Math.floor(
|
|
592
|
+
if (lineBreak >= Math.floor(maxChars * 0.8)) endOffset = lineBreak + 1;
|
|
578
593
|
}
|
|
579
594
|
const text = content.slice(0, endOffset);
|
|
580
595
|
return {
|
|
@@ -1621,7 +1636,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1621
1636
|
name: toolNames.webSearch,
|
|
1622
1637
|
label: "Web Search",
|
|
1623
1638
|
description:
|
|
1624
|
-
`Search the web using OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, SearXNG, DuckDuckGo, Exa, Perplexity, Gemini, AnySearch, xAI, Bright Data, or SerpBase. Pass a provider array to search only those providers simultaneously, or use provider "all" to search every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase. Returns an AI-synthesized answer with source citations. OpenAI search uses a Codex subscription or OpenAI API key; xAI search uses a SuperGrok/X Premium subscription or xAI API key. DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are available only when explicitly selected. For comprehensive research, prefer queries (plural) with 2-4 varied angles over a single query — each query gets its own synthesized answer, so varying phrasing and scope gives much broader coverage. When includeContent is true, full page content is fetched in the background. Searches auto-open the interactive browser curator and stream results live; set workflow to "none" to skip curation or "auto-summary" for a model-generated summary without the browser curator. The configured provider is used when provider is omitted or set to auto; omit provider unless explicitly overriding it. Without a configured provider, auto-selects OpenAI when suitable and available, then Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, Perplexity, Gemini API, or Gemini Web. When SearXNG is configured, it is preferred first for local/private search.`,
|
|
1639
|
+
`Search the web using OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, SearXNG, DuckDuckGo, Exa, Perplexity, Gemini, AnySearch, xAI, Bright Data, or SerpBase. Pass a provider array to search only those providers simultaneously, or use provider "all" to search every eligible provider except DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase. Returns an AI-synthesized answer with source citations. OpenAI search uses a Codex subscription or OpenAI API key; xAI search uses a SuperGrok/X Premium subscription or xAI API key. DuckDuckGo, AnySearch, xAI, Bright Data, and SerpBase are available only when explicitly selected. For comprehensive research, prefer queries (plural) with 2-4 varied angles over a single query — each query gets its own synthesized answer, so varying phrasing and scope gives much broader coverage. When includeContent is true, full page content is fetched in the background. Searches auto-open the interactive browser curator and stream results live; set workflow to "none" to skip curation or "auto-summary" for a model-generated summary without the browser curator. The configured provider is used when provider is omitted or set to auto; omit provider unless explicitly overriding it. Without a configured provider, auto-selects OpenAI when suitable and available, then Exa, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Bocha, Ollama, Perplexity, Gemini API, or Gemini Web. When SearXNG is configured, it is preferred first for local/private search.`,
|
|
1625
1640
|
promptSnippet:
|
|
1626
1641
|
"Use for web research questions. Prefer {queries:[...]} with 2-4 varied angles over a single query for broader coverage. Omit provider unless explicitly overriding the configured default.",
|
|
1627
1642
|
parameters: Type.Object({
|
|
@@ -2402,7 +2417,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2402
2417
|
}
|
|
2403
2418
|
|
|
2404
2419
|
const fullLength = result.content.length;
|
|
2405
|
-
const slice = initialContentSlice(result.content);
|
|
2420
|
+
const slice = initialContentSlice(result.content, getMaxInlineContentChars());
|
|
2406
2421
|
const truncated = slice.endOffset < fullLength;
|
|
2407
2422
|
let output = slice.text;
|
|
2408
2423
|
|
|
@@ -2614,7 +2629,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2614
2629
|
url: Type.Optional(Type.String({ description: "Get content for this URL" })),
|
|
2615
2630
|
urlIndex: Type.Optional(Type.Number({ description: "Get content for URL at index" })),
|
|
2616
2631
|
offset: Type.Optional(Type.Number({ description: "Character offset for fetched URL content slices (default 0). Cannot be combined with findText." })),
|
|
2617
|
-
limit: Type.Optional(Type.Number({ description:
|
|
2632
|
+
limit: Type.Optional(Type.Number({ description: "Maximum characters to return for fetched URL content slices (default and max are set by maxInlineContentChars). Cannot be combined with findText." })),
|
|
2618
2633
|
findText: Type.Optional(Type.Union([
|
|
2619
2634
|
Type.String({ minLength: 1, maxLength: 500 }),
|
|
2620
2635
|
Type.Array(Type.String({ minLength: 1, maxLength: 500 }), { minItems: 1, maxItems: 10 }),
|
|
@@ -2646,13 +2661,14 @@ export default function (pi: ExtensionAPI) {
|
|
|
2646
2661
|
};
|
|
2647
2662
|
}
|
|
2648
2663
|
const serialized = JSON.stringify(artifact, null, 2);
|
|
2664
|
+
const maxInlineContentChars = getMaxInlineContentChars();
|
|
2649
2665
|
const offset = params.offset ?? 0;
|
|
2650
|
-
const limit = params.limit ??
|
|
2666
|
+
const limit = params.limit ?? maxInlineContentChars;
|
|
2651
2667
|
if (!Number.isInteger(offset) || offset < 0) {
|
|
2652
2668
|
return { content: [{ type: "text", text: "offset must be a non-negative integer" }], details: { error: "Invalid offset", offset } };
|
|
2653
2669
|
}
|
|
2654
|
-
if (!Number.isInteger(limit) || limit <= 0 || limit >
|
|
2655
|
-
return { content: [{ type: "text", text: `limit must be an integer from 1 to ${
|
|
2670
|
+
if (!Number.isInteger(limit) || limit <= 0 || limit > maxInlineContentChars) {
|
|
2671
|
+
return { content: [{ type: "text", text: `limit must be an integer from 1 to ${maxInlineContentChars}` }], details: { error: "Invalid limit", limit, maxLimit: maxInlineContentChars } };
|
|
2656
2672
|
}
|
|
2657
2673
|
if (offset > serialized.length) {
|
|
2658
2674
|
return { content: [{ type: "text", text: `offset ${offset} is out of range (0-${serialized.length})` }], details: { error: "Offset out of range", offset, contentLength: serialized.length } };
|
|
@@ -2774,18 +2790,19 @@ export default function (pi: ExtensionAPI) {
|
|
|
2774
2790
|
}
|
|
2775
2791
|
}
|
|
2776
2792
|
|
|
2793
|
+
const maxInlineContentChars = getMaxInlineContentChars();
|
|
2777
2794
|
const offset = params.offset ?? 0;
|
|
2778
|
-
const limit = params.limit ??
|
|
2795
|
+
const limit = params.limit ?? maxInlineContentChars;
|
|
2779
2796
|
if (!Number.isInteger(offset) || offset < 0) {
|
|
2780
2797
|
return {
|
|
2781
2798
|
content: [{ type: "text", text: "offset must be a non-negative integer" }],
|
|
2782
2799
|
details: { error: "Invalid offset", offset },
|
|
2783
2800
|
};
|
|
2784
2801
|
}
|
|
2785
|
-
if (!Number.isInteger(limit) || limit <= 0 || limit >
|
|
2802
|
+
if (!Number.isInteger(limit) || limit <= 0 || limit > maxInlineContentChars) {
|
|
2786
2803
|
return {
|
|
2787
|
-
content: [{ type: "text", text: `limit must be an integer from 1 to ${
|
|
2788
|
-
details: { error: "Invalid limit", limit, maxLimit:
|
|
2804
|
+
content: [{ type: "text", text: `limit must be an integer from 1 to ${maxInlineContentChars}` }],
|
|
2805
|
+
details: { error: "Invalid limit", limit, maxLimit: maxInlineContentChars },
|
|
2789
2806
|
};
|
|
2790
2807
|
}
|
|
2791
2808
|
if (offset > urlData.content.length) {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-web-access",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.22.0",
|
|
4
4
|
"description": "Web search, URL fetching, GitHub repo cloning, PDF extraction, YouTube video understanding, and local video analysis for Pi coding agent. Supports OpenAI, Brave, Parallel, TinyFish, Search1API, Searchinfinity, Querit, Tavily, Jina, SERPdive, Kagi, Ollama, AnySearch, Bright Data SERP, SerpBase, SearXNG, Firecrawl extraction, Exa, Perplexity, and Gemini.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"scripts": {
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { closeSync, constants, fchmodSync, fstatSync, fsyncSync, lstatSync, mkdirSync, openSync, readFileSync, readdirSync, renameSync, type Stats, unlinkSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { randomBytes } from "node:crypto";
|
|
2
3
|
import { join } from "node:path";
|
|
3
4
|
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
4
5
|
import type { ExtractedContent } from "./extract.ts";
|
|
@@ -9,8 +10,25 @@ const CACHE_TTL_MS = 60 * 60 * 1000;
|
|
|
9
10
|
const FETCH_CACHE_DIR = "web-search-cache";
|
|
10
11
|
const FETCH_CACHE_VERSION = 1;
|
|
11
12
|
const CACHE_KEY_PATTERN = /^[A-Za-z0-9_-]+\.json$/;
|
|
13
|
+
const CACHE_TMP_PATTERN = /^[A-Za-z0-9_-]+\.json\.\d+\.\d+(?:\.[a-f0-9]{32})?\.tmp$/;
|
|
12
14
|
const CACHE_ID_PATTERN = /^[A-Za-z0-9_-]+$/;
|
|
13
15
|
const MAX_METADATA_TEXT = 8192;
|
|
16
|
+
const DEFAULT_CACHE_LIMITS = { maxEntries: 128, maxBytes: 128 * 1024 * 1024 };
|
|
17
|
+
const O_DIRECTORY = process.platform === "win32" ? 0 : (constants.O_DIRECTORY ?? 0);
|
|
18
|
+
const O_NOFOLLOW = process.platform === "win32" ? 0 : (constants.O_NOFOLLOW ?? 0);
|
|
19
|
+
|
|
20
|
+
interface FetchCacheLimits {
|
|
21
|
+
maxEntries: number;
|
|
22
|
+
maxBytes: number;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
interface CacheFile {
|
|
26
|
+
name: string;
|
|
27
|
+
size: number;
|
|
28
|
+
mtimeMs: number;
|
|
29
|
+
dev: number;
|
|
30
|
+
ino: number;
|
|
31
|
+
}
|
|
14
32
|
|
|
15
33
|
export interface QueryResultData {
|
|
16
34
|
query: string;
|
|
@@ -120,17 +138,220 @@ function isInlineFetchData(data: StoredSearchData): data is StoredSearchData & {
|
|
|
120
138
|
return data.type === "fetch" && Array.isArray(data.urls) && data.urls.every(isInlineFetchedUrl);
|
|
121
139
|
}
|
|
122
140
|
|
|
123
|
-
function
|
|
141
|
+
function cacheLimits(limits?: Partial<FetchCacheLimits>): FetchCacheLimits {
|
|
142
|
+
const resolved = {
|
|
143
|
+
maxEntries: limits?.maxEntries ?? DEFAULT_CACHE_LIMITS.maxEntries,
|
|
144
|
+
maxBytes: limits?.maxBytes ?? DEFAULT_CACHE_LIMITS.maxBytes,
|
|
145
|
+
};
|
|
146
|
+
if (!Number.isFinite(resolved.maxEntries) || !Number.isInteger(resolved.maxEntries) || resolved.maxEntries <= 0 ||
|
|
147
|
+
!Number.isFinite(resolved.maxBytes) || !Number.isInteger(resolved.maxBytes) || resolved.maxBytes <= 0) {
|
|
148
|
+
throw new Error("Fetched content cache limits must be finite positive integers");
|
|
149
|
+
}
|
|
150
|
+
return resolved;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
function enforceDirectoryMode(fd: number): void {
|
|
154
|
+
try {
|
|
155
|
+
fchmodSync(fd, 0o700);
|
|
156
|
+
} catch (err) {
|
|
157
|
+
if (process.platform !== "win32") throw err;
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function enforceFileMode(fd: number): void {
|
|
162
|
+
try {
|
|
163
|
+
fchmodSync(fd, 0o600);
|
|
164
|
+
} catch (err) {
|
|
165
|
+
if (process.platform !== "win32") throw err;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
function safeFetchCacheDir(create: true): string;
|
|
170
|
+
function safeFetchCacheDir(create: false): string | null;
|
|
171
|
+
function safeFetchCacheDir(create: boolean): string | null {
|
|
124
172
|
const dir = getFetchCacheDir();
|
|
125
|
-
mkdirSync(dir, { recursive: true, mode: 0o700 });
|
|
173
|
+
if (create) mkdirSync(dir, { recursive: true, mode: 0o700 });
|
|
174
|
+
let before: Stats;
|
|
175
|
+
try {
|
|
176
|
+
before = lstatSync(dir);
|
|
177
|
+
} catch (err) {
|
|
178
|
+
if (!create && (err as NodeJS.ErrnoException).code === "ENOENT") return null;
|
|
179
|
+
throw err;
|
|
180
|
+
}
|
|
181
|
+
if (before.isSymbolicLink() || !before.isDirectory()) {
|
|
182
|
+
throw new Error("Fetched content cache path is not a safe directory");
|
|
183
|
+
}
|
|
184
|
+
if (process.platform === "win32") {
|
|
185
|
+
const after = lstatSync(dir);
|
|
186
|
+
if (after.isSymbolicLink() || !after.isDirectory() || after.dev !== before.dev || after.ino !== before.ino) {
|
|
187
|
+
throw new Error("Fetched content cache directory changed while securing it");
|
|
188
|
+
}
|
|
189
|
+
return dir;
|
|
190
|
+
}
|
|
191
|
+
let fd: number | null = null;
|
|
192
|
+
try {
|
|
193
|
+
fd = openSync(dir, constants.O_RDONLY | O_DIRECTORY | O_NOFOLLOW);
|
|
194
|
+
const opened = fstatSync(fd);
|
|
195
|
+
if (!opened.isDirectory() || opened.dev !== before.dev || opened.ino !== before.ino) {
|
|
196
|
+
throw new Error("Fetched content cache directory changed while opening");
|
|
197
|
+
}
|
|
198
|
+
enforceDirectoryMode(fd);
|
|
199
|
+
closeSync(fd);
|
|
200
|
+
fd = null;
|
|
201
|
+
const after = lstatSync(dir);
|
|
202
|
+
if (after.isSymbolicLink() || !after.isDirectory() || after.dev !== before.dev || after.ino !== before.ino) {
|
|
203
|
+
throw new Error("Fetched content cache directory changed while securing it");
|
|
204
|
+
}
|
|
205
|
+
return dir;
|
|
206
|
+
} finally {
|
|
207
|
+
if (fd !== null) try { closeSync(fd); } catch {}
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
function openRegularFile(path: string): { fd: number; info: Stats } {
|
|
212
|
+
const before = lstatSync(path);
|
|
213
|
+
if (before.isSymbolicLink() || !before.isFile()) throw new Error("Fetched content cache entry is not a regular file");
|
|
214
|
+
const fd = openSync(path, constants.O_RDONLY | O_NOFOLLOW);
|
|
215
|
+
try {
|
|
216
|
+
const info = fstatSync(fd);
|
|
217
|
+
if (!info.isFile() || info.dev !== before.dev || info.ino !== before.ino) {
|
|
218
|
+
throw new Error("Fetched content cache entry changed while opening");
|
|
219
|
+
}
|
|
220
|
+
return { fd, info };
|
|
221
|
+
} catch (err) {
|
|
222
|
+
closeSync(fd);
|
|
223
|
+
throw err;
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
type CacheUnlinkResult = "removed" | "missing" | "changed" | "error";
|
|
228
|
+
|
|
229
|
+
function unlinkCacheFile(dir: string, file: CacheFile): CacheUnlinkResult {
|
|
230
|
+
try {
|
|
231
|
+
const root = lstatSync(dir);
|
|
232
|
+
if (root.isSymbolicLink() || !root.isDirectory()) return "changed";
|
|
233
|
+
const path = join(dir, file.name);
|
|
234
|
+
let current: Stats;
|
|
235
|
+
try {
|
|
236
|
+
current = lstatSync(path);
|
|
237
|
+
} catch (err) {
|
|
238
|
+
return (err as NodeJS.ErrnoException).code === "ENOENT" ? "missing" : "error";
|
|
239
|
+
}
|
|
240
|
+
if (current.isSymbolicLink() || !current.isFile() || current.dev !== file.dev || current.ino !== file.ino) return "changed";
|
|
241
|
+
try {
|
|
242
|
+
unlinkSync(path);
|
|
243
|
+
return "removed";
|
|
244
|
+
} catch (err) {
|
|
245
|
+
return (err as NodeJS.ErrnoException).code === "ENOENT" ? "missing" : "error";
|
|
246
|
+
}
|
|
247
|
+
} catch {
|
|
248
|
+
return "error";
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
function pruneFetchCache(now: number, limits: FetchCacheLimits, preferredKey?: string, reservation?: { key: string; bytes: number }): boolean {
|
|
253
|
+
let dir: string | null;
|
|
254
|
+
try {
|
|
255
|
+
dir = safeFetchCacheDir(false);
|
|
256
|
+
} catch {
|
|
257
|
+
return false;
|
|
258
|
+
}
|
|
259
|
+
if (!dir) return true;
|
|
260
|
+
|
|
261
|
+
let entries: string[];
|
|
262
|
+
try {
|
|
263
|
+
entries = readdirSync(dir);
|
|
264
|
+
} catch {
|
|
265
|
+
return false;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
const files: CacheFile[] = [];
|
|
269
|
+
for (const entry of entries) {
|
|
270
|
+
if (!CACHE_KEY_PATTERN.test(entry) && !CACHE_TMP_PATTERN.test(entry)) continue;
|
|
271
|
+
const path = join(dir, entry);
|
|
272
|
+
let opened: ReturnType<typeof openRegularFile>;
|
|
273
|
+
try {
|
|
274
|
+
opened = openRegularFile(path);
|
|
275
|
+
} catch {
|
|
276
|
+
continue;
|
|
277
|
+
}
|
|
278
|
+
try {
|
|
279
|
+
enforceFileMode(opened.fd);
|
|
280
|
+
} catch {
|
|
281
|
+
closeSync(opened.fd);
|
|
282
|
+
continue;
|
|
283
|
+
}
|
|
284
|
+
closeSync(opened.fd);
|
|
285
|
+
const file = { name: entry, size: opened.info.size, mtimeMs: opened.info.mtimeMs, dev: opened.info.dev, ino: opened.info.ino };
|
|
286
|
+
if (now - file.mtimeMs >= CACHE_TTL_MS) {
|
|
287
|
+
const removed = unlinkCacheFile(dir, file);
|
|
288
|
+
if (removed !== "removed" && removed !== "missing") return false;
|
|
289
|
+
continue;
|
|
290
|
+
}
|
|
291
|
+
if (CACHE_KEY_PATTERN.test(entry)) files.push(file);
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
files.sort((a, b) => a.mtimeMs - b.mtimeMs || a.name.localeCompare(b.name));
|
|
295
|
+
const projectedUsage = () => {
|
|
296
|
+
const replaced = reservation ? files.find((file) => file.name === reservation.key) : undefined;
|
|
297
|
+
return {
|
|
298
|
+
entries: files.length + (reservation && !replaced ? 1 : 0),
|
|
299
|
+
bytes: files.reduce((total, file) => total + file.size, 0) + (reservation ? reservation.bytes - (replaced?.size ?? 0) : 0),
|
|
300
|
+
};
|
|
301
|
+
};
|
|
302
|
+
const attempted = new Set<string>();
|
|
303
|
+
let usage = projectedUsage();
|
|
304
|
+
while (usage.entries > limits.maxEntries || usage.bytes > limits.maxBytes) {
|
|
305
|
+
const index = files.findIndex((file) => file.name !== preferredKey && !attempted.has(file.name));
|
|
306
|
+
if (index < 0) break;
|
|
307
|
+
const file = files[index];
|
|
308
|
+
attempted.add(file.name);
|
|
309
|
+
const removed = unlinkCacheFile(dir, file);
|
|
310
|
+
if (removed === "removed" || removed === "missing") files.splice(index, 1);
|
|
311
|
+
usage = projectedUsage();
|
|
312
|
+
}
|
|
313
|
+
return usage.entries <= limits.maxEntries && usage.bytes <= limits.maxBytes;
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
function writeFetchCache(data: StoredSearchData & { urls: ExtractedContent[] }): FetchCacheRef {
|
|
317
|
+
const limits = DEFAULT_CACHE_LIMITS;
|
|
318
|
+
const serialized = JSON.stringify(data);
|
|
319
|
+
const size = Buffer.byteLength(serialized);
|
|
320
|
+
if (size > limits.maxBytes) throw new Error(`Fetched content cache entry exceeds ${limits.maxBytes} bytes`);
|
|
321
|
+
|
|
322
|
+
const dir = safeFetchCacheDir(true);
|
|
126
323
|
const key = cacheKeyForId(data.id);
|
|
324
|
+
if (!pruneFetchCache(Date.now(), limits, key, { key, bytes: size })) {
|
|
325
|
+
throw new Error("Fetched content cache could not reserve space for a new entry");
|
|
326
|
+
}
|
|
127
327
|
const finalPath = join(dir, key);
|
|
128
|
-
const
|
|
328
|
+
const tmpName = `${key}.${process.pid}.${Date.now()}.${randomBytes(16).toString("hex")}.tmp`;
|
|
329
|
+
const tmpPath = join(dir, tmpName);
|
|
330
|
+
let fd: number | null = null;
|
|
331
|
+
let tmpFile: CacheFile | null = null;
|
|
332
|
+
let renamed = false;
|
|
129
333
|
try {
|
|
130
|
-
|
|
334
|
+
fd = openSync(tmpPath, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY | O_NOFOLLOW, 0o600);
|
|
335
|
+
const tmpInfo = fstatSync(fd);
|
|
336
|
+
tmpFile = { name: tmpName, size: tmpInfo.size, mtimeMs: tmpInfo.mtimeMs, dev: tmpInfo.dev, ino: tmpInfo.ino };
|
|
337
|
+
enforceFileMode(fd);
|
|
338
|
+
writeFileSync(fd, serialized, "utf8");
|
|
339
|
+
fsyncSync(fd);
|
|
340
|
+
closeSync(fd);
|
|
341
|
+
fd = null;
|
|
342
|
+
safeFetchCacheDir(false);
|
|
131
343
|
renameSync(tmpPath, finalPath);
|
|
344
|
+
renamed = true;
|
|
345
|
+
const written = lstatSync(finalPath);
|
|
346
|
+
if (!written.isFile() || written.dev !== tmpFile.dev || written.ino !== tmpFile.ino) {
|
|
347
|
+
throw new Error("Fetched content cache entry changed after writing");
|
|
348
|
+
}
|
|
349
|
+
if (!pruneFetchCache(Date.now(), limits, key)) {
|
|
350
|
+
throw new Error("Fetched content cache could not meet its limits after writing");
|
|
351
|
+
}
|
|
132
352
|
} catch (err) {
|
|
133
|
-
try {
|
|
353
|
+
if (fd !== null) try { closeSync(fd); } catch {}
|
|
354
|
+
if (tmpFile) unlinkCacheFile(dir, { ...tmpFile, name: renamed ? key : tmpName });
|
|
134
355
|
throw err;
|
|
135
356
|
}
|
|
136
357
|
return { version: FETCH_CACHE_VERSION, key, storedAt: Date.now() };
|
|
@@ -182,37 +403,32 @@ function readCachedFetchData(data: StoredSearchData, now = Date.now()): StoredSe
|
|
|
182
403
|
return unavailableFetchData(data, data.fetchCacheError ?? "Cached fetched content is unavailable");
|
|
183
404
|
}
|
|
184
405
|
const path = fetchCachePath(data.fetchCache.key);
|
|
185
|
-
if (!path
|
|
186
|
-
|
|
187
|
-
}
|
|
406
|
+
if (!path) return unavailableFetchData(data, "Cached fetched content is missing or expired");
|
|
407
|
+
let fd: number | null = null;
|
|
188
408
|
try {
|
|
189
|
-
|
|
409
|
+
if (!safeFetchCacheDir(false)) return unavailableFetchData(data, "Cached fetched content is missing or expired");
|
|
410
|
+
const opened = openRegularFile(path);
|
|
411
|
+
fd = opened.fd;
|
|
412
|
+
enforceFileMode(fd);
|
|
413
|
+
const parsed: unknown = JSON.parse(readFileSync(fd, "utf8"));
|
|
190
414
|
if (!isValidStoredData(parsed) || parsed.type !== "fetch" || parsed.id !== data.id || !isInlineFetchData(parsed)) {
|
|
191
415
|
return unavailableFetchData(data, "Cached fetched content is invalid");
|
|
192
416
|
}
|
|
193
417
|
return { ...parsed, fetchCache: data.fetchCache, urlMetadata: data.urlMetadata };
|
|
194
418
|
} catch (err) {
|
|
419
|
+
if ((err as NodeJS.ErrnoException).code === "ENOENT") {
|
|
420
|
+
return unavailableFetchData(data, "Cached fetched content is missing or expired");
|
|
421
|
+
}
|
|
195
422
|
const message = err instanceof Error ? err.message : String(err);
|
|
196
423
|
return unavailableFetchData(data, `Cached fetched content could not be read: ${message}`);
|
|
424
|
+
} finally {
|
|
425
|
+
if (fd !== null) try { closeSync(fd); } catch {}
|
|
197
426
|
}
|
|
198
427
|
}
|
|
199
428
|
|
|
200
|
-
export function pruneExpiredFetchCache(now = Date.now()): void {
|
|
201
|
-
const
|
|
202
|
-
|
|
203
|
-
let entries: string[];
|
|
204
|
-
try {
|
|
205
|
-
entries = readdirSync(dir);
|
|
206
|
-
} catch {
|
|
207
|
-
return;
|
|
208
|
-
}
|
|
209
|
-
for (const entry of entries) {
|
|
210
|
-
if (!CACHE_KEY_PATTERN.test(entry)) continue;
|
|
211
|
-
const path = join(dir, entry);
|
|
212
|
-
try {
|
|
213
|
-
if (now - statSync(path).mtimeMs >= CACHE_TTL_MS) unlinkSync(path);
|
|
214
|
-
} catch {}
|
|
215
|
-
}
|
|
429
|
+
export function pruneExpiredFetchCache(now = Date.now(), requestedLimits?: Partial<FetchCacheLimits>): void {
|
|
430
|
+
const limits = cacheLimits(requestedLimits);
|
|
431
|
+
try { pruneFetchCache(now, limits); } catch {}
|
|
216
432
|
}
|
|
217
433
|
|
|
218
434
|
export function storeResult(id: string, data: StoredSearchData): void {
|
|
@@ -220,7 +436,6 @@ export function storeResult(id: string, data: StoredSearchData): void {
|
|
|
220
436
|
}
|
|
221
437
|
|
|
222
438
|
export function storeFetchedContentResult(id: string, data: StoredSearchData & { type: "fetch"; urls: ExtractedContent[] }): StoredSearchData {
|
|
223
|
-
pruneExpiredFetchCache();
|
|
224
439
|
let ref: FetchCacheRef | null = null;
|
|
225
440
|
let cacheError: string | undefined;
|
|
226
441
|
try {
|
|
@@ -247,8 +462,16 @@ export function getAllResults(): StoredSearchData[] {
|
|
|
247
462
|
export function deleteResult(id: string): boolean {
|
|
248
463
|
const data = storedResults.get(id);
|
|
249
464
|
if (data?.fetchCache) {
|
|
250
|
-
|
|
251
|
-
|
|
465
|
+
try {
|
|
466
|
+
const dir = safeFetchCacheDir(false);
|
|
467
|
+
const path = fetchCachePath(data.fetchCache.key);
|
|
468
|
+
if (dir && path) {
|
|
469
|
+
const info = lstatSync(path);
|
|
470
|
+
if (!info.isSymbolicLink() && info.isFile()) {
|
|
471
|
+
unlinkCacheFile(dir, { name: data.fetchCache.key, size: info.size, mtimeMs: info.mtimeMs, dev: info.dev, ino: info.ino });
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
} catch {}
|
|
252
475
|
}
|
|
253
476
|
return storedResults.delete(id);
|
|
254
477
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "bestony-pi-preset",
|
|
3
|
-
"version": "0.0.
|
|
3
|
+
"version": "0.0.22",
|
|
4
4
|
"description": "Bestony's personal Pi coding agent preset — skills, extensions, prompts, and themes.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "bestony",
|
|
@@ -40,9 +40,9 @@
|
|
|
40
40
|
"@tintinweb/pi-subagents": "^0.15.0",
|
|
41
41
|
"pi-cache-optimizer": "^2.8.2",
|
|
42
42
|
"pi-init": "^1.0.0",
|
|
43
|
-
"pi-mcp-adapter": "^2.
|
|
43
|
+
"pi-mcp-adapter": "^2.23.0",
|
|
44
44
|
"pi-session-name": "^0.1.2",
|
|
45
|
-
"pi-web-access": "^0.
|
|
45
|
+
"pi-web-access": "^0.22.0",
|
|
46
46
|
"pi-xai": "^0.17.1"
|
|
47
47
|
},
|
|
48
48
|
"bundledDependencies": [
|