@asterxsk/kiln 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (263) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +170 -0
  3. package/agent/AGENTS.md +67 -0
  4. package/agent/README.md +5 -0
  5. package/agent/extensions/AGENTS.md +68 -0
  6. package/agent/extensions/ask-user/index.ts +418 -0
  7. package/agent/extensions/ask-user/install.ps1 +23 -0
  8. package/agent/extensions/ask-user/install.sh +21 -0
  9. package/agent/extensions/ask-user/package-lock.json +769 -0
  10. package/agent/extensions/ask-user/package.json +19 -0
  11. package/agent/extensions/ask-user/prompt.ts +45 -0
  12. package/agent/extensions/ask-user/tsconfig.json +7 -0
  13. package/agent/extensions/background-terminals/docs/implementation-guide.md +942 -0
  14. package/agent/extensions/background-terminals/index.ts +627 -0
  15. package/agent/extensions/background-terminals/install.ps1 +23 -0
  16. package/agent/extensions/background-terminals/install.sh +21 -0
  17. package/agent/extensions/background-terminals/manager.test.ts +735 -0
  18. package/agent/extensions/background-terminals/output.test.ts +109 -0
  19. package/agent/extensions/background-terminals/package-lock.json +769 -0
  20. package/agent/extensions/background-terminals/package.json +17 -0
  21. package/agent/extensions/background-terminals/prompt.test.ts +125 -0
  22. package/agent/extensions/background-terminals/ps.test.ts +82 -0
  23. package/agent/extensions/background-terminals/result-delivery.test.ts +44 -0
  24. package/agent/extensions/background-terminals/src/domain.ts +87 -0
  25. package/agent/extensions/background-terminals/src/manager.ts +907 -0
  26. package/agent/extensions/background-terminals/src/output.ts +84 -0
  27. package/agent/extensions/background-terminals/src/prompt.ts +142 -0
  28. package/agent/extensions/background-terminals/src/result-delivery.ts +27 -0
  29. package/agent/extensions/background-terminals/src/runtime.ts +36 -0
  30. package/agent/extensions/background-terminals/src/ui/output-view.ts +79 -0
  31. package/agent/extensions/background-terminals/src/ui/ps.ts +621 -0
  32. package/agent/extensions/background-terminals/tsconfig.json +7 -0
  33. package/agent/extensions/destructive/README.md +31 -0
  34. package/agent/extensions/destructive/index.ts +88 -0
  35. package/agent/extensions/destructive/install.ps1 +23 -0
  36. package/agent/extensions/destructive/install.sh +21 -0
  37. package/agent/extensions/file-search/index.spec.ts +443 -0
  38. package/agent/extensions/file-search/index.ts +459 -0
  39. package/agent/extensions/file-search/install.ps1 +23 -0
  40. package/agent/extensions/file-search/install.sh +21 -0
  41. package/agent/extensions/file-search/package-lock.json +2253 -0
  42. package/agent/extensions/file-search/package.json +23 -0
  43. package/agent/extensions/file-search/src/args.ts +122 -0
  44. package/agent/extensions/file-search/src/binaries.ts +422 -0
  45. package/agent/extensions/file-search/src/output.ts +126 -0
  46. package/agent/extensions/file-search/src/process.ts +146 -0
  47. package/agent/extensions/file-search/src/prompt.ts +52 -0
  48. package/agent/extensions/file-search/tsconfig.json +7 -0
  49. package/agent/extensions/goal/README.md +50 -0
  50. package/agent/extensions/goal/index.ts +155 -0
  51. package/agent/extensions/goal/install.ps1 +23 -0
  52. package/agent/extensions/goal/install.sh +21 -0
  53. package/agent/extensions/modelconf/PLAN.md +915 -0
  54. package/agent/extensions/modelconf/README.md +66 -0
  55. package/agent/extensions/modelconf/index.ts +296 -0
  56. package/agent/extensions/modelconf/install.ps1 +23 -0
  57. package/agent/extensions/modelconf/install.sh +21 -0
  58. package/agent/extensions/modelconf/package-lock.json +1809 -0
  59. package/agent/extensions/modelconf/package.json +13 -0
  60. package/agent/extensions/modelconf/src/fuzzy.ts +26 -0
  61. package/agent/extensions/modelconf/src/glob.ts +13 -0
  62. package/agent/extensions/modelconf/src/persistence.test.ts +46 -0
  63. package/agent/extensions/modelconf/src/persistence.ts +164 -0
  64. package/agent/extensions/modelconf/src/ui/ModelConfView.test.ts +75 -0
  65. package/agent/extensions/modelconf/src/ui/ModelConfView.ts +1101 -0
  66. package/agent/extensions/modelconf/tsconfig.json +15 -0
  67. package/agent/extensions/pi-web-access/CHANGELOG.md +690 -0
  68. package/agent/extensions/pi-web-access/LICENSE +21 -0
  69. package/agent/extensions/pi-web-access/README.md +470 -0
  70. package/agent/extensions/pi-web-access/SECURITY.md +5 -0
  71. package/agent/extensions/pi-web-access/activity.ts +101 -0
  72. package/agent/extensions/pi-web-access/auth-fetch.ts +148 -0
  73. package/agent/extensions/pi-web-access/banner.png +0 -0
  74. package/agent/extensions/pi-web-access/brightdata-unlocker.ts +272 -0
  75. package/agent/extensions/pi-web-access/chrome-cookies.ts +669 -0
  76. package/agent/extensions/pi-web-access/content-find.ts +139 -0
  77. package/agent/extensions/pi-web-access/credential-source.ts +191 -0
  78. package/agent/extensions/pi-web-access/data-uri-sanitize.ts +406 -0
  79. package/agent/extensions/pi-web-access/datalab-pdf-extract.ts +568 -0
  80. package/agent/extensions/pi-web-access/declared-web-links.ts +173 -0
  81. package/agent/extensions/pi-web-access/evidence/CONTRACT-EVIDENCE.md +496 -0
  82. package/agent/extensions/pi-web-access/evidence/contract-probe.mjs +140 -0
  83. package/agent/extensions/pi-web-access/exa.ts +526 -0
  84. package/agent/extensions/pi-web-access/extract.ts +1196 -0
  85. package/agent/extensions/pi-web-access/feature-config.ts +29 -0
  86. package/agent/extensions/pi-web-access/fetch-params.ts +111 -0
  87. package/agent/extensions/pi-web-access/gemini-adc.ts +298 -0
  88. package/agent/extensions/pi-web-access/gemini-api.ts +353 -0
  89. package/agent/extensions/pi-web-access/gemini-pdf-extract.ts +108 -0
  90. package/agent/extensions/pi-web-access/gemini-search.ts +21 -0
  91. package/agent/extensions/pi-web-access/gemini-url-context.ts +128 -0
  92. package/agent/extensions/pi-web-access/gemini-web-config.ts +101 -0
  93. package/agent/extensions/pi-web-access/gemini-web.ts +487 -0
  94. package/agent/extensions/pi-web-access/github-api.ts +197 -0
  95. package/agent/extensions/pi-web-access/github-extract.ts +746 -0
  96. package/agent/extensions/pi-web-access/github-issue-pr.ts +700 -0
  97. package/agent/extensions/pi-web-access/index.ts +1737 -0
  98. package/agent/extensions/pi-web-access/package-lock.json +5808 -0
  99. package/agent/extensions/pi-web-access/package.json +64 -0
  100. package/agent/extensions/pi-web-access/page-query.ts +96 -0
  101. package/agent/extensions/pi-web-access/pdf-extract.ts +409 -0
  102. package/agent/extensions/pi-web-access/pi-web-fetch-demo.mp4 +0 -0
  103. package/agent/extensions/pi-web-access/promise-try.d.ts +7 -0
  104. package/agent/extensions/pi-web-access/query-rewrite.ts +51 -0
  105. package/agent/extensions/pi-web-access/render-search-error.ts +170 -0
  106. package/agent/extensions/pi-web-access/rsc-extract.ts +338 -0
  107. package/agent/extensions/pi-web-access/search-types.ts +20 -0
  108. package/agent/extensions/pi-web-access/source-check.ts +282 -0
  109. package/agent/extensions/pi-web-access/ssrf-protection.ts +526 -0
  110. package/agent/extensions/pi-web-access/storage.ts +521 -0
  111. package/agent/extensions/pi-web-access/summary-model-scope.ts +125 -0
  112. package/agent/extensions/pi-web-access/test/auth-fetch.test.mjs +208 -0
  113. package/agent/extensions/pi-web-access/test/brightdata-unlocker.test.mjs +840 -0
  114. package/agent/extensions/pi-web-access/test/chrome-cookie-extraction.test.mjs +441 -0
  115. package/agent/extensions/pi-web-access/test/config-path.test.mjs +283 -0
  116. package/agent/extensions/pi-web-access/test/content-find.test.mjs +25 -0
  117. package/agent/extensions/pi-web-access/test/credential-source.test.mjs +118 -0
  118. package/agent/extensions/pi-web-access/test/data-uri-sanitize.test.mjs +210 -0
  119. package/agent/extensions/pi-web-access/test/datalab-pdf-extract.test.mjs +552 -0
  120. package/agent/extensions/pi-web-access/test/declared-web-links.test.mjs +212 -0
  121. package/agent/extensions/pi-web-access/test/fetch-answer-storage.test.mjs +40 -0
  122. package/agent/extensions/pi-web-access/test/fetch-cache-storage.test.mjs +334 -0
  123. package/agent/extensions/pi-web-access/test/fetch-content-domain-policy.test.mjs +95 -0
  124. package/agent/extensions/pi-web-access/test/fetch-modes.test.mjs +53 -0
  125. package/agent/extensions/pi-web-access/test/fetch-not-found-guidance.test.mjs +92 -0
  126. package/agent/extensions/pi-web-access/test/fetch-params.test.mjs +86 -0
  127. package/agent/extensions/pi-web-access/test/fetch-render-call.test.mjs +34 -0
  128. package/agent/extensions/pi-web-access/test/fetch-routing.test.mjs +173 -0
  129. package/agent/extensions/pi-web-access/test/gemini-adc-auth.test.mjs +257 -0
  130. package/agent/extensions/pi-web-access/test/gemini-api-transport.test.mjs +170 -0
  131. package/agent/extensions/pi-web-access/test/gemini-pdf-extract.test.mjs +133 -0
  132. package/agent/extensions/pi-web-access/test/gemini-web-cookie-opt-in.test.mjs +178 -0
  133. package/agent/extensions/pi-web-access/test/gemini-web-header-overflow.test.mjs +148 -0
  134. package/agent/extensions/pi-web-access/test/get-search-content.test.mjs +223 -0
  135. package/agent/extensions/pi-web-access/test/github-extract.test.mjs +378 -0
  136. package/agent/extensions/pi-web-access/test/github-issue-pr.test.mjs +565 -0
  137. package/agent/extensions/pi-web-access/test/inline-content-config.test.mjs +99 -0
  138. package/agent/extensions/pi-web-access/test/lazy-extract-load.test.mjs +118 -0
  139. package/agent/extensions/pi-web-access/test/local-video-oversize.test.mjs +52 -0
  140. package/agent/extensions/pi-web-access/test/package-typebox-dependency.test.mjs +50 -0
  141. package/agent/extensions/pi-web-access/test/page-query.test.mjs +51 -0
  142. package/agent/extensions/pi-web-access/test/pdf-config.test.mjs +140 -0
  143. package/agent/extensions/pi-web-access/test/pdf-extract.test.mjs +500 -0
  144. package/agent/extensions/pi-web-access/test/proxy-transport.test.mjs +286 -0
  145. package/agent/extensions/pi-web-access/test/query-rewrite.test.mjs +52 -0
  146. package/agent/extensions/pi-web-access/test/rsc-fallback.test.mjs +102 -0
  147. package/agent/extensions/pi-web-access/test/search-error-render.test.mjs +152 -0
  148. package/agent/extensions/pi-web-access/test/search-providers.test.mjs +274 -0
  149. package/agent/extensions/pi-web-access/test/source-check.test.mjs +179 -0
  150. package/agent/extensions/pi-web-access/test/ssrf-allow-ranges-config.test.mjs +205 -0
  151. package/agent/extensions/pi-web-access/test/ssrf-protection.test.mjs +456 -0
  152. package/agent/extensions/pi-web-access/test/summary-model-scope.test.mjs +106 -0
  153. package/agent/extensions/pi-web-access/test/tool-registration-config.test.mjs +182 -0
  154. package/agent/extensions/pi-web-access/test/web-search-answer-render.test.mjs +66 -0
  155. package/agent/extensions/pi-web-access/test/youtube-extract-errors.test.mjs +64 -0
  156. package/agent/extensions/pi-web-access/tsconfig.json +11 -0
  157. package/agent/extensions/pi-web-access/utils.ts +451 -0
  158. package/agent/extensions/pi-web-access/video-extract.ts +392 -0
  159. package/agent/extensions/pi-web-access/youtube-extract.ts +328 -0
  160. package/agent/extensions/shared/activity-status.ts +31 -0
  161. package/agent/extensions/shared/child-session.test.ts +270 -0
  162. package/agent/extensions/shared/child-session.ts +148 -0
  163. package/agent/extensions/shared/context-utilization.test.ts +48 -0
  164. package/agent/extensions/shared/context-utilization.ts +47 -0
  165. package/agent/extensions/shared/dashboard-state.ts +99 -0
  166. package/agent/extensions/shared/install.ps1 +23 -0
  167. package/agent/extensions/shared/install.sh +21 -0
  168. package/agent/extensions/shared/tool-call-timeout.test.ts +117 -0
  169. package/agent/extensions/shared/tool-call-timeout.ts +104 -0
  170. package/agent/extensions/skillsconf/README.md +74 -0
  171. package/agent/extensions/skillsconf/index.ts +92 -0
  172. package/agent/extensions/skillsconf/install.ps1 +23 -0
  173. package/agent/extensions/skillsconf/install.sh +21 -0
  174. package/agent/extensions/skillsconf/package-lock.json +1809 -0
  175. package/agent/extensions/skillsconf/package.json +18 -0
  176. package/agent/extensions/skillsconf/src/delete-skill.test.ts +64 -0
  177. package/agent/extensions/skillsconf/src/delete-skill.ts +45 -0
  178. package/agent/extensions/skillsconf/src/filter.test.ts +80 -0
  179. package/agent/extensions/skillsconf/src/filter.ts +74 -0
  180. package/agent/extensions/skillsconf/src/fuzzy.ts +26 -0
  181. package/agent/extensions/skillsconf/src/persistence.test.ts +72 -0
  182. package/agent/extensions/skillsconf/src/persistence.ts +169 -0
  183. package/agent/extensions/skillsconf/src/ui/SkillConfView.test.ts +337 -0
  184. package/agent/extensions/skillsconf/src/ui/SkillConfView.ts +758 -0
  185. package/agent/extensions/skillsconf/src/ui/text-input.ts +91 -0
  186. package/agent/extensions/skillsconf/src/ui/tui-helpers.ts +144 -0
  187. package/agent/extensions/skillsconf/tsconfig.json +15 -0
  188. package/agent/extensions/status line/index.ts +282 -0
  189. package/agent/extensions/status line/install.ps1 +23 -0
  190. package/agent/extensions/status line/install.sh +21 -0
  191. package/agent/extensions/subagents/by-the-way.test.ts +29 -0
  192. package/agent/extensions/subagents/claude.test.ts +119 -0
  193. package/agent/extensions/subagents/codex.test.ts +102 -0
  194. package/agent/extensions/subagents/context-usage.test.ts +107 -0
  195. package/agent/extensions/subagents/docs/design-plan.md +568 -0
  196. package/agent/extensions/subagents/docs/effect-v4-extension-guide.md +354 -0
  197. package/agent/extensions/subagents/docs/effect-v4-notes.md +571 -0
  198. package/agent/extensions/subagents/index.ts +779 -0
  199. package/agent/extensions/subagents/install.ps1 +23 -0
  200. package/agent/extensions/subagents/install.sh +21 -0
  201. package/agent/extensions/subagents/manager.test.ts +276 -0
  202. package/agent/extensions/subagents/package-lock.json +2244 -0
  203. package/agent/extensions/subagents/package.json +19 -0
  204. package/agent/extensions/subagents/result-delivery.test.ts +27 -0
  205. package/agent/extensions/subagents/src/backend.ts +73 -0
  206. package/agent/extensions/subagents/src/backends/claude.ts +701 -0
  207. package/agent/extensions/subagents/src/backends/codex.ts +1060 -0
  208. package/agent/extensions/subagents/src/backends/pi.ts +575 -0
  209. package/agent/extensions/subagents/src/backends/stub.ts +300 -0
  210. package/agent/extensions/subagents/src/by-the-way.ts +21 -0
  211. package/agent/extensions/subagents/src/domain.ts +253 -0
  212. package/agent/extensions/subagents/src/format.ts +74 -0
  213. package/agent/extensions/subagents/src/manager.ts +736 -0
  214. package/agent/extensions/subagents/src/prompt.ts +92 -0
  215. package/agent/extensions/subagents/src/result-delivery.ts +20 -0
  216. package/agent/extensions/subagents/src/runtime.ts +53 -0
  217. package/agent/extensions/subagents/src/ui/takeover.ts +583 -0
  218. package/agent/extensions/subagents/src/ui/transcript.ts +201 -0
  219. package/agent/extensions/subagents/takeover.test.ts +29 -0
  220. package/agent/extensions/subagents/tsconfig.json +7 -0
  221. package/agent/extensions/taste/index.ts +443 -0
  222. package/agent/extensions/taste/install.ps1 +23 -0
  223. package/agent/extensions/taste/install.sh +21 -0
  224. package/agent/extensions/todo/AGENTS.md +38 -0
  225. package/agent/extensions/todo/LICENSE +21 -0
  226. package/agent/extensions/todo/config.ts +55 -0
  227. package/agent/extensions/todo/index.ts +151 -0
  228. package/agent/extensions/todo/install.ps1 +23 -0
  229. package/agent/extensions/todo/install.sh +21 -0
  230. package/agent/extensions/todo/locales/de.json +17 -0
  231. package/agent/extensions/todo/locales/en.json +15 -0
  232. package/agent/extensions/todo/locales/es.json +17 -0
  233. package/agent/extensions/todo/locales/fr.json +17 -0
  234. package/agent/extensions/todo/locales/pt-BR.json +17 -0
  235. package/agent/extensions/todo/locales/pt.json +17 -0
  236. package/agent/extensions/todo/locales/ru.json +17 -0
  237. package/agent/extensions/todo/locales/uk.json +17 -0
  238. package/agent/extensions/todo/locales/zh.json +17 -0
  239. package/agent/extensions/todo/package-lock.json +3358 -0
  240. package/agent/extensions/todo/package.json +67 -0
  241. package/agent/extensions/todo/state/i18n-bridge.ts +64 -0
  242. package/agent/extensions/todo/state/invariants.ts +20 -0
  243. package/agent/extensions/todo/state/replay.ts +38 -0
  244. package/agent/extensions/todo/state/selectors.ts +107 -0
  245. package/agent/extensions/todo/state/state-reducer.ts +326 -0
  246. package/agent/extensions/todo/state/state.ts +18 -0
  247. package/agent/extensions/todo/state/store.ts +82 -0
  248. package/agent/extensions/todo/state/task-graph.ts +57 -0
  249. package/agent/extensions/todo/todo-overlay.ts +200 -0
  250. package/agent/extensions/todo/todo.ts +155 -0
  251. package/agent/extensions/todo/tool/response-envelope.ts +109 -0
  252. package/agent/extensions/todo/tool/types.ts +206 -0
  253. package/agent/extensions/todo/verify-ref-system.js +0 -0
  254. package/agent/extensions/todo/view/format.ts +177 -0
  255. package/agent/extensions/trim-context/README.md +54 -0
  256. package/agent/extensions/trim-context/index.ts +487 -0
  257. package/agent/extensions/trim-context/install.ps1 +23 -0
  258. package/agent/extensions/trim-context/install.sh +21 -0
  259. package/agent/install.ps1 +527 -0
  260. package/agent/install.sh +511 -0
  261. package/agent/keybindings.json +7 -0
  262. package/bin/kiln.js +359 -0
  263. package/package.json +21 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Nico Bailon
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,470 @@
1
+ <p>
2
+ <img src="banner.png" alt="pi-web-access" width="1100">
3
+ </p>
4
+
5
+ # Pi Web Access
6
+
7
+ **Web search via Exa, content extraction, and video understanding for Pi agent. Zero-config Exa search with no API key needed, or bring your own Exa API key for direct API access.**
8
+
9
+ [![npm version](https://img.shields.io/npm/v/pi-web-access?style=for-the-badge)](https://www.npmjs.com/package/pi-web-access)
10
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg?style=for-the-badge)](https://opensource.org/licenses/MIT)
11
+ [![Platform](https://img.shields.io/badge/Platform-macOS%20%7C%20Linux%20%7C%20Windows*-blue?style=for-the-badge)]()
12
+
13
+ <https://github.com/user-attachments/assets/cac6a17a-1eeb-4dde-9818-cdf85d8ea98f>
14
+
15
+ ## Why Pi Web Access
16
+
17
+ **Zero Config** — Works out of the box with Exa MCP (no API key needed). Add an Exa API key for direct API access with higher limits.
18
+
19
+ **Video Understanding** — Point it at a YouTube video or local screen recording and ask questions about what's on screen. Full transcripts, visual descriptions, and frame extraction at exact timestamps.
20
+
21
+ **Smart Fallbacks** — Every capability has a fallback chain. Search uses Exa (direct API if keyed, MCP if not). YouTube tries Gemini Web when enabled, then API, then Exa. Blocked pages fall back to Jina Reader, Bright Data Web Unlocker, or Gemini extraction. Third-party hosted page fetchers require explicit `fetchRouting.allowRemoteHostedProviders` opt-in for remote HTTP(S) targets.
22
+
23
+ **GitHub Cloning** — GitHub URLs are cloned locally instead of scraped. The agent gets real file contents and a local path to explore, not rendered HTML.
24
+
25
+ ## Install
26
+
27
+ ```bash
28
+ pi install npm:pi-web-access
29
+ ```
30
+
31
+ Works immediately with no API keys — Exa MCP provides zero-config search. For direct API access, add your key to `~/.pi/web-search.json`:
32
+
33
+ ```json
34
+ {
35
+ "exaApiKey": "exa-..."
36
+ }
37
+ ```
38
+
39
+ `web_search` uses Exa (direct API if keyed, MCP if not). For sandboxed networks that provide outbound proxy transport through environment variables, set `ssrf.trustEnvProxy` to `true` to skip local DNS preflight for proxied hostnames:
40
+
41
+ ```json
42
+ {
43
+ "ssrf": {
44
+ "trustEnvProxy": true
45
+ }
46
+ }
47
+ ```
48
+
49
+ This is an opt-in DNS-preflight adjustment, not proxy transport configuration. `HTTP_PROXY`, `HTTPS_PROXY`, and `ALL_PROXY` are recognized; `NO_PROXY` hosts still undergo DNS validation, and localhost or literal private IP targets remain blocked.
50
+
51
+ Optional dependencies for video frame extraction:
52
+
53
+ ```bash
54
+ brew install ffmpeg # frame extraction, video thumbnails, local video duration
55
+ brew install yt-dlp # YouTube stream URLs for frame extraction
56
+ ```
57
+
58
+ Without these, video content analysis (transcripts, visual descriptions via Gemini) still works. The binaries are only needed for extracting individual frames as images.
59
+
60
+ Requires Pi v0.37.3+.
61
+
62
+ ## Quick Start
63
+
64
+ ```typescript
65
+ // Search the web
66
+ web_search({ query: "TypeScript best practices 2025" })
67
+
68
+ // Fetch a page
69
+ fetch_content({ url: "https://docs.example.com/guide" })
70
+
71
+ // Clone a GitHub repo
72
+ fetch_content({ url: "https://github.com/owner/repo" })
73
+
74
+ // Understand a YouTube video
75
+ fetch_content({ url: "https://youtube.com/watch?v=abc", prompt: "What libraries are shown?" })
76
+
77
+ // Analyze a screen recording
78
+ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on screen?" })
79
+ ```
80
+
81
+ ## Tools
82
+
83
+ ### web_search
84
+
85
+ Search the web via Exa. Returns a synthesized answer with source citations.
86
+
87
+ ```typescript
88
+ web_search({ query: "rust async programming" })
89
+ web_search({ queries: ["query 1", "query 2"] })
90
+ web_search({ query: "latest news", numResults: 10, recencyFilter: "week" })
91
+ web_search({ query: "...", domainFilter: ["github.com"] })
92
+ web_search({ query: "...", includeContent: true })
93
+ ```
94
+
95
+ | Parameter | Description |
96
+ | ----------- | ------------- |
97
+ | `query` / `queries` | Single query or batch of queries |
98
+ | `numResults` | Results per query (default: 5, max: 20) |
99
+ | `recencyFilter` | `day`, `week`, `month`, or `year` |
100
+ | `domainFilter` | Limit to domains (prefix with `-` to exclude) |
101
+ | `includeContent` | Fetch full page content from sources in background |
102
+
103
+ ### fetch_content
104
+
105
+ Fetch URL(s) as readable markdown, exact textual HTTP bodies, direct images, or page-grounded answers. Automatically detects and handles GitHub repos, GitHub PRs and issues, YouTube videos, PDFs, local video files, images, and regular web pages.
106
+
107
+ ```typescript
108
+ fetch_content({ url: "https://example.com/article" })
109
+ fetch_content({ urls: ["url1", "url2", "url3"] })
110
+ fetch_content({ url: "https://github.com/owner/repo" })
111
+ fetch_content({ url: "https://github.com/owner/repo/pull/123#discussion_r456" })
112
+ fetch_content({ url: "https://youtube.com/watch?v=abc", prompt: "What libraries are shown?" })
113
+ fetch_content({ url: "/path/to/recording.mp4", prompt: "What error appears on screen?" })
114
+ fetch_content({ url: "https://youtube.com/watch?v=abc", timestamp: "23:41-25:00", frames: 4 })
115
+ fetch_content({ url: "https://example.com/api", mode: "raw" })
116
+ fetch_content({ url: "https://example.com/guide", mode: "answer", prompt: "What are the installation steps?" })
117
+ fetch_content({ url: "https://example.com/account", auth: "work", mode: "raw" })
118
+ fetch_content({ url: "https://example.com/diagram.png" })
119
+ ```
120
+
121
+ | Parameter | Description |
122
+ | ----------- | ------------- |
123
+ | `url` / `urls` | Single URL/path or multiple URLs |
124
+ | `prompt` | Question for video analysis, or the page-local question required by `mode: "answer"` |
125
+ | `mode` | `readable` (default), `raw` for exact textual HTTP bodies, or `answer` for a grounded answer from fetched content |
126
+ | `answerModel` | Optional `provider/model-id` override for answer mode; defaults to the current enabled Pi model |
127
+ | `timestamp` | Extract frame(s) — single (`"23:41"`), range (`"23:41-25:00"`), or seconds (`"85"`) |
128
+ | `frames` | Number of frames to extract (max 12) |
129
+ | `forceClone` | Clone GitHub repos that exceed the 350MB size threshold |
130
+
131
+ ### get_search_content
132
+
133
+ Retrieve stored content from previous searches or fetches. Fetched URL content is stored in full in a private `web-search-cache` directory inside the extension folder (not the Pi config dir), rather than in the session JSONL. This includes `fetch_content` answer mode, which stores the original page content. The cache has a one-hour lifetime and fixed limits of 128 entries and 128 MiB; when either limit is reached, the oldest entries are removed first. On macOS and Linux the cache directory and files are kept at permissions `0700` and `0600`, respectively. Use `findText` to locate bounded matching passages without paging through a large page, or use `offset` and `limit` to retrieve slices intentionally.
134
+
135
+ ```typescript
136
+ get_search_content({ responseId: "abc123", urlIndex: 0 })
137
+ get_search_content({ responseId: "abc123", url: "https://...", offset: 30000 })
138
+ get_search_content({ responseId: "abc123", query: "original query" })
139
+ get_search_content({ responseId: "abc123", urlIndex: 0, findText: "installation" })
140
+ get_search_content({ responseId: "abc123", urlIndex: 0, findText: ["timeout", "retry"], findMode: "fuzzy" })
141
+ ```
142
+
143
+ `findMode` supports `exact`, `case-insensitive` (default), and `fuzzy`. Finder output is capped at 20,000 characters with match counts and nearby context. `findText` cannot be combined with `offset` or `limit`. The default `limit` and maximum permitted `limit` use `maxInlineContentChars`.
144
+
145
+ ### source_check
146
+
147
+ Check a claim and return a machine-readable artifact with exact passage citations. Search results are deduplicated and capped at 20 sources; `fetchContent` fetches at most 5 pages, while stored and retrieved content remains subject to the configured `maxInlineContentChars` `offset`/`limit` bounds.
148
+
149
+ ```typescript
150
+ source_check({ claim: "The API supports streaming responses" })
151
+ source_check({
152
+ claim: "The API supports streaming responses",
153
+ queries: ["API streaming responses documentation", "API streaming limitations"],
154
+ fetchContent: true,
155
+ domainFilter: ["docs.example.com", "-old.example.com"]
156
+ })
157
+ ```
158
+
159
+ The artifact includes `supported`, `contradicted`, `unclear`, or `missing-evidence` claim status, source quality hints, SHA-256 content hashes, and passage IDs with exact source offsets. Search and fetch errors remain in the artifact instead of being silently discarded. Artifacts are stored with the session and retrieved through `get_search_content` using the returned `responseId`; paged artifact responses are JSON slices, so request the next `offset` when needed.
160
+
161
+ ## Capabilities
162
+
163
+ ### GitHub repos
164
+
165
+ GitHub URLs are cloned locally instead of scraped. The agent gets real file contents and a local path to explore with `read` and `bash`. Root URLs return the repo tree + README, `/tree/` paths return directory listings, `/blob/` paths return file contents.
166
+
167
+ Repos over 350MB get a lightweight API-based view instead of a full clone (override with `forceClone: true`). Commit SHA URLs are handled via the API. Clones are cached for the session and wiped on session change. Private repos require the `gh` CLI. Set `githubClone.enabled` to `false` to skip this GitHub-specific clone/API handling; `fetch_content` remains available, so the URL can continue through the normal HTTP extraction path.
168
+
169
+ Pull request and issue URLs are rendered as one priority-ordered markdown document instead of scraped HTML. PR views include status, body, checks when `gh` supports them, review verdicts, linked references, files, commits, conversation comments, review thread comments, truncation markers, and escalation commands. Issue views include state, metadata, body, linked closing PRs when available, and comments. Comment anchors such as `#issuecomment-...` and `#discussion_r...` are forced inline. The full rendered document is stored for `get_search_content` offsets and `findText`.
170
+
171
+ `fetch_content` uses `gh pr view` or `gh issue view` first with prompts disabled. It retries with a smaller field set when an older `gh` does not know a requested field. Public unauthenticated REST is used as a bounded fallback under the same `fetchContent.domainPolicy` and SSRF rules; REST cannot include checks. Set `githubPrIssue.enabled` to `false` to skip PR/issue specialization and keep normal HTTP extraction.
172
+
173
+ ### YouTube videos
174
+
175
+ YouTube URLs are processed via Gemini for full video understanding — visual descriptions, transcripts with timestamps, and chapter markers. Pass a `prompt` to ask specific questions about the video. Results include the video thumbnail so the agent gets visual context alongside the transcript.
176
+
177
+ Fallback: Gemini Web when browser cookies are enabled → Gemini API → Exa (text summary only). Handles all URL formats: `/watch?v=`, `youtu.be/`, `/shorts/`, `/live/`, `/embed/`, `/v/`.
178
+
179
+ ### Local video files
180
+
181
+ Pass a file path (`/`, `./`, `../`, or `file://` prefix) to analyze video content via Gemini. Supports MP4, MOV, WebM, AVI, and other common formats up to 50MB for Gemini analysis. Pass a `prompt` to ask about specific content. If ffmpeg is installed, a thumbnail frame is included alongside the analysis. Timestamp/frame extraction uses ffmpeg directly and can still operate on larger local files.
182
+
183
+ Fallback: Gemini API (Files API upload) → Gemini Web when browser cookies are enabled.
184
+
185
+ ### Video frame extraction
186
+
187
+ Use `timestamp` and/or `frames` on any YouTube URL or local video file to extract visual frames as images.
188
+
189
+ ```typescript
190
+ fetch_content({ url: "...", timestamp: "23:41" }) // single frame
191
+ fetch_content({ url: "...", timestamp: "23:41-25:00" }) // range, 6 frames
192
+ fetch_content({ url: "...", timestamp: "23:41-25:00", frames: 3 }) // range, custom count
193
+ fetch_content({ url: "...", timestamp: "23:41", frames: 5 }) // 5 frames at 5s intervals
194
+ fetch_content({ url: "...", frames: 6 }) // sample whole video
195
+ ```
196
+
197
+ Requires `ffmpeg` (and `yt-dlp` for YouTube). Timestamps accept `H:MM:SS`, `MM:SS`, or bare seconds.
198
+
199
+ ### PDFs
200
+
201
+ PDF URLs are converted to Markdown and saved under the temporary `pi-web-pdf` directory by default so the agent can `read` specific sections without loading the full document into context. Three engines are available, selected with `pdf.provider` (`"auto"` is the default):
202
+
203
+ | Provider | Engine | Trade-offs |
204
+ | --- | --- | --- |
205
+ | `datalab` | Datalab hosted conversion (Marker) | Deterministic layout-aware output — tables, multi-column reading order, headings, math; `accurate` mode handles scanned pages; may return a `parse_quality_score`; requires a Datalab key, billed per page with a free monthly credit |
206
+ | `gemini` | Gemini API (vision LLM) | Best on scanned/complex pages; LLM transcription can occasionally drift or truncate; requires a Gemini key |
207
+ | `unpdf` | Local pdf.js text extraction | Free, offline, no key; flattened text only — no layout, no tables, no OCR |
208
+
209
+ `auto` order: Datalab (when a key is configured) → Gemini (when a key is configured) → local `unpdf`. Datalab runs first for layout-aware conversion. If its request fails — including after free-tier credit is exhausted — the chain continues to Gemini, then `unpdf`, automatically. Setting `pdf.provider` to `gemini`, `datalab`, or `unpdf` pins that engine and skips the other remote tiers (an explicit engine still falls back to `unpdf` when it errors, except for credential/config errors and caller cancellation). No Datalab key means the `datalab` tier is simply skipped — behavior is unchanged for existing users.
210
+
211
+ **Why Datalab.** The hosted converter uses a dedicated extraction engine (Marker) intended to retain document structure such as tables, multi-column reading order, headings, links, and math, where local `unpdf` extraction only yields flattened text. It is deterministic rather than LLM-based. Completed responses may include a `parse_quality_score` (0–5) for optional quality gating. Pricing is per processed page: **fast / balanced** $4 / 1,000 pages; **accurate** $10 / 1,000 pages. The free tier gives a **$10 monthly credit** (personal email; $20 with a work email) at **25 requests/minute** — roughly **2,500 pages/month free in `fast` mode** or 1,000 in `accurate` mode. Processing defaults to the **US region**. EU data residency uses **1.25× usage**; opt in with `DATALAB_PROCESSING_LOCATION=eu`.
212
+
213
+ Configure Datalab via the web-search config:
214
+
215
+ ```jsonc
216
+ {
217
+ "datalabApiKey": "$DATALAB_API_KEY",
218
+ "pdf": {
219
+ "maxSizeMB": 20,
220
+ "maxPages": 100,
221
+ "provider": "auto", // "auto" | "gemini" | "datalab" | "unpdf"
222
+ "datalabMode": "balanced", // "fast" | "balanced" | "accurate"
223
+ "datalabTimeoutMs": 120000
224
+ }
225
+ }
226
+ ```
227
+
228
+ Env vars: `DATALAB_API_KEY` (or `datalabApiKey` in config), `DATALAB_PROCESSING_LOCATION` (`us` default; `eu` enables EU data residency at 1.25× usage), `DATALAB_MODE` (`fast` / `balanced` / `accurate`), and `DATALAB_API_BASE` (custom gateway). `pdf.datalabMode` overrides `DATALAB_MODE`. The default `datalabTimeoutMs` is 120s and is capped at 300s.
229
+
230
+ > Privacy note: like the Gemini tier, the PDF bytes are sent to the Datalab cloud for conversion. Files are uploaded to the selected region's storage and deleted best-effort after conversion.
231
+
232
+ ### Blocked pages
233
+
234
+ Raw and direct-image HTTP requests use the same SSRF validation, hostname domain policy, redirect checks, timeout, and 5MB streamed response bound as normal extraction. Raw mode returns textual bodies even for non-2xx responses and exposes the HTTP status in tool details; it does not run readability or hosted extraction fallbacks.
235
+
236
+ `fetch_content` can opt into local browser-cookie auth with `auth: "profile"`, or `auth: true` when exactly one `authFetch` profile exists. Configure profiles in `~/.pi/web-search.json`, for example `{ "authFetch": { "social": ["x.com", "instagram.com"], "work": { "hosts": ["docs.company.com"], "chromeProfile": "Profile 2", "cache": "off" } } }`. Auth fetch uses only the local direct HTTP path, requires HTTPS, allows only configured hosts and their subdomains, refuses cross-origin redirects, and never sends cookies or authenticated content to hosted extraction providers. Browser cookie extraction remains opt-in through `allowBrowserCookies: true` or `PI_ALLOW_BROWSER_COOKIES=1`.
237
+
238
+ #### Proxy (`proxy`)
239
+
240
+ `web_search`, `source_check`, and `fetch_content` all accept an optional `proxy` string (e.g. `"http://mcr:4444"`). When provided, every outbound HTTP(S) request is routed through `curl` instead of Node's built-in fetch — this works around Node fetch ignoring `HTTP(S)_PROXY` env vars and undici `ProxyAgent` failing the TLS handshake against several common HTTP proxies (ERR_SSL_WRONG_VERSION_NUMBER).
241
+
242
+ An empty string (`""`) forces a direct connection even when a config-level proxy is set. Omitting the parameter falls back to the global `proxy` in `~/.pi/web-search.json`.
243
+
244
+ ```jsonc
245
+ // ~/.pi/web-search.json — global proxy for all tools
246
+ {
247
+ "proxy": "http://mcr:4444"
248
+ }
249
+ ```
250
+
251
+ Localhost, `127.0.0.1`, `[::1]`, and any host matching the `NO_PROXY` environment variable are never proxied.
252
+
253
+ When Readability fails or returns only a cookie notice, the extension can retry Jina Reader (handles JS rendering server-side, no API key needed), Bright Data Web Unlocker, Gemini URL Context API, and Gemini Web extraction when browser cookies are enabled. Configure `fetchRouting.providers` to change the order or set of `fetch_content` providers. Supported values are `http`, `jina`, `brightdata`, and `gemini`; when absent, the default order is unchanged. For remote HTTP(S) targets, third-party hosted providers are disabled unless `fetchRouting.allowRemoteHostedProviders` is `true`, because hosted services perform their own fetch and can see a different redirect chain than the local safety gate. Bright Data Web Unlocker runs ahead of only the Gemini fallbacks, because it is billed per request against a paid account; it is skipped unless both a key and an `unblocker` zone are configured. It applies no minimum-length check, so any non-empty body it returns — including a short consent or paywall stub — is the final answer for that URL and the Gemini fallbacks are not tried. Handles SPAs, JS-heavy pages, and anti-bot protections transparently. Also parses Next.js RSC flight data when present. HTML extraction also surfaces registered discovery relations (`service-desc`, `service-doc`, `service-meta`, `api-catalog`, `describedby`) from the HTTP `Link` header and matching `link`/`a[rel]` markup. Readable or rendered content remains primary; on an empty shell, the normal extraction fallbacks run before declared links are returned on their own.
254
+
255
+ ## How It Works
256
+
257
+ ```
258
+ web_search(query)
259
+ → Exa (direct API if keyed, MCP if not)
260
+
261
+ fetch_content(url)
262
+ → Video file? Gemini API (Files API) → Gemini Web (if browser cookies enabled)
263
+ → GitHub URL? Clone repo, return file contents + local path
264
+ → YouTube URL? Gemini Web (if browser cookies enabled) → Gemini API → Exa
265
+ → HTTP fetch → PDF? Datalab → Gemini API → local text extraction, save to temp pi-web-pdf
266
+ → HTML? Readability (+ declared Link/rel discovery) → RSC parser → third-party hosted fallbacks only when fetchRouting.allowRemoteHostedProviders is enabled
267
+ → Text/JSON/Markdown? Return directly
268
+ ```
269
+
270
+ ## Commands
271
+
272
+ ### /search
273
+
274
+ Browse stored search results interactively. Lists all results from the current session with their response IDs for easy retrieval.
275
+
276
+ ### /google-account
277
+
278
+ Show the active Google account currently authenticated for Gemini Web. If cookie extraction fails, it reports sanitized attempted browser/profile entries and whether the failure was missing required cookies, password-store access, decryption, SQLite, or profile lookup.
279
+
280
+ ## Activity Monitor
281
+
282
+ Toggle with **Ctrl+Shift+W** to see live request/response activity:
283
+
284
+ ```
285
+ ─── Web Search Activity ────────────────────────────────────
286
+ API "typescript best practices" 200 2.1s ✓
287
+ GET docs.example.com/article 200 0.8s ✓
288
+ GET blog.example.com/post 404 0.3s ✗
289
+ ────────────────────────────────────────────────────────────
290
+ ```
291
+
292
+ ## Configuration
293
+
294
+ Config defaults to `~/.pi/web-search.json`, or `web-search.json` under `PI_CODING_AGENT_DIR` / `XDG_CONFIG_HOME/pi` when set. Every field is optional.
295
+
296
+ ```json
297
+ {
298
+ "exaApiKey": "exa-...",
299
+ "exaBaseUrl": "https://gateway.example.com/exa",
300
+ "brightdataApiKey": "$BRIGHTDATA_API_KEY",
301
+ "brightdataUnlockerZone": "pi_unlocker",
302
+ "geminiApiKey": "AIza...",
303
+ "geminiBaseUrl": "https://my-gateway.example.com/gemini",
304
+ "cloudflareApiKey": "...",
305
+ "geminiAuth": "adc",
306
+ "geminiProject": "my-gcp-project",
307
+ "geminiLocation": "us-central1",
308
+ "fetchRouting": {
309
+ "providers": ["http", "jina", "brightdata", "gemini"],
310
+ "allowRemoteHostedProviders": false
311
+ },
312
+ "webSearch": {
313
+ "enabled": true
314
+ },
315
+ "tools": {
316
+ "webSearch": { "enabled": true },
317
+ "sourceCheck": { "enabled": true },
318
+ "fetchContent": { "enabled": true },
319
+ "getSearchContent": { "enabled": true }
320
+ },
321
+ "commands": {
322
+ "websearch": { "enabled": true },
323
+ "search": { "enabled": true },
324
+ "google-account": { "enabled": true }
325
+ },
326
+ "image": {
327
+ "enabled": true
328
+ },
329
+ "browserCookies": {
330
+ "browser": "helium",
331
+ "profile": "Profile 2"
332
+ },
333
+ "allowBrowserCookies": false,
334
+ "maxInlineContentChars": 30000,
335
+ "githubClone": {
336
+ "enabled": true,
337
+ "maxRepoSizeMB": 350,
338
+ "cloneTimeoutSeconds": 30,
339
+ "clonePath": "/tmp/pi-github-repos"
340
+ },
341
+ "githubPrIssue": {
342
+ "enabled": true
343
+ },
344
+ "youtube": {
345
+ "enabled": true,
346
+ "preferredModel": "gemini-3.6-flash"
347
+ },
348
+ "video": {
349
+ "enabled": true,
350
+ "preferredModel": "gemini-3.6-flash",
351
+ "maxSizeMB": 50
352
+ },
353
+ "pdf": {
354
+ "enabled": true,
355
+ "maxSizeMB": 20,
356
+ "provider": "auto"
357
+ },
358
+ "fetchContent": {
359
+ "domainPolicy": {
360
+ "allow": ["example.com"],
361
+ "deny": ["blocked.example.com"]
362
+ }
363
+ },
364
+ "shortcuts": {
365
+ "activity": "ctrl+shift+w"
366
+ },
367
+ "ssrf": {
368
+ "allowRanges": ["198.18.0.0/15"],
369
+ "trustEnvProxy": false
370
+ }
371
+ }
372
+ ```
373
+
374
+
375
+ API-key fields (`exaApiKey`, `geminiApiKey`, `datalabApiKey`, `cloudflareApiKey`, and `brightdataApiKey`) accept explicit credential sources. Use `$NAME` or `${NAME}` to read one named environment variable, or prefix a trusted local shell command with `!` to resolve one value at provider request time. Escape `$$` as a literal leading `$` and `$!` as a literal leading `!`:
376
+
377
+ ```json
378
+ {
379
+ "exaApiKey": "!/absolute/path/to/secret-manager read exa",
380
+ "geminiApiKey": "${SCOPED_GEMINI_API_KEY}",
381
+ "datalabApiKey": "$$literal-key",
382
+ "cloudflareApiKey": "$!literal-command"
383
+ }
384
+ ```
385
+
386
+ This syntax applies to provider credentials only; other configuration fields are not interpolated. `exaApiKey`, `geminiApiKey`, `datalabApiKey`, `cloudflareApiKey`, and `brightdataApiKey` use the same credential-source rules, while `exaBaseUrl`, `geminiBaseUrl`, and `brightdataUnlockerZone` are literal config values.
387
+
388
+ A command source is not run while the extension loads or registers tools. Each selected provider request runs it again with a five-second timeout, a 16 KiB output limit, a minimized environment, and a one-line non-empty stdout requirement. Command text and stderr are omitted from errors. These commands are trusted local configuration, not a same-user process isolation boundary; use absolute executable paths and protect the config file. `OP_SESSION_*` variables are forwarded to trusted resolver commands so shell-local 1Password sessions can be reused without storing them in config. An explicit source overrides legacy provider environment variables and fails that provider locally rather than falling back with a stale credential. Direct Google Gemini API requests send the resolved key only in the `x-goog-api-key` header, never in the URL.
389
+
390
+ Set `exaBaseUrl` to route Exa through a compatible HTTPS API gateway. `EXA_BASE_URL` is the environment-variable equivalent and takes precedence over config. Exa appends `/answer` or `/search`. The default remains Exa's official API root. `exaBaseUrl` applies only to keyed direct API calls; zero-config Exa MCP search continues to use Exa's hosted MCP endpoint. Invalid overrides fail before a request is sent instead of falling back to an official endpoint, and credential headers are removed if a gateway redirects to another origin.
391
+
392
+ `authFetch` configures named local browser-cookie auth profiles for explicit `fetch_content` calls. A profile can be a host array (`"work": ["docs.company.com"]`) or an object with `hosts`, optional `chromeProfile`, `redirects: "same-origin"`, and `cache: "session" | "off"`.
393
+
394
+ `browserCookies` selects the Chromium browser preset and profile used for Gemini Web cookies, for example `{ "browserCookies": { "browser": "helium", "profile": "Profile 1" } }`. When `browser` is set, cookie discovery checks only that browser, which avoids unrelated password-store prompts. Supported preset names are `helium`, `chrome`, `brave`, `arc`, `chromium`, and `edge`, subject to platform availability. Omit `browser` to keep automatic browser discovery. `profile` must be a profile directory name. The old top-level `chromeProfile` field is rejected; move it to `browserCookies.profile`. Arbitrary profile paths and `profilePath` are intentionally not supported.
395
+
396
+ `fetchContent.domainPolicy` is an optional hostname allow/deny policy for `fetch_content` target URLs. It is off when omitted. Each bare hostname matches itself and its subdomains; `deny` wins when a hostname matches both lists. The policy is checked before HTTP(S) target handling and before each redirect followed by this extension's own fetch path. Local file paths and non-HTTP sources are not subject to this policy. It is an additional restriction: the existing SSRF guard still blocks private and internal destinations. Remote extraction services can still perform their own DNS, redirects, and egress after this extension preflights the submitted target URL, so third-party hosted HTTP(S) fallbacks stay disabled unless `fetchRouting.allowRemoteHostedProviders` is enabled for separately isolated provider deployments.
397
+
398
+
399
+
400
+
401
+ **Bright Data.** Set `brightdataApiKey` or `BRIGHTDATA_API_KEY` plus `brightdataUnlockerZone` or `BRIGHTDATA_UNLOCKER_ZONE` (a zone of type `unblocker`) to enable the Bright Data Web Unlocker `fetch_content` fallback.
402
+
403
+
404
+ Bright Data Web Unlocker is a paid `fetch_content` fallback after the direct fetch and before Gemini. It validates the target URL with the local SSRF guard before resolving credentials or sending any request to Bright Data, validates redirects from the Bright Data API endpoint, and strips authorization across cross-origin API redirects. As with any remote extraction service, Bright Data fetches the submitted target from its own infrastructure; keep `brightdataUnlockerZone` unset for URLs that must not be disclosed to a third party. Successful Unlocker responses are returned as Markdown, including short pages or consent stubs, because the request has already been billed and discarding the body would hide what Bright Data saw.
405
+
406
+
407
+
408
+
409
+
410
+
411
+
412
+ Without an explicit `$` or `!` source, `BRIGHTDATA_API_KEY`, `BRIGHTDATA_UNLOCKER_ZONE`, `EXA_API_KEY`, `EXA_BASE_URL`, `GEMINI_API_KEY`, `GOOGLE_GEMINI_BASE_URL`, `CLOUDFLARE_API_KEY`, `DATALAB_API_KEY`, `DATALAB_PROCESSING_LOCATION`, `DATALAB_MODE`, and `DATALAB_API_BASE` env vars retain their existing precedence over literal config file values. Configured Exa API keys use Exa's own account limits directly; any legacy local `exa-usage.json` file is ignored. `GOOGLE_GEMINI_BASE_URL` overrides the Gemini API host for Gemini generate-content calls such as search, URL context, YouTube, and local video analysis. Set it to a bare host with no trailing slash and no version segment, for example `https://my-gateway.example.com/gemini`; `geminiBaseUrl` is the config-file equivalent. When the configured host contains `gateway.ai.cloudflare.com`, authentication uses `cf-aig-authorization: Bearer <token>` from `CLOUDFLARE_API_KEY` or `cloudflareApiKey`, and `GEMINI_API_KEY` is not required for generate-content calls. Alternatively, set `geminiAuth` to `"adc"` to authenticate Gemini generate-content calls with Google Application Default Credentials (ADC) instead of an API key; calls go to the Vertex AI endpoint (`aiplatform.googleapis.com`) with an OAuth bearer token minted from the ADC file (`GOOGLE_APPLICATION_CREDENTIALS` or `~/.config/gcloud/application_default_credentials.json`, i.e. `gcloud auth application-default login`). `geminiProject`/`geminiLocation` set the Vertex project and location and fall back to the `GOOGLE_CLOUD_PROJECT`/`GOOGLE_CLOUD_LOCATION` (or `GCLOUD_PROJECT`) env vars; project and location are required. ADC supports `authorized_user` (OAuth refresh token) and `service_account` (JWT assertion) credential files, and tokens are cached and refreshed from expiry. ADC mode covers search, URL context, and PDF/inline-data extraction; YouTube and local video analysis still go through the Gemini Files API, so they fall back to Gemini Web unless a `GEMINI_API_KEY` is also configured. The access token is treated as a credential and is redacted from errors. Local video file upload still uses Google's Files API directly, so gateway-only video extraction falls back to Gemini Web unless a `GEMINI_API_KEY` is also configured. Set `webSearch.enabled` to `false` to unregister the configured search and source-check tools while leaving fetch/content tools available. `toolNames` can opt into alternate public tool names for environments where another extension or model reserves the defaults, without changing behavior: `webSearch`, `sourceCheck`, `fetchContent`, and `getSearchContent` default to `web_search`, `source_check`, `fetch_content`, and `get_search_content`. `browserCookies.profile` pins Gemini Web cookie lookup to a specific Chromium profile. When omitted, detected Chromium profiles are scanned in stable order and the first profile containing the required Gemini cookies is used. macOS discovery supports Helium, Chrome, Brave, and Arc; Linux discovery supports Chromium and Chrome. `allowBrowserCookies` enables Chromium cookie extraction for Gemini Web; it defaults to `false` to avoid browser data access and surprise macOS Keychain prompts. You can also set `PI_ALLOW_BROWSER_COOKIES=1`. Cookie databases are copied to a temporary read-only working copy; the reader uses `node:sqlite` when available and otherwise tries the `sqlite3` CLI or Python's standard-library SQLite module. `ssrf.allowRanges` lists CIDR ranges (e.g. `"198.18.0.0/15"`, `"fd00::/8"`) exempted from the SSRF guard that otherwise blocks private/reserved IP ranges. This unblocks `fetch_content`/`web_search` on hosts whose network proxy runs in TUN + fake-IP mode (Surge, Clash, Mihomo, Stash, ...), where public domains resolve into a synthetic reserved range. It is **off by default** — the guard stays fully enabled unless you list ranges here. Use the narrowest range that covers your proxy's fake-IP pool. All-address CIDRs such as `0.0.0.0/0` and `::/0` are rejected. `ssrf.trustEnvProxy` is a separate opt-in for sandboxed environments with valid HTTP(S) proxy env vars; it skips local DNS preflight only for proxied hostnames and still blocks localhost, literal private IPs, and `NO_PROXY` matches. It does not configure proxy transport.
413
+ ### Shortcuts
414
+
415
+ The shortcut is configurable via `~/.pi/web-search.json`:
416
+
417
+ ```json
418
+ {
419
+ "shortcuts": {
420
+ "activity": "ctrl+shift+w"
421
+ }
422
+ }
423
+ ```
424
+
425
+ Values use the same format as pi keybindings (e.g. `ctrl+s`, `ctrl+shift+s`, `alt+r`). Changes take effect on next pi restart.
426
+
427
+ Set `"enabled": false` under `tools`, `commands`, `image`, or `pdf` to disable that feature. Tool-specific settings override the legacy `webSearch.enabled` shorthand; without an override, it still disables `web_search` and `source_check`. `image.enabled: false` blocks direct image fetches and video frame extraction, and prevents video thumbnails. `pdf.enabled: false` blocks PDF extraction. For GitHub specifically, `githubClone.enabled: false` only skips clone/API specialization, and `githubPrIssue.enabled: false` only skips PR/issue specialization; neither setting unregisters `fetch_content` or blocks generic URL extraction. Pi restart is required for tool and command registration changes.
428
+
429
+ Rate limits: Content fetches run 3 concurrent with a 30s timeout for the direct HTTP fetch of each URL. Remote extraction fallbacks carry their own budgets and are not covered by that number: Jina Reader 30s, Bright Data Web Unlocker 60s, Gemini 120s, Datalab 120s (capped at 300s, rate-limited to 25 requests/minute on the free tier). `pdf.maxSizeMB` defaults to 20 and is capped at 50. `pdf.maxPages` defaults to 100 and limits every PDF provider to the first N pages.
430
+
431
+ ## Limitations
432
+
433
+ - Chromium cookie extraction for Gemini Web is opt-in via `allowBrowserCookies: true` or `PI_ALLOW_BROWSER_COOKIES=1`; no browser data or password store is touched while it is disabled. On macOS, enabling it may trigger a Keychain dialog. On Windows, Chrome and Edge v10 cookies use the current user's DPAPI key; v20 app-bound cookies are not supported. Required cookie names are checked before password-store access, and browser encryption passwords are cached only in-process. If `node:sqlite` is unavailable, the reader falls back to the `sqlite3` CLI or Python stdlib; `/google-account` reports sanitized browser/profile attempts and classifies SQLite, profile, missing-cookie, password-store, and decryption failures.
434
+ - YouTube private/age-restricted videos may fail on all extraction paths.
435
+ - Gemini can process videos up to ~1 hour; longer videos may be truncated.
436
+ - PDFs are text-extracted only (no OCR for scanned documents).
437
+ - GitHub branch names with slashes may misresolve file paths; the clone still works and the agent can navigate manually.
438
+ - GitHub wiki, discussion, and other non-code pages still fall through to normal web extraction. PR review-thread resolution state is unknown in the specialized PR view, and REST fallback omits checks.
439
+
440
+ <details>
441
+ <summary>Files</summary>
442
+
443
+ | File | Purpose |
444
+ | ------ | --------- |
445
+ | `index.ts` | Extension entry, tool definitions, commands, widget |
446
+ | `brightdata-unlocker.ts` | Bright Data Web Unlocker extraction fallback |
447
+ | `exa.ts` | Exa.ai search provider — direct API and MCP proxy |
448
+ | `extract.ts` | URL/file path routing, HTTP extraction, fallback orchestration |
449
+ | `content-find.ts` | Bounded exact, case-insensitive, and fuzzy passage lookup |
450
+ | `page-query.ts` | Grounded page-local answer generation with model context budgeting |
451
+ | `gemini-search.ts` | Exa search entry point (direct API if keyed, MCP if not) |
452
+ | `search-types.ts` | Shared search result/option types |
453
+ | `gemini-url-context.ts` | Gemini URL Context + Web extraction fallbacks |
454
+ | `gemini-web.ts` | Gemini Web client (cookie auth, StreamGenerate) |
455
+ | `gemini-web-config.ts` | Gemini Web profile and browser-cookie opt-in config |
456
+ | `gemini-api.ts` | Gemini REST API client (generateContent) |
457
+ | `chrome-cookies.ts` | Chromium-based cookie extraction (macOS Keychain, Linux secret-tool, Windows DPAPI + SQLite) |
458
+ | `youtube-extract.ts` | YouTube detection, three-tier extraction, frame extraction |
459
+ | `video-extract.ts` | Local video detection, Files API upload, Gemini analysis |
460
+ | `github-extract.ts` | GitHub URL parsing, clone cache, content generation |
461
+ | `github-api.ts` | GitHub API fallback for large repos and commit SHAs |
462
+ | `github-issue-pr.ts` | GitHub PR and issue URL parsing, gh/REST fetch, markdown rendering |
463
+ | `datalab-pdf-extract.ts` | Datalab hosted PDF-to-Markdown conversion client (upload → convert → poll) |
464
+ | `pdf-extract.ts` | PDF text extraction, saves to markdown |
465
+ | `rsc-extract.ts` | RSC flight data parser for Next.js pages |
466
+ | `utils.ts` | Shared formatting and error helpers |
467
+ | `storage.ts` | Session-aware result storage |
468
+ | `activity.ts` | Activity tracking for the observability widget |
469
+
470
+ </details>
@@ -0,0 +1,5 @@
1
+ # Security Policy
2
+
3
+ Please report suspected vulnerabilities through GitHub private vulnerability reporting for this repository. Do not post exploit details, secrets, or proof-of-concept payloads in public issues or pull requests.
4
+
5
+ If private vulnerability reporting is unavailable for your account or this repository, open a minimal public issue asking for a private contact path without including technical details.
@@ -0,0 +1,101 @@
1
+ // Types
2
+ export interface ActivityEntry {
3
+ id: string;
4
+ type: "api" | "fetch";
5
+ startTime: number;
6
+ endTime?: number;
7
+
8
+ // For API calls
9
+ query?: string;
10
+
11
+ // For URL fetches
12
+ url?: string;
13
+
14
+ // Result - status is number (HTTP code) or null (pending/network error)
15
+ status: number | null;
16
+ error?: string;
17
+ }
18
+
19
+ export interface RateLimitInfo {
20
+ used: number;
21
+ max: number;
22
+ oldestTimestamp: number | null;
23
+ windowMs: number;
24
+ }
25
+
26
+ export class ActivityMonitor {
27
+ private entries: ActivityEntry[] = [];
28
+ private readonly maxEntries = 10;
29
+ private listeners = new Set<() => void>();
30
+ private rateLimitInfo: RateLimitInfo = { used: 0, max: 10, oldestTimestamp: null, windowMs: 60000 };
31
+ private nextId = 1;
32
+
33
+ logStart(partial: Omit<ActivityEntry, "id" | "startTime" | "status">): string {
34
+ const id = `act-${this.nextId++}`;
35
+ const entry: ActivityEntry = {
36
+ ...partial,
37
+ id,
38
+ startTime: Date.now(),
39
+ status: null,
40
+ };
41
+ this.entries.push(entry);
42
+ if (this.entries.length > this.maxEntries) {
43
+ this.entries.shift();
44
+ }
45
+ this.notify();
46
+ return id;
47
+ }
48
+
49
+ logComplete(id: string, status: number): void {
50
+ const entry = this.entries.find((e) => e.id === id);
51
+ if (entry) {
52
+ entry.endTime = Date.now();
53
+ entry.status = status;
54
+ this.notify();
55
+ }
56
+ }
57
+
58
+ logError(id: string, error: string): void {
59
+ const entry = this.entries.find((e) => e.id === id);
60
+ if (entry) {
61
+ entry.endTime = Date.now();
62
+ entry.error = error;
63
+ this.notify();
64
+ }
65
+ }
66
+
67
+ getEntries(): readonly ActivityEntry[] {
68
+ return this.entries;
69
+ }
70
+
71
+ getRateLimitInfo(): RateLimitInfo {
72
+ return this.rateLimitInfo;
73
+ }
74
+
75
+ updateRateLimit(info: RateLimitInfo): void {
76
+ this.rateLimitInfo = info;
77
+ this.notify();
78
+ }
79
+
80
+ onUpdate(callback: () => void): () => void {
81
+ this.listeners.add(callback);
82
+ return () => this.listeners.delete(callback);
83
+ }
84
+
85
+ clear(): void {
86
+ this.entries = [];
87
+ this.rateLimitInfo = { used: 0, max: 10, oldestTimestamp: null, windowMs: 60000 };
88
+ this.notify();
89
+ }
90
+
91
+ private notify(): void {
92
+ for (const cb of this.listeners) {
93
+ try {
94
+ cb();
95
+ } catch {
96
+ }
97
+ }
98
+ }
99
+ }
100
+
101
+ export const activityMonitor = new ActivityMonitor();