pi-unsloth-webtools 0.7.5 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -44
- package/config-ui.ts +2 -2
- package/entities.ts +2 -2
- package/html-to-md.ts +5 -5
- package/index.ts +58 -116
- package/package.json +1 -1
- package/proxy.ts +2 -1
- package/web-access.ts +4 -0
- package/web-fetch.ts +103 -25
- package/web-render.ts +2 -2
- package/web-search.ts +1 -1
package/README.md
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
# pi-unsloth-webtools
|
|
2
2
|
|
|
3
|
-
A [pi](https://github.com/earendil-works/pi-coding-agent) extension providing `web_search
|
|
4
|
-
`web_fetch
|
|
3
|
+
A [pi](https://github.com/earendil-works/pi-coding-agent) extension providing `web_search` and
|
|
4
|
+
`web_fetch` tools. It began as a port of the Unsloth Studio codebase
|
|
5
5
|
([`unslothai/unsloth`](https://github.com/unslothai/unsloth), `studio/backend/core/inference/`);
|
|
6
6
|
the engine, extraction, and PDF layers are still derived from it, but the package is no longer
|
|
7
7
|
behavior-identical to Studio — it enables local file and private-address fetching by default
|
|
8
|
-
and adds a fetch cache, Wayback fallbacks, page metadata,
|
|
9
|
-
|
|
8
|
+
and adds a fetch cache, Wayback fallbacks, page metadata, automatic Jina Reader rendering, and
|
|
9
|
+
other behavior Studio does not have. See
|
|
10
10
|
[Known differences from Studio](#known-differences-from-studio). The `unsloth` in the name marks
|
|
11
11
|
provenance, not affiliation.
|
|
12
12
|
|
|
@@ -26,7 +26,7 @@ pi install /path/to/pi-unsloth-webtools
|
|
|
26
26
|
|
|
27
27
|
## What it does
|
|
28
28
|
|
|
29
|
-
|
|
29
|
+
Both tools display their target in the TUI tool row: `web_search "query"` and `web_fetch <url>`.
|
|
30
30
|
|
|
31
31
|
### web_search
|
|
32
32
|
|
|
@@ -39,11 +39,8 @@ Mirrors Unsloth Studio's `web_search` tool:
|
|
|
39
39
|
re-serialized, collapsing host-case, default-port, and trailing-slash variants — so the
|
|
40
40
|
same page found via different tracking links collapses), and the same `SimpleFilterRanker`
|
|
41
41
|
re-ranking. Formats results identically: `Title:` / `URL:` /
|
|
42
|
-
`Snippet:` blocks separated by `---`, ending with the hint to
|
|
42
|
+
`Snippet:` blocks separated by `---`, ending with the hint to call `web_fetch` to
|
|
43
43
|
read a full page.
|
|
44
|
-
- Accepts an optional `url` parameter; when given, fetches that page's text instead of
|
|
45
|
-
searching (optionally truncated with `maxChars`). An HTTP 403 on that fetch falls back
|
|
46
|
-
to `web_render` when the tool is enabled.
|
|
47
44
|
- Rate-limit, timeout, and empty-result messages mirror Studio's `_search_failure_message`.
|
|
48
45
|
- Transient engine failures (network errors or null responses) are retried once with a short
|
|
49
46
|
backoff inside the same timeout budget (a retry that cannot fit in the remaining budget is
|
|
@@ -83,7 +80,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
83
80
|
fetch; the deadline abort cuts a retry short when no budget remains.
|
|
84
81
|
- Proxy environment variables are honored when they name a SOCKS5 proxy: `HTTPS_PROXY` /
|
|
85
82
|
`HTTP_PROXY` / `ALL_PROXY` (with `NO_PROXY` exclusions) tunnel the pinned connection, so a
|
|
86
|
-
Tor-mode agent routes `web_fetch`
|
|
83
|
+
Tor-mode agent routes `web_fetch` through its exit. `socks5h` is
|
|
87
84
|
treated like `socks5`: the host is resolved locally for the guard and the pinned IP is what the
|
|
88
85
|
proxy connects to. Other proxy schemes are ignored (direct connection).
|
|
89
86
|
- GitHub repo root pages are rewritten to the unauthenticated README API
|
|
@@ -116,7 +113,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
116
113
|
with link-density header stripping; boilerplate-line removal.
|
|
117
114
|
- No page-size budget: fetched pages and PDFs are returned in full (Studio's window-aware
|
|
118
115
|
cap is deliberately dropped; the optional `maxChars` parameter still truncates when given,
|
|
119
|
-
on `web_fetch`
|
|
116
|
+
on `web_fetch`).
|
|
120
117
|
The 512 KiB / 10 MiB download caps still bound the raw fetch.
|
|
121
118
|
- HTML entity decoding replicates CPython's `html.unescape` (full 2,231-entry HTML5 table,
|
|
122
119
|
longest-prefix rule, Windows-1252 numeric mappings), matching Studio byte-for-byte.
|
|
@@ -125,28 +122,35 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
125
122
|
`Date:` (`article:published_time` / `dc.date` / `date`) and `Site:` (`og:site_name` /
|
|
126
123
|
`application-name`) lines are added when declared, so the model can judge recency and
|
|
127
124
|
provenance.
|
|
128
|
-
- A direct fetch refused with HTTP 403 is retried through
|
|
129
|
-
|
|
125
|
+
- A direct fetch refused with HTTP 403 is retried through the Jina Reader automatically when
|
|
126
|
+
rendering is enabled; the rendered page is prefixed with a note saying so. When rendering is
|
|
130
127
|
disabled or the render also fails, the original `Failed to fetch URL: HTTP 403 ...` is returned.
|
|
128
|
+
- Pages that look JavaScript-rendered are re-fetched through the Jina Reader the same way. When
|
|
129
|
+
rendering is disabled or does not produce more text, the page gets a
|
|
130
|
+
`*(JavaScript-rendered page; content may be incomplete)*` note, so a shell is not mistaken for
|
|
131
|
+
the whole page.
|
|
131
132
|
|
|
132
|
-
###
|
|
133
|
+
### JavaScript rendering
|
|
133
134
|
|
|
134
|
-
|
|
135
|
-
that
|
|
135
|
+
`web_fetch` retries through the third-party Jina Reader (`r.jina.ai`) when
|
|
136
|
+
a direct fetch is refused with HTTP 403 or returns a page that looks JavaScript-rendered (thin
|
|
137
|
+
converted text plus SPA markers, script-heavy markup, a noscript body, or a description meta tag):
|
|
136
138
|
|
|
137
139
|
- Every target is validated and resolved locally first: http/https only, and any private,
|
|
138
|
-
loopback, link-local, or otherwise non-public address is refused. Local files are
|
|
139
|
-
|
|
140
|
-
the URL is sent to Jina.
|
|
140
|
+
loopback, link-local, or otherwise non-public address is refused. Local files are never sent,
|
|
141
|
+
regardless of `webFetch.allowPrivateAddresses` / `webFetch.allowLocalFiles`.
|
|
141
142
|
- `unslothWebTools.jinaApiKey` (or `webRender.jinaApiKey`, or the `JINA_API_KEY` environment
|
|
142
143
|
variable) raises the Reader's rate limits; without a key it still works at Jina's free limits.
|
|
143
|
-
-
|
|
144
|
-
provenance line. An optional `maxChars`
|
|
144
|
+
- Rendered output is Markdown prefixed with `Title:` / `URL:` lines and a `Rendered via the Jina
|
|
145
|
+
Reader` provenance line; the 403 fallback prefixes a note instead. An optional `maxChars`
|
|
146
|
+
truncates, like any fetch.
|
|
147
|
+
- When rendering is disabled or does not produce more text, a page that looks JavaScript-rendered
|
|
148
|
+
is returned with a `*(JavaScript-rendered page; content may be incomplete)*` note, so a shell is
|
|
149
|
+
not mistaken for the whole page.
|
|
150
|
+
- Enabled by default. Disable it with `/webtools-config` (`webRenderEnabled` in
|
|
151
|
+
`~/.config/pi-unsloth-webtools/config.json`), which turns off both escalation paths.
|
|
145
152
|
- Keyless Reader requests are rate-limited per outgoing IP; see
|
|
146
153
|
[Companion: rotating exit IPs](#companion-rotating-exit-ips).
|
|
147
|
-
- Enabled by default. Disable it with `/webtools-config` (`webRenderEnabled` in
|
|
148
|
-
`~/.config/pi-unsloth-webtools/config.json`), which deactivates the tool for the session.
|
|
149
|
-
- Also used automatically when a `web_fetch` or `web_search` url-mode fetch is refused with HTTP 403.
|
|
150
154
|
|
|
151
155
|
## Known differences from Studio
|
|
152
156
|
|
|
@@ -155,11 +159,10 @@ that need JavaScript to render:
|
|
|
155
159
|
(`file://` URLs and absolute, `~/`, `./` paths) by default; opt out with
|
|
156
160
|
`webFetch.allowPrivateAddresses: false` and `webFetch.allowLocalFiles: false` to restore
|
|
157
161
|
Studio's behavior.
|
|
158
|
-
- Third-party rendering:
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
targets also leave the machine in that case.
|
|
162
|
+
- Third-party rendering: when rendering is enabled, a fetch refused with HTTP 403 or returning a
|
|
163
|
+
page that looks JavaScript-rendered is retried through the Jina Reader (`r.jina.ai`), so the
|
|
164
|
+
target URL leaves the machine. Studio has no third-party rendering path. This path always refuses
|
|
165
|
+
local files and non-public addresses, regardless of the local-access settings.
|
|
163
166
|
- PDF styling: MuPDF.js exposes one font per line, so mixed-style lines style the
|
|
164
167
|
whole line instead of per-span; superscript, subscript, underline, strikeout, and
|
|
165
168
|
highlight markers are not emitted. Tables use a conservative text-grid detector:
|
|
@@ -182,7 +185,7 @@ that need JavaScript to render:
|
|
|
182
185
|
- Proxies: Studio routes through environment proxies; this port resolves and pins the target IP and
|
|
183
186
|
tunnels that connection through `HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` when the proxy is a
|
|
184
187
|
SOCKS5 proxy (`NO_PROXY` exclusions respected; DNS stays local for the guard). Other proxy
|
|
185
|
-
schemes fall back to a direct connection. The search and
|
|
188
|
+
schemes fall back to a direct connection. The search and Jina rendering paths use the process-wide
|
|
186
189
|
`fetch`, so an agent-level proxy dispatcher applies there too — see
|
|
187
190
|
[Companion: rotating exit IPs](#companion-rotating-exit-ips).
|
|
188
191
|
- Dedup and titles: the aggregator keys on canonicalized hrefs (`utm_*`/tracking parameters
|
|
@@ -204,7 +207,7 @@ is `false`. For other tradeoffs, prefer:
|
|
|
204
207
|
| Need | Use |
|
|
205
208
|
|---|---|
|
|
206
209
|
| Browser-like TLS/HTTP fingerprinting to unblock bot-defended pages | `pi-smart-fetch` (`wreq-js` `chrome_145`) |
|
|
207
|
-
| Headless Chrome for JS-rendered SPAs/YouTube/Reddit threads | Built-in
|
|
210
|
+
| Headless Chrome for JS-rendered SPAs/YouTube/Reddit threads | Built-in automatic Jina Reader rendering in `web_fetch` first; `georgebashi/pi-web-fetch` (puppeteer + trafilatura) when the Reader falls short |
|
|
208
211
|
| Hosted search with semantic ranking and no scraping | `Brave Search API` / `Tavily` / `Exa` via `pi-ollama-web-search` |
|
|
209
212
|
| Prompt-focused page distillation to save context | `pi-web-fetch` `prompt` -> sub-agent or Claude Code `WebFetch(url,prompt)` |
|
|
210
213
|
| Batch fetching many URLs concurrently | `pi-smart-fetch` `batch_web_fetch` or call `web_fetch` in parallel |
|
|
@@ -216,7 +219,7 @@ choose the best tool per URL. No need to fork this package to add those features
|
|
|
216
219
|
|
|
217
220
|
[`pi-tor-proxy`](https://github.com/YuGiMob/pi-tor-proxy) routes pi's in-process `fetch` traffic
|
|
218
221
|
through Tor (it downloads and manages its own Tor binary) and gives each pi instance its own
|
|
219
|
-
circuit and exit IP. The search sweep and
|
|
222
|
+
circuit and exit IP. The search sweep and Jina rendering both use `fetch`, so they leave through
|
|
220
223
|
the current Tor exit, and Jina rate-limits keyless Reader requests per outgoing IP —
|
|
221
224
|
`/tor-cycle` swaps the exit those limits are counted against, while `/tor-country` and
|
|
222
225
|
`/tor-exclude` constrain which exits are used.
|
|
@@ -225,7 +228,7 @@ the current Tor exit, and Jina rate-limits keyless Reader requests per outgoing
|
|
|
225
228
|
pi install npm:pi-unsloth-webtools npm:pi-tor-proxy
|
|
226
229
|
```
|
|
227
230
|
|
|
228
|
-
`web_fetch`
|
|
231
|
+
`web_fetch` also routes: it resolves and pins the target IP, then tunnels
|
|
229
232
|
the connection through the SOCKS5 proxy named by `HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` (with
|
|
230
233
|
`NO_PROXY` exclusions, so localhost and local files stay direct). DNS is still resolved locally for
|
|
231
234
|
the SSRF guard, and the proxy connects to that pinned IP.
|
|
@@ -254,13 +257,13 @@ Optional settings in `~/.pi/agent/settings.json` or `.pi/settings.json` (project
|
|
|
254
257
|
| Key | Default | Description |
|
|
255
258
|
|---|---|---|
|
|
256
259
|
| `unslothWebTools.maxResults` | `5` | Default `maxResults` for `web_search` (clamped 1-20) |
|
|
257
|
-
| `unslothWebTools.maxChars` / `webFetch.maxChars` / `smartFetchDefaultMaxChars` | tool param | Default `maxChars` for `web_fetch`
|
|
258
|
-
| `unslothWebTools.timeoutMs` / `webFetch.timeoutMs` / `smartFetchDefaultTimeoutMs` | `60000` fetch, `300000` search | Default `timeoutMs` when the tool param is absent (>=1000). Fetch
|
|
260
|
+
| `unslothWebTools.maxChars` / `webFetch.maxChars` / `smartFetchDefaultMaxChars` | tool param | Default `maxChars` for `web_fetch` |
|
|
261
|
+
| `unslothWebTools.timeoutMs` / `webFetch.timeoutMs` / `smartFetchDefaultTimeoutMs` | `60000` fetch, `300000` search | Default `timeoutMs` when the tool param is absent (>=1000). Fetch falls back to 60000; `web_search` query mode falls back to 300000 |
|
|
259
262
|
| `webSearch.maxResults` / `smartWebSearch.resultsPerQuery` | same as above | Legacy aliases for `maxResults` |
|
|
260
263
|
| `websitePolicy` | none | Not read from settings. Tools run unrestricted by default; `websitePolicy` is a programmatic option the host passes to `webSearch` / `fetchPageText` |
|
|
261
264
|
| `unslothWebTools.allowPrivateAddresses` / `webFetch.allowPrivateAddresses` | `true` | Opt out to restore the resolved-IP SSRF guard: private/loopback/link-local hosts (localhost, LAN IPs) are refused again. Non-canonical numeric IP encodings stay blocked either way |
|
|
262
|
-
| `unslothWebTools.allowLocalFiles` / `webFetch.allowLocalFiles` | `true` | Opt out to refuse local files in `web_fetch`
|
|
263
|
-
| `unslothWebTools.jinaApiKey` / `webRender.jinaApiKey` | none (`JINA_API_KEY` fallback) | API key for
|
|
265
|
+
| `unslothWebTools.allowLocalFiles` / `webFetch.allowLocalFiles` | `true` | Opt out to refuse local files in `web_fetch` (`file://` URLs, absolute, `~/`, or `./` paths); when enabled, PDFs are extracted and HTML converted |
|
|
266
|
+
| `unslothWebTools.jinaApiKey` / `webRender.jinaApiKey` | none (`JINA_API_KEY` fallback) | API key for automatic Jina Reader rendering; raises its rate limits. Settings keys win over the environment variable |
|
|
264
267
|
|
|
265
268
|
### Settings window
|
|
266
269
|
|
|
@@ -276,14 +279,14 @@ first changed:
|
|
|
276
279
|
|
|
277
280
|
| Key | Default | Description |
|
|
278
281
|
|---|---|---|
|
|
279
|
-
| `webRenderEnabled` | `true` | When `false`,
|
|
282
|
+
| `webRenderEnabled` | `true` | When `false`, automatic Jina Reader rendering is disabled: HTTP 403 and JavaScript-page escalation in `web_fetch` no longer run |
|
|
280
283
|
|
|
281
284
|
On non-Windows platforms the directory honors `XDG_CONFIG_HOME` when set (falling back to
|
|
282
285
|
`~/.config`); on Windows it always uses `~/.config`.
|
|
283
286
|
|
|
284
287
|
Tool params always win over file defaults. Search dedup also strips default ports, so `https://example.com:443/a` and `https://example.com/a` collapse.
|
|
285
288
|
|
|
286
|
-
Environment overrides: `PI_UNSLOTH_CACHE_DIR` changes the fetch cache directory, `PI_UNSLOTH_WEBTOOLS_STATS` opts into append-only sweep stats JSONL, `PI_CODING_AGENT_DIR` / `PI_AGENT_DIR` change the global settings directory, and `JINA_API_KEY` supplies the
|
|
289
|
+
Environment overrides: `PI_UNSLOTH_CACHE_DIR` changes the fetch cache directory, `PI_UNSLOTH_WEBTOOLS_STATS` opts into append-only sweep stats JSONL, `PI_CODING_AGENT_DIR` / `PI_AGENT_DIR` change the global settings directory, and `JINA_API_KEY` supplies the Jina Reader key when no settings key is set. Cache entries live 1 hour and stale copies are served only after a network failure. SOCKS5 proxies named by `HTTPS_PROXY`, `HTTP_PROXY`, or `ALL_PROXY` are honored on every fetch (`NO_PROXY` exclusions apply).
|
|
287
290
|
|
|
288
291
|
## Troubleshooting
|
|
289
292
|
|
|
@@ -300,16 +303,14 @@ Match on the exact prefix. Do not retry blocked hosts with spelling tricks.
|
|
|
300
303
|
| Private address blocked | `Blocked: refusing to fetch the non-public address ...` | The SSRF guard is active (`allowPrivateAddresses: false`); remove it or set `true` to reach localhost/LAN, and write the scheme explicitly (`http://localhost:3000`). |
|
|
301
304
|
| Local file blocked | `Blocked: the URL has an invalid hostname or port.` for paths | Local files are disabled: remove `allowLocalFiles: false` to read `file://`, absolute, `~/`, or `./` paths. |
|
|
302
305
|
| File read failed | `Failed to read file: ...` | Check the path exists and is a regular file. |
|
|
303
|
-
| HTTP failure | `Failed to fetch URL: HTTP ...` | Fix the URL. A 404 automatically tries a Wayback snapshot; a 403 retries through
|
|
306
|
+
| HTTP failure | `Failed to fetch URL: HTTP ...` | Fix the URL. A 404 automatically tries a Wayback snapshot; a 403 retries through the Jina Reader when rendering is enabled. |
|
|
304
307
|
| Proxy failure | `Failed to fetch URL: SOCKS5 proxy ...` | The SOCKS5 proxy refused or failed (for example Tor is stopping). Check the proxy, or unset the proxy variables for a direct fetch. |
|
|
305
308
|
| Non-text / binary | `(non-text content:` / `(binary content,` | Not readable as text by design. |
|
|
306
309
|
| PDF without text | `(PDF contains no extractable text)` / `(PDF content could not be read as text...)` | Scanned or encrypted PDF. |
|
|
307
310
|
| Download cap hit | `... (page truncated at the download limit)` | Raw fetch hit 512 KiB (10 MiB for PDFs). |
|
|
308
311
|
| maxChars cut | `... (truncated, N chars total)` | Raise `maxChars` for the full text. |
|
|
309
|
-
| Empty page | `(page returned no readable text)` | Page had no extractable text;
|
|
310
|
-
|
|
|
311
|
-
| Render blocked (private host) | `Blocked: refusing to fetch the non-public address ...` | `web_render` only reaches public hosts; use `web_fetch` for localhost/LAN. |
|
|
312
|
-
| Render failure | `Failed to render URL: ...` | Jina rejected or failed the request (rate limit, bad key, unreachable page). Retry, or check `unslothWebTools.jinaApiKey` / `JINA_API_KEY`. |
|
|
312
|
+
| Empty page | `(page returned no readable text)` | Page had no extractable text; `web_fetch` retries through the Jina Reader automatically when rendering is enabled. |
|
|
313
|
+
| JavaScript-rendered page | `*(JavaScript-rendered page; content may be incomplete)*` | Rendering was disabled or produced no more text; the page likely needs a browser. |
|
|
313
314
|
| GitHub rewrite | `README of ... (fetched via the GitHub README API):` | Expected repo-root rewrite, not the HTML chrome. |
|
|
314
315
|
| Cache fallback | `Served from cache` / `STALE cache from YYYY-MM-DD` | Network failed; output is the cached copy with its date. |
|
|
315
316
|
| Wayback fallback | `Fetched from Wayback Machine snapshot (YYYY-MM-DD) for ...` | Original 404'd; output is the archived copy with its date. |
|
package/config-ui.ts
CHANGED
|
@@ -15,8 +15,8 @@ export function configRows(config: Config): ConfigRow[] {
|
|
|
15
15
|
return [
|
|
16
16
|
{
|
|
17
17
|
key: "webRenderEnabled",
|
|
18
|
-
label: "
|
|
19
|
-
hint: "
|
|
18
|
+
label: "JavaScript rendering",
|
|
19
|
+
hint: "Jina Reader fallback",
|
|
20
20
|
enabled: config.webRenderEnabled !== false,
|
|
21
21
|
},
|
|
22
22
|
];
|
package/entities.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export const NAMED_ENTITIES: Record<string, string> = {
|
|
1
|
+
export const NAMED_ENTITIES: Record<string, string> = Object.assign(Object.create(null), {
|
|
2
2
|
"AElig": "\u00c6",
|
|
3
3
|
"AElig;": "\u00c6",
|
|
4
4
|
"AMP": "&",
|
|
@@ -2230,7 +2230,7 @@ export const NAMED_ENTITIES: Record<string, string> = {
|
|
|
2230
2230
|
"zscr;": "\ud835\udccf",
|
|
2231
2231
|
"zwj;": "\u200d",
|
|
2232
2232
|
"zwnj;": "\u200c",
|
|
2233
|
-
};
|
|
2233
|
+
});
|
|
2234
2234
|
|
|
2235
2235
|
export const INVALID_CHARREFS: Record<number, string> = {
|
|
2236
2236
|
0: "\ufffd",
|
package/html-to-md.ts
CHANGED
|
@@ -66,7 +66,7 @@ const P_CLOSING_TAGS = new Set([
|
|
|
66
66
|
"ul",
|
|
67
67
|
]);
|
|
68
68
|
|
|
69
|
-
const IMPLICIT_CLOSERS: Record<string, Set<string>> = {
|
|
69
|
+
const IMPLICIT_CLOSERS: Record<string, Set<string>> = Object.assign(Object.create(null), {
|
|
70
70
|
p: P_CLOSING_TAGS,
|
|
71
71
|
li: new Set(["li"]),
|
|
72
72
|
dt: new Set(["dt", "dd"]),
|
|
@@ -76,9 +76,9 @@ const IMPLICIT_CLOSERS: Record<string, Set<string>> = {
|
|
|
76
76
|
th: new Set(["td", "th", "tr"]),
|
|
77
77
|
option: new Set(["option", "optgroup"]),
|
|
78
78
|
optgroup: new Set(["optgroup"]),
|
|
79
|
-
};
|
|
79
|
+
});
|
|
80
80
|
|
|
81
|
-
const CLOSE_BARRIERS: Record<string, Set<string>> = {
|
|
81
|
+
const CLOSE_BARRIERS: Record<string, Set<string>> = Object.assign(Object.create(null), {
|
|
82
82
|
li: new Set(["ul", "ol", "menu"]),
|
|
83
83
|
dt: new Set(["dl"]),
|
|
84
84
|
dd: new Set(["dl"]),
|
|
@@ -87,7 +87,7 @@ const CLOSE_BARRIERS: Record<string, Set<string>> = {
|
|
|
87
87
|
th: new Set(["table"]),
|
|
88
88
|
option: new Set(["select", "datalist"]),
|
|
89
89
|
optgroup: new Set(["select", "datalist"]),
|
|
90
|
-
};
|
|
90
|
+
});
|
|
91
91
|
|
|
92
92
|
const BLOCK_TAGS = new Set([
|
|
93
93
|
"p",
|
|
@@ -106,7 +106,7 @@ const BLOCK_TAGS = new Set([
|
|
|
106
106
|
]);
|
|
107
107
|
|
|
108
108
|
const HEADING_TAGS = new Set(["h1", "h2", "h3", "h4", "h5", "h6"]);
|
|
109
|
-
const INLINE_EMPHASIS: Record<string, string> = { strong: "**", b: "**", em: "*", i: "*" };
|
|
109
|
+
const INLINE_EMPHASIS: Record<string, string> = Object.assign(Object.create(null), { strong: "**", b: "**", em: "*", i: "*" });
|
|
110
110
|
|
|
111
111
|
const HEADER_LINK_DENSITY = 0.93;
|
|
112
112
|
const HEADER_MIN_CHARS = 150;
|
package/index.ts
CHANGED
|
@@ -1,9 +1,14 @@
|
|
|
1
1
|
import { defineTool, type ExtensionAPI, type ExtensionContext, type Theme } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import { Type } from "typebox";
|
|
3
3
|
import { truncateToWidth } from "@earendil-works/pi-tui";
|
|
4
|
-
import { collapseWhitespace } from "./html-to-md.ts";
|
|
4
|
+
import { collapseWhitespace, visibleChars } from "./html-to-md.ts";
|
|
5
5
|
import { SEARCH_TIMEOUT_MS, webSearch as defaultWebSearch } from "./web-search.ts";
|
|
6
|
-
import {
|
|
6
|
+
import {
|
|
7
|
+
DEFAULT_FETCH_TIMEOUT_MS,
|
|
8
|
+
fetchPageOutcome as defaultFetchPageOutcome,
|
|
9
|
+
fetchPageText as defaultFetchPageText,
|
|
10
|
+
type FetchPageOutcome,
|
|
11
|
+
} from "./web-fetch.ts";
|
|
7
12
|
import { renderPageText as defaultRenderPageText } from "./web-render.ts";
|
|
8
13
|
import { loadDefaultFetchSettings, loadDefaultFetchTimeoutMs, loadJinaApiKey } from "./settings.ts";
|
|
9
14
|
import { readConfig, readConfigWithStatus, toggleWebRender } from "./config.ts";
|
|
@@ -39,24 +44,26 @@ function fetchWasForbidden(text: string): boolean {
|
|
|
39
44
|
return /^Failed to fetch URL: HTTP 403\b/m.test(text);
|
|
40
45
|
}
|
|
41
46
|
|
|
47
|
+
function renderFailed(rendered: string): boolean {
|
|
48
|
+
return (
|
|
49
|
+
rendered.startsWith("Failed to render URL:") ||
|
|
50
|
+
rendered.startsWith("Blocked:") ||
|
|
51
|
+
rendered.startsWith("(page returned no readable text)")
|
|
52
|
+
);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function renderImprovesPage(rendered: string, fetched: string): boolean {
|
|
56
|
+
return visibleChars(rendered) > visibleChars(fetched) * 1.5;
|
|
57
|
+
}
|
|
58
|
+
|
|
42
59
|
const FORBIDDEN_FALLBACK_NOTE = "Direct fetch failed with HTTP 403; rendered via the Jina Reader instead.";
|
|
60
|
+
const THIN_FALLBACK_NOTE = "Direct fetch returned little content; rendered via the Jina Reader instead.";
|
|
61
|
+
const INCOMPLETE_NOTE = "\n\n*(JavaScript-rendered page; content may be incomplete)*";
|
|
43
62
|
|
|
44
63
|
const WebSearchParams = Type.Object({
|
|
45
64
|
query: Type.Optional(
|
|
46
65
|
Type.String({ description: "The search query" }),
|
|
47
66
|
),
|
|
48
|
-
url: Type.Optional(
|
|
49
|
-
Type.String({
|
|
50
|
-
description:
|
|
51
|
-
"A URL to fetch full page content from (instead of searching). Use this to read a page found in search results.",
|
|
52
|
-
}),
|
|
53
|
-
),
|
|
54
|
-
maxChars: Type.Optional(
|
|
55
|
-
Type.Number({
|
|
56
|
-
description:
|
|
57
|
-
"Truncate the fetched page to this many characters (only used with the url parameter)",
|
|
58
|
-
}),
|
|
59
|
-
),
|
|
60
67
|
maxResults: Type.Optional(
|
|
61
68
|
Type.Number({
|
|
62
69
|
minimum: 1,
|
|
@@ -87,43 +94,37 @@ const WebFetchParams = Type.Object({
|
|
|
87
94
|
),
|
|
88
95
|
});
|
|
89
96
|
|
|
90
|
-
const WebRenderParams = Type.Object({
|
|
91
|
-
url: Type.String({ description: "Public URL of the page to render" }),
|
|
92
|
-
maxChars: Type.Optional(
|
|
93
|
-
Type.Number({
|
|
94
|
-
description: "Truncate the returned content to this many characters (default: no limit)",
|
|
95
|
-
}),
|
|
96
|
-
),
|
|
97
|
-
timeoutMs: Type.Optional(
|
|
98
|
-
Type.Number({
|
|
99
|
-
minimum: 1000,
|
|
100
|
-
description: "Overall timeout in milliseconds (default: 60000)",
|
|
101
|
-
}),
|
|
102
|
-
),
|
|
103
|
-
});
|
|
104
|
-
|
|
105
97
|
export interface WebToolsDeps {
|
|
106
98
|
fetchPageText?: typeof defaultFetchPageText;
|
|
99
|
+
fetchPageOutcome?: typeof defaultFetchPageOutcome;
|
|
107
100
|
webSearch?: typeof defaultWebSearch;
|
|
108
101
|
renderPageText?: typeof defaultRenderPageText;
|
|
109
102
|
webRenderEnabled?: () => Promise<boolean>;
|
|
110
103
|
}
|
|
111
104
|
|
|
112
105
|
export function createWebTools(deps: WebToolsDeps = {}) {
|
|
113
|
-
const fetchPageText = deps.fetchPageText ?? defaultFetchPageText;
|
|
114
106
|
const webSearch = deps.webSearch ?? defaultWebSearch;
|
|
115
107
|
const renderPageText = deps.renderPageText ?? defaultRenderPageText;
|
|
108
|
+
const legacyFetchPageText = deps.fetchPageText;
|
|
109
|
+
const fetchOutcome: typeof defaultFetchPageOutcome =
|
|
110
|
+
deps.fetchPageOutcome ??
|
|
111
|
+
(legacyFetchPageText
|
|
112
|
+
? async (url, options) => ({ text: await legacyFetchPageText(url, options), hint: null })
|
|
113
|
+
: defaultFetchPageOutcome);
|
|
116
114
|
const webRenderEnabled =
|
|
117
115
|
deps.webRenderEnabled ?? (async () => (await readConfig()).webRenderEnabled !== false);
|
|
118
116
|
|
|
119
|
-
const
|
|
117
|
+
const renderFallback = async (
|
|
120
118
|
url: string,
|
|
121
|
-
|
|
119
|
+
outcome: FetchPageOutcome,
|
|
122
120
|
options: { timeoutMs: number; maxChars?: number; signal?: AbortSignal; cwd?: string },
|
|
123
121
|
): Promise<string | null> => {
|
|
124
|
-
|
|
122
|
+
const forbidden = fetchWasForbidden(outcome.text);
|
|
123
|
+
const thin = outcome.hint !== null && !forbidden;
|
|
124
|
+
if (!forbidden && !thin) return null;
|
|
125
|
+
const incomplete = outcome.text + INCOMPLETE_NOTE;
|
|
125
126
|
try {
|
|
126
|
-
if (!(await webRenderEnabled())) return null;
|
|
127
|
+
if (!(await webRenderEnabled())) return thin ? incomplete : null;
|
|
127
128
|
const apiKey = await loadJinaApiKey(options.cwd);
|
|
128
129
|
const rendered = await renderPageText(url, {
|
|
129
130
|
timeoutMs: options.timeoutMs,
|
|
@@ -131,10 +132,11 @@ export function createWebTools(deps: WebToolsDeps = {}) {
|
|
|
131
132
|
signal: options.signal,
|
|
132
133
|
apiKey,
|
|
133
134
|
});
|
|
134
|
-
if (rendered
|
|
135
|
-
|
|
135
|
+
if (renderFailed(rendered)) return thin ? incomplete : null;
|
|
136
|
+
if (thin && !renderImprovesPage(rendered, outcome.text)) return incomplete;
|
|
137
|
+
return (forbidden ? FORBIDDEN_FALLBACK_NOTE : THIN_FALLBACK_NOTE) + "\n\n" + rendered;
|
|
136
138
|
} catch {
|
|
137
|
-
return null;
|
|
139
|
+
return thin ? incomplete : null;
|
|
138
140
|
}
|
|
139
141
|
};
|
|
140
142
|
|
|
@@ -143,40 +145,17 @@ export function createWebTools(deps: WebToolsDeps = {}) {
|
|
|
143
145
|
name: "web_search",
|
|
144
146
|
label: "Web Search",
|
|
145
147
|
description:
|
|
146
|
-
"Search the web and
|
|
147
|
-
|
|
148
|
-
"A direct fetch refused with HTTP 403 falls back to web_render when that tool is enabled.",
|
|
149
|
-
promptSnippet: "Search the web and fetch page content",
|
|
148
|
+
"Search the web and return snippets for the top results. Use web_fetch to read a page found in the results.",
|
|
149
|
+
promptSnippet: "Search the web and return snippets",
|
|
150
150
|
promptGuidelines: [
|
|
151
|
-
|
|
151
|
+
"Web tool order: web_search to discover, then web_fetch to read a page.",
|
|
152
152
|
],
|
|
153
153
|
parameters: WebSearchParams,
|
|
154
154
|
renderCall(args, theme) {
|
|
155
|
-
const url = collapsedArg(args.url);
|
|
156
155
|
const query = collapsedArg(args.query);
|
|
157
|
-
return toolCallLine(theme, "web_search",
|
|
156
|
+
return toolCallLine(theme, "web_search", query ? `"${query}"` : "");
|
|
158
157
|
},
|
|
159
158
|
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
160
|
-
if (params.url?.trim()) {
|
|
161
|
-
const url = params.url.trim();
|
|
162
|
-
onUpdate?.({ content: [{ type: "text", text: `Fetching ${url}...` }], details: {} });
|
|
163
|
-
const cwd = (_ctx as ExtensionContext | undefined)?.cwd;
|
|
164
|
-
const { timeoutMs, maxChars, allowPrivateAddresses, allowLocalFiles } = await fetchDefaults(cwd, params);
|
|
165
|
-
const text = await fetchPageText(url, {
|
|
166
|
-
timeoutMs,
|
|
167
|
-
signal: signal ?? undefined,
|
|
168
|
-
maxChars,
|
|
169
|
-
allowPrivateAddresses,
|
|
170
|
-
allowLocalFiles,
|
|
171
|
-
});
|
|
172
|
-
const rendered = await renderOnForbidden(url, text, {
|
|
173
|
-
timeoutMs,
|
|
174
|
-
maxChars,
|
|
175
|
-
signal: signal ?? undefined,
|
|
176
|
-
cwd,
|
|
177
|
-
});
|
|
178
|
-
return { content: [{ type: "text", text: rendered ?? text }], details: {} };
|
|
179
|
-
}
|
|
180
159
|
onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
|
|
181
160
|
const timeoutParam = positiveNumber(params.timeoutMs);
|
|
182
161
|
const searchCwd = (_ctx as ExtensionContext | undefined)?.cwd;
|
|
@@ -196,14 +175,15 @@ export function createWebTools(deps: WebToolsDeps = {}) {
|
|
|
196
175
|
description:
|
|
197
176
|
"Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
|
|
198
177
|
"main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
|
|
199
|
-
"stripping.
|
|
178
|
+
"stripping. Pages that look JavaScript-rendered are retried through the Jina Reader " +
|
|
179
|
+
"automatically when rendering is enabled, as are HTTP 403 responses. " +
|
|
180
|
+
"Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
|
|
200
181
|
"README API, so the README is returned instead of the repo page's UI chrome. " +
|
|
201
182
|
"Private/loopback/link-local targets and local files (file:// URLs, absolute, ~/ or ./ paths, including " +
|
|
202
183
|
"PDFs) are supported by default; opt out with webFetch.allowPrivateAddresses: false or " +
|
|
203
|
-
"webFetch.allowLocalFiles: false in settings. The download size is capped.
|
|
204
|
-
"A direct fetch refused with HTTP 403 falls back to web_render (Jina Reader) when that tool is enabled.",
|
|
184
|
+
"webFetch.allowLocalFiles: false in settings. The download size is capped.",
|
|
205
185
|
promptGuidelines: [
|
|
206
|
-
"web_fetch
|
|
186
|
+
"web_fetch escalates JavaScript-rendered pages and HTTP 403 responses to the Jina Reader automatically when rendering is enabled.",
|
|
207
187
|
],
|
|
208
188
|
promptSnippet: "Fetch a web page and return readable text content",
|
|
209
189
|
parameters: WebFetchParams,
|
|
@@ -214,77 +194,45 @@ export function createWebTools(deps: WebToolsDeps = {}) {
|
|
|
214
194
|
onUpdate?.({ content: [{ type: "text", text: `Fetching ${params.url}...` }], details: {} });
|
|
215
195
|
const cwd = (_ctx as ExtensionContext | undefined)?.cwd;
|
|
216
196
|
const { timeoutMs, maxChars, allowPrivateAddresses, allowLocalFiles } = await fetchDefaults(cwd, params);
|
|
217
|
-
const
|
|
197
|
+
const deadlineMs = Date.now() + timeoutMs;
|
|
198
|
+
const outcome = await fetchOutcome(params.url, {
|
|
218
199
|
timeoutMs,
|
|
200
|
+
deadlineMs,
|
|
219
201
|
signal: signal ?? undefined,
|
|
220
202
|
maxChars,
|
|
221
203
|
allowPrivateAddresses,
|
|
222
204
|
allowLocalFiles,
|
|
223
205
|
});
|
|
224
|
-
const rendered = await
|
|
225
|
-
timeoutMs,
|
|
206
|
+
const rendered = await renderFallback(params.url, outcome, {
|
|
207
|
+
timeoutMs: Math.max(1, deadlineMs - Date.now()),
|
|
226
208
|
maxChars,
|
|
227
209
|
signal: signal ?? undefined,
|
|
228
210
|
cwd,
|
|
229
211
|
});
|
|
230
|
-
return { content: [{ type: "text", text: rendered ?? text }], details: {} };
|
|
231
|
-
},
|
|
232
|
-
}),
|
|
233
|
-
webRenderTool: defineTool({
|
|
234
|
-
name: "web_render",
|
|
235
|
-
label: "Web Render",
|
|
236
|
-
description:
|
|
237
|
-
"Render a public web page to Markdown through the third-party Jina Reader (r.jina.ai). " +
|
|
238
|
-
"Use it for JavaScript-rendered pages that web_fetch cannot read. " +
|
|
239
|
-
"The target URL is sent to Jina; local files and non-public addresses are refused. " +
|
|
240
|
-
"An optional JINA_API_KEY or unslothWebTools.jinaApiKey raises the Reader's rate limits.",
|
|
241
|
-
promptSnippet: "Render a JavaScript-rendered page to Markdown via the Jina Reader",
|
|
242
|
-
promptGuidelines: [
|
|
243
|
-
'Use web_render when web_fetch reports "(page returned no readable text)" or the page only fills in through JavaScript.',
|
|
244
|
-
],
|
|
245
|
-
parameters: WebRenderParams,
|
|
246
|
-
renderCall(args, theme) {
|
|
247
|
-
return toolCallLine(theme, "web_render", collapsedArg(args.url));
|
|
248
|
-
},
|
|
249
|
-
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
250
|
-
onUpdate?.({ content: [{ type: "text", text: `Rendering ${params.url}...` }], details: {} });
|
|
251
|
-
const cwd = (_ctx as ExtensionContext | undefined)?.cwd;
|
|
252
|
-
const { timeoutMs, maxChars } = await fetchDefaults(cwd, params);
|
|
253
|
-
const apiKey = await loadJinaApiKey(cwd);
|
|
254
|
-
const text = await renderPageText(params.url, {
|
|
255
|
-
timeoutMs,
|
|
256
|
-
maxChars,
|
|
257
|
-
signal: signal ?? undefined,
|
|
258
|
-
apiKey,
|
|
259
|
-
});
|
|
260
|
-
return { content: [{ type: "text", text }], details: {} };
|
|
212
|
+
return { content: [{ type: "text", text: rendered ?? outcome.text }], details: {} };
|
|
261
213
|
},
|
|
262
214
|
}),
|
|
263
215
|
};
|
|
264
216
|
}
|
|
265
217
|
|
|
266
218
|
export default function (pi: ExtensionAPI) {
|
|
267
|
-
const { webSearchTool, webFetchTool
|
|
219
|
+
const { webSearchTool, webFetchTool } = createWebTools();
|
|
268
220
|
pi.registerTool(webSearchTool);
|
|
269
221
|
pi.registerTool(webFetchTool);
|
|
270
|
-
pi.registerTool(webRenderTool);
|
|
271
222
|
|
|
272
223
|
pi.on("session_start", async (_event, ctx) => {
|
|
273
224
|
try {
|
|
274
|
-
const {
|
|
225
|
+
const { corrupted } = await readConfigWithStatus();
|
|
275
226
|
if (corrupted && ctx.hasUI) {
|
|
276
227
|
ctx.ui.notify("Web tools config was corrupt and was reset to defaults", "warning");
|
|
277
228
|
}
|
|
278
|
-
if (!config.webRenderEnabled) {
|
|
279
|
-
pi.setActiveTools(pi.getActiveTools().filter((name) => name !== "web_render"));
|
|
280
|
-
}
|
|
281
229
|
} catch (error) {
|
|
282
230
|
console.error("Failed to load web tools config:", error);
|
|
283
231
|
}
|
|
284
232
|
});
|
|
285
233
|
|
|
286
234
|
pi.registerCommand("webtools-config", {
|
|
287
|
-
description: "Open the web tools settings window (
|
|
235
|
+
description: "Open the web tools settings window (JavaScript rendering on/off)",
|
|
288
236
|
handler: async (_args, ctx) => {
|
|
289
237
|
if (!ctx.hasUI) {
|
|
290
238
|
ctx.ui.notify("/webtools-config requires interactive mode", "error");
|
|
@@ -298,13 +246,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
298
246
|
done,
|
|
299
247
|
onToggle: async (key) => {
|
|
300
248
|
if (key !== "webRenderEnabled") return;
|
|
301
|
-
|
|
302
|
-
const active = pi.getActiveTools();
|
|
303
|
-
pi.setActiveTools(
|
|
304
|
-
enabled
|
|
305
|
-
? [...new Set([...active, "web_render"])]
|
|
306
|
-
: active.filter((name) => name !== "web_render"),
|
|
307
|
-
);
|
|
249
|
+
await toggleWebRender();
|
|
308
250
|
},
|
|
309
251
|
});
|
|
310
252
|
await overlay.load();
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.8.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools that began as a port of the Unsloth Studio codebase and now diverge from it (multi-engine search, opt-in SSRF guard, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
package/proxy.ts
CHANGED
|
@@ -3,6 +3,7 @@ import https from "node:https";
|
|
|
3
3
|
import net from "node:net";
|
|
4
4
|
import tls from "node:tls";
|
|
5
5
|
import type { Duplex } from "node:stream";
|
|
6
|
+
import { stripIpv6Brackets } from "./web-access.ts";
|
|
6
7
|
|
|
7
8
|
export interface SocksProxy {
|
|
8
9
|
host: string;
|
|
@@ -124,7 +125,7 @@ export function socksProxyForUrl(url: URL): SocksProxy | null {
|
|
|
124
125
|
const proxy = parseSocksProxy(firstEnv(url.protocol === "https:" ? HTTPS_PROXY_VARS : HTTP_PROXY_VARS));
|
|
125
126
|
if (!proxy) return null;
|
|
126
127
|
const port = url.port ? Number(url.port) : url.protocol === "https:" ? 443 : 80;
|
|
127
|
-
if (bypassesProxy(url.hostname, port)) return null;
|
|
128
|
+
if (bypassesProxy(stripIpv6Brackets(url.hostname), port)) return null;
|
|
128
129
|
return proxy;
|
|
129
130
|
}
|
|
130
131
|
|
package/web-access.ts
CHANGED
|
@@ -70,6 +70,10 @@ export function normalizeDomain(value: unknown): string {
|
|
|
70
70
|
return asciiDomain;
|
|
71
71
|
}
|
|
72
72
|
|
|
73
|
+
export function stripIpv6Brackets(hostname: string): string {
|
|
74
|
+
return hostname.startsWith("[") && hostname.endsWith("]") ? hostname.slice(1, -1) : hostname;
|
|
75
|
+
}
|
|
76
|
+
|
|
73
77
|
function compressIpv6(ip: string): string {
|
|
74
78
|
const segments = ip.toLowerCase().split("::");
|
|
75
79
|
if (segments.length > 2) throw new Error("invalid ipv6");
|
package/web-fetch.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { lookup as dnsLookup } from "node:dns/promises";
|
|
|
2
2
|
import type { LookupAllOptions } from "node:dns";
|
|
3
3
|
import http from "node:http";
|
|
4
4
|
import https from "node:https";
|
|
5
|
+
import { isIP } from "node:net";
|
|
5
6
|
import { open } from "node:fs/promises";
|
|
6
7
|
import type { FileHandle } from "node:fs/promises";
|
|
7
8
|
import { homedir } from "node:os";
|
|
@@ -19,9 +20,10 @@ import {
|
|
|
19
20
|
isPublicIp,
|
|
20
21
|
MAX_SIGNAL_TIMEOUT_MS,
|
|
21
22
|
normalizeUrlScheme,
|
|
23
|
+
stripIpv6Brackets,
|
|
22
24
|
type WebsitePolicy,
|
|
23
25
|
} from "./web-access.ts";
|
|
24
|
-
import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown } from "./html-to-md.ts";
|
|
26
|
+
import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown, visibleChars } from "./html-to-md.ts";
|
|
25
27
|
import type { AttrDict } from "./html-to-md.ts";
|
|
26
28
|
import { INVALID_CHARREFS } from "./entities.ts";
|
|
27
29
|
import { getCached, isFresh, setCached, staleNotice } from "./cache.ts";
|
|
@@ -125,6 +127,14 @@ const HTML_LEADING_RE = new RegExp(
|
|
|
125
127
|
);
|
|
126
128
|
const HTML_DOCUMENT_RE = /^<(?:!doctype\s+html\b|\/?(?:html|head|body)\b)/;
|
|
127
129
|
|
|
130
|
+
const RENDER_THIN_PROSE_CHARS = 400;
|
|
131
|
+
const RENDER_SCRIPT_SHARE = 0.5;
|
|
132
|
+
const RENDER_SCRIPT_MIN_HTML_BYTES = 8192;
|
|
133
|
+
const RENDER_MIN_SCRIPTS = 3;
|
|
134
|
+
const RENDER_NOSCRIPT_MIN_CHARS = 40;
|
|
135
|
+
const RENDER_SPA_MARKERS =
|
|
136
|
+
/(?:id\s*=\s*["']?(?:root|app|__next|__nuxt)["']?|data-reactroot|ng-version|__NEXT_DATA__|__NUXT__|__PRELOADED_STATE__)/i;
|
|
137
|
+
|
|
128
138
|
const MIN_SINGLE_BYTE_ASCII_RATIO = 3 / 4;
|
|
129
139
|
const ASCII_TEXT_BYTES = new Set<number>([
|
|
130
140
|
...Array.from({ length: 0x7f - 0x20 }, (_, i) => i + 0x20),
|
|
@@ -235,6 +245,16 @@ export interface RawFetchResult {
|
|
|
235
245
|
contentType: string;
|
|
236
246
|
}
|
|
237
247
|
|
|
248
|
+
export interface RenderHint {
|
|
249
|
+
reason: "empty" | "js-shell";
|
|
250
|
+
evidence: string[];
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
export interface FetchPageOutcome {
|
|
254
|
+
text: string;
|
|
255
|
+
hint: RenderHint | null;
|
|
256
|
+
}
|
|
257
|
+
|
|
238
258
|
function htmlProbe(body: string, re: RegExp): boolean {
|
|
239
259
|
let i = 0;
|
|
240
260
|
const n = body.length;
|
|
@@ -265,6 +285,46 @@ export function looksLikeHtmlDocument(body: string): boolean {
|
|
|
265
285
|
return htmlProbe(body, HTML_DOCUMENT_RE);
|
|
266
286
|
}
|
|
267
287
|
|
|
288
|
+
function noscriptText(html: string): string {
|
|
289
|
+
const parts: string[] = [];
|
|
290
|
+
for (const match of html.match(/<noscript\b[^>]*>[\s\S]*?<\/noscript>/gi) ?? []) {
|
|
291
|
+
parts.push(match.replace(/<[^>]*>/g, " "));
|
|
292
|
+
}
|
|
293
|
+
return collapseWhitespace(decodeHtmlEntities(parts.join(" ")));
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
export function needsRenderHint(html: string, converted: string): RenderHint | null {
|
|
297
|
+
const prose = visibleChars(converted);
|
|
298
|
+
if (prose >= RENDER_THIN_PROSE_CHARS) return null;
|
|
299
|
+
if (prose === 0) return { reason: "empty", evidence: [] };
|
|
300
|
+
const evidence: string[] = [];
|
|
301
|
+
let score = 0;
|
|
302
|
+
if (RENDER_SPA_MARKERS.test(html)) {
|
|
303
|
+
evidence.push("spa-markers");
|
|
304
|
+
score += 2;
|
|
305
|
+
}
|
|
306
|
+
const scriptCount = (html.match(/<script\b/gi) ?? []).length;
|
|
307
|
+
let scriptBytes = 0;
|
|
308
|
+
for (const script of html.match(/<script\b[^>]*>[\s\S]*?<\/script>/gi) ?? []) scriptBytes += script.length;
|
|
309
|
+
if (html.length >= RENDER_SCRIPT_MIN_HTML_BYTES && scriptBytes >= html.length * RENDER_SCRIPT_SHARE) {
|
|
310
|
+
evidence.push("script-heavy");
|
|
311
|
+
score += 2;
|
|
312
|
+
}
|
|
313
|
+
if (scriptCount >= RENDER_MIN_SCRIPTS) {
|
|
314
|
+
evidence.push("scripts");
|
|
315
|
+
score += 1;
|
|
316
|
+
}
|
|
317
|
+
if (noscriptText(html).length >= RENDER_NOSCRIPT_MIN_CHARS) {
|
|
318
|
+
evidence.push("noscript");
|
|
319
|
+
score += 1;
|
|
320
|
+
}
|
|
321
|
+
if (extractPageMeta(html).description) {
|
|
322
|
+
evidence.push("description");
|
|
323
|
+
score += 1;
|
|
324
|
+
}
|
|
325
|
+
return score >= 2 ? { reason: "js-shell", evidence } : null;
|
|
326
|
+
}
|
|
327
|
+
|
|
268
328
|
function parseContentType(value: string | null | undefined): string {
|
|
269
329
|
return /^[\w.+-]+\/[\w.+-]+/.exec(value ?? "")?.[0] ?? "";
|
|
270
330
|
}
|
|
@@ -320,7 +380,7 @@ function hasSingleByteTextEvidence(data: Buffer): boolean {
|
|
|
320
380
|
return ascii / data.length >= MIN_SINGLE_BYTE_ASCII_RATIO;
|
|
321
381
|
}
|
|
322
382
|
|
|
323
|
-
const CHARSET_ALIASES: Record<string, string> = {
|
|
383
|
+
const CHARSET_ALIASES: Record<string, string> = Object.assign(Object.create(null), {
|
|
324
384
|
gbk: "gbk",
|
|
325
385
|
gb2312: "gbk",
|
|
326
386
|
"gb-2312": "gbk",
|
|
@@ -346,7 +406,7 @@ const CHARSET_ALIASES: Record<string, string> = {
|
|
|
346
406
|
"windows-874": "windows-874",
|
|
347
407
|
cp874: "windows-874",
|
|
348
408
|
"tis-620": "tis-620",
|
|
349
|
-
};
|
|
409
|
+
});
|
|
350
410
|
|
|
351
411
|
function normalizeCharset(name: string): string | null {
|
|
352
412
|
const n = name.trim().replace(/["']/g, "").toLowerCase();
|
|
@@ -637,9 +697,10 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
637
697
|
const url = opts.url;
|
|
638
698
|
const transport = url.protocol === "https:" ? https : http;
|
|
639
699
|
const port = url.port ? Number(url.port) : url.protocol === "https:" ? 443 : 80;
|
|
700
|
+
const hostname = stripIpv6Brackets(url.hostname);
|
|
640
701
|
const options: https.RequestOptions = {
|
|
641
702
|
method: "GET",
|
|
642
|
-
host:
|
|
703
|
+
host: hostname,
|
|
643
704
|
port,
|
|
644
705
|
path: url.pathname + url.search,
|
|
645
706
|
headers: opts.headers,
|
|
@@ -652,12 +713,12 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
652
713
|
ip: opts.pinnedIp,
|
|
653
714
|
family: opts.family,
|
|
654
715
|
port,
|
|
655
|
-
servername:
|
|
716
|
+
servername: hostname,
|
|
656
717
|
timeoutMs: Math.max(1, opts.inactivityMs),
|
|
657
718
|
signal: opts.signal,
|
|
658
719
|
});
|
|
659
720
|
} else {
|
|
660
|
-
options.servername = url.protocol === "https:" ?
|
|
721
|
+
options.servername = url.protocol === "https:" && !isIP(hostname) ? hostname : undefined;
|
|
661
722
|
options.lookup = ((_hostname: string, _options: unknown, _callback: unknown) => {
|
|
662
723
|
const done = (typeof _options === "function" ? _options : _callback) as (err: unknown, address: unknown, family?: unknown) => void;
|
|
663
724
|
if ((_options as { all?: boolean } | undefined)?.all) {
|
|
@@ -881,9 +942,10 @@ export async function fetchUrlRaw(
|
|
|
881
942
|
const budgetResult = checkBudget();
|
|
882
943
|
if (budgetResult !== null) return budgetResult;
|
|
883
944
|
const parsed = new URL(currentUrl);
|
|
884
|
-
const
|
|
885
|
-
|
|
886
|
-
|
|
945
|
+
const parsedHostname = stripIpv6Brackets(parsed.hostname);
|
|
946
|
+
const hostHeader = parsedHostname.includes(":")
|
|
947
|
+
? `[${parsedHostname}]${parsed.port ? `:${parsed.port}` : ""}`
|
|
948
|
+
: parsedHostname + (parsed.port ? `:${parsed.port}` : "");
|
|
887
949
|
const headers: Record<string, string> = {
|
|
888
950
|
"User-Agent": userAgent,
|
|
889
951
|
Host: hostHeader,
|
|
@@ -1133,9 +1195,10 @@ interface PageMeta {
|
|
|
1133
1195
|
author: string;
|
|
1134
1196
|
date: string;
|
|
1135
1197
|
site: string;
|
|
1198
|
+
description: string;
|
|
1136
1199
|
}
|
|
1137
1200
|
|
|
1138
|
-
const META_KEYS: Record<string, keyof PageMeta> = {
|
|
1201
|
+
const META_KEYS: Record<string, keyof PageMeta> = Object.assign(Object.create(null), {
|
|
1139
1202
|
author: "author",
|
|
1140
1203
|
"article:author": "author",
|
|
1141
1204
|
"dc.creator": "author",
|
|
@@ -1145,7 +1208,9 @@ const META_KEYS: Record<string, keyof PageMeta> = {
|
|
|
1145
1208
|
datepublished: "date",
|
|
1146
1209
|
"og:site_name": "site",
|
|
1147
1210
|
"application-name": "site",
|
|
1148
|
-
|
|
1211
|
+
description: "description",
|
|
1212
|
+
"og:description": "description",
|
|
1213
|
+
});
|
|
1149
1214
|
|
|
1150
1215
|
function cutAtCharBoundary(text: string, maxChars: number): string {
|
|
1151
1216
|
const sliced = text.slice(0, maxChars);
|
|
@@ -1165,7 +1230,7 @@ function capMetaValue(value: string): string {
|
|
|
1165
1230
|
}
|
|
1166
1231
|
|
|
1167
1232
|
function extractPageMeta(html: string): PageMeta {
|
|
1168
|
-
const meta: PageMeta = { title: extractPageTitle(html), author: "", date: "", site: "" };
|
|
1233
|
+
const meta: PageMeta = { title: extractPageTitle(html), author: "", date: "", site: "", description: "" };
|
|
1169
1234
|
const seen = new Set<string>();
|
|
1170
1235
|
const record = (name: string, attrs: AttrDict) => {
|
|
1171
1236
|
if (name !== "meta") return;
|
|
@@ -1291,6 +1356,13 @@ function localFileFailure(err: unknown): string {
|
|
|
1291
1356
|
return `Failed to read file: ${err instanceof Error ? err.message : String(err)}`;
|
|
1292
1357
|
}
|
|
1293
1358
|
|
|
1359
|
+
const HTML_FILE_EXTENSIONS = new Set([".htm", ".html", ".xht", ".xhtml"]);
|
|
1360
|
+
|
|
1361
|
+
function localFileContentType(filePath: string): string {
|
|
1362
|
+
const dot = filePath.lastIndexOf(".");
|
|
1363
|
+
return dot !== -1 && HTML_FILE_EXTENSIONS.has(filePath.slice(dot).toLowerCase()) ? "text/html" : "";
|
|
1364
|
+
}
|
|
1365
|
+
|
|
1294
1366
|
async function readLocalFile(
|
|
1295
1367
|
filePath: string,
|
|
1296
1368
|
options: { signal?: AbortSignal; maxChars?: number; maxBytes?: number; maxPdfBytes?: number },
|
|
@@ -1333,16 +1405,16 @@ async function readLocalFile(
|
|
|
1333
1405
|
}
|
|
1334
1406
|
const text = decodeWithCodec(body, bomCodecFor(body) ?? "utf-8");
|
|
1335
1407
|
if (looksBinary(text)) return `(binary content, ${body.length} bytes; not readable as text)`;
|
|
1336
|
-
return truncatePageText(withTruncation(renderBody(text,
|
|
1408
|
+
return truncatePageText(withTruncation(renderBody(text, localFileContentType(filePath)), truncated), options.maxChars);
|
|
1337
1409
|
} finally {
|
|
1338
1410
|
await handle.close().catch(() => {});
|
|
1339
1411
|
}
|
|
1340
1412
|
}
|
|
1341
1413
|
|
|
1342
|
-
export async function
|
|
1414
|
+
export async function fetchPageOutcome(
|
|
1343
1415
|
url: string,
|
|
1344
1416
|
options: FetchPageOptions = {},
|
|
1345
|
-
): Promise<
|
|
1417
|
+
): Promise<FetchPageOutcome> {
|
|
1346
1418
|
const timeoutMs = options.timeoutMs ?? DEFAULT_FETCH_TIMEOUT_MS;
|
|
1347
1419
|
const now = options.nowMs ?? Date.now;
|
|
1348
1420
|
const deadlineMs = options.deadlineMs ?? now() + timeoutMs;
|
|
@@ -1353,18 +1425,19 @@ export async function fetchPageText(
|
|
|
1353
1425
|
if (options.allowLocalFiles !== false) {
|
|
1354
1426
|
const localPath = parseLocalPath(url);
|
|
1355
1427
|
if (localPath !== null) {
|
|
1356
|
-
if (!localPath) return "Failed to read file: invalid file URL.";
|
|
1357
|
-
|
|
1428
|
+
if (!localPath) return { text: "Failed to read file: invalid file URL.", hint: null };
|
|
1429
|
+
const localText = await readLocalFile(localPath, {
|
|
1358
1430
|
signal,
|
|
1359
1431
|
maxChars,
|
|
1360
1432
|
maxBytes: options.maxBytes,
|
|
1361
1433
|
maxPdfBytes: options.maxPdfBytes,
|
|
1362
1434
|
});
|
|
1435
|
+
return { text: localText, hint: null };
|
|
1363
1436
|
}
|
|
1364
1437
|
}
|
|
1365
1438
|
url = normalizeUrlScheme(url);
|
|
1366
1439
|
const [allowed, reason] = checkUrlAccess(url, policy);
|
|
1367
|
-
if (!allowed) return reason;
|
|
1440
|
+
if (!allowed) return { text: reason, hint: null };
|
|
1368
1441
|
const rawFetchOptions = {
|
|
1369
1442
|
deadlineMs,
|
|
1370
1443
|
signal,
|
|
@@ -1381,7 +1454,7 @@ export async function fetchPageText(
|
|
|
1381
1454
|
if (rawResult.error === null) {
|
|
1382
1455
|
const out = renderBody(rawResult.body, rawResult.contentType);
|
|
1383
1456
|
await persistCache(url, rawResult.body, rawResult.contentType, useCache);
|
|
1384
|
-
return truncatePageText(out, maxChars);
|
|
1457
|
+
return { text: truncatePageText(out, maxChars), hint: null };
|
|
1385
1458
|
}
|
|
1386
1459
|
}
|
|
1387
1460
|
const readmeApiUrl = githubRepoReadmeApiUrl(url);
|
|
@@ -1397,7 +1470,7 @@ export async function fetchPageText(
|
|
|
1397
1470
|
if (apiBody.trim()) {
|
|
1398
1471
|
const rendered = `README of ${url} (fetched via the GitHub README API):\n\n` + apiBody;
|
|
1399
1472
|
await persistCache(url, rendered, "text/markdown", useCache);
|
|
1400
|
-
return truncatePageText(rendered, maxChars);
|
|
1473
|
+
return { text: truncatePageText(rendered, maxChars), hint: null };
|
|
1401
1474
|
}
|
|
1402
1475
|
const rawReadmeUrl = githubRepoRawReadmeUrl(url);
|
|
1403
1476
|
if (rawReadmeUrl) {
|
|
@@ -1406,7 +1479,7 @@ export async function fetchPageText(
|
|
|
1406
1479
|
if (rawBody.trim()) {
|
|
1407
1480
|
const rendered = `README of ${url} (fetched via the GitHub raw README URL):\n\n` + rawBody;
|
|
1408
1481
|
await persistCache(url, rendered, "text/markdown", useCache);
|
|
1409
|
-
return truncatePageText(rendered, maxChars);
|
|
1482
|
+
return { text: truncatePageText(rendered, maxChars), hint: null };
|
|
1410
1483
|
}
|
|
1411
1484
|
}
|
|
1412
1485
|
}
|
|
@@ -1421,7 +1494,7 @@ export async function fetchPageText(
|
|
|
1421
1494
|
let cachedOut = renderBody(cached.body, cached.contentType);
|
|
1422
1495
|
if (!isFresh(cached, now())) cachedOut += staleNotice(cached);
|
|
1423
1496
|
else cachedOut += "\n\n*Served from cache — network fetch failed*";
|
|
1424
|
-
return truncatePageText(cachedOut, maxChars);
|
|
1497
|
+
return { text: truncatePageText(cachedOut, maxChars), hint: null };
|
|
1425
1498
|
}
|
|
1426
1499
|
} catch {}
|
|
1427
1500
|
}
|
|
@@ -1433,14 +1506,19 @@ export async function fetchPageText(
|
|
|
1433
1506
|
const ts = wb.timestamp ? `${wb.timestamp.slice(0, 4)}-${wb.timestamp.slice(4, 6)}-${wb.timestamp.slice(6, 8)}` : "unknown date";
|
|
1434
1507
|
out = `*Fetched from Wayback Machine snapshot (${ts}) for ${originalUrl}:*\n\n` + out;
|
|
1435
1508
|
await persistCache(originalUrl, wb.body, wb.contentType, useCache);
|
|
1436
|
-
return truncatePageText(out, maxChars);
|
|
1509
|
+
return { text: truncatePageText(out, maxChars), hint: null };
|
|
1437
1510
|
}
|
|
1438
1511
|
} catch {}
|
|
1439
1512
|
}
|
|
1440
1513
|
}
|
|
1441
|
-
return result.error;
|
|
1514
|
+
return { text: result.error, hint: null };
|
|
1442
1515
|
}
|
|
1443
1516
|
const finalOut = renderBody(result.body, result.contentType);
|
|
1517
|
+
const hint = isHtmlContent(result.body, result.contentType) ? needsRenderHint(result.body, finalOut) : null;
|
|
1444
1518
|
await persistCache(originalUrl, result.body, result.contentType, useCache);
|
|
1445
|
-
return truncatePageText(finalOut, maxChars);
|
|
1519
|
+
return { text: truncatePageText(finalOut, maxChars), hint };
|
|
1520
|
+
}
|
|
1521
|
+
|
|
1522
|
+
export async function fetchPageText(url: string, options: FetchPageOptions = {}): Promise<string> {
|
|
1523
|
+
return (await fetchPageOutcome(url, options)).text;
|
|
1446
1524
|
}
|
package/web-render.ts
CHANGED
|
@@ -6,7 +6,7 @@ const DEFAULT_RENDER_TIMEOUT_MS = 60_000;
|
|
|
6
6
|
const JINA_READER_URL = "https://r.jina.ai/";
|
|
7
7
|
const CANCELLED_MESSAGE = "Failed to render URL: cancelled.";
|
|
8
8
|
const TIMED_OUT_MESSAGE = "Failed to render URL: timed out.";
|
|
9
|
-
const LOCAL_FILE_MESSAGE = "Blocked:
|
|
9
|
+
const LOCAL_FILE_MESSAGE = "Blocked: the Jina Reader cannot fetch local files.";
|
|
10
10
|
const EMPTY_READER_MESSAGE = "Failed to render URL: the Jina Reader returned no content.";
|
|
11
11
|
|
|
12
12
|
export interface RenderPageOptions {
|
|
@@ -74,7 +74,7 @@ export async function renderPageText(url: string, options: RenderPageOptions = {
|
|
|
74
74
|
const resolved = await resolveAndValidateHost(hostname, signal, false);
|
|
75
75
|
const resolutionBudget = budgetMessage(options.signal, signal);
|
|
76
76
|
if (resolutionBudget !== null) return resolutionBudget;
|
|
77
|
-
if (!resolved.ok) return resolved.reason
|
|
77
|
+
if (!resolved.ok) return resolved.reason.startsWith("Blocked:") ? resolved.reason : `Failed to render URL: ${resolved.reason}`;
|
|
78
78
|
const headers: Record<string, string> = { Accept: "application/json" };
|
|
79
79
|
const apiKey = options.apiKey?.trim();
|
|
80
80
|
if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
|
package/web-search.ts
CHANGED
|
@@ -114,6 +114,6 @@ export function formatSearchResults(results: SearchResult[]): string {
|
|
|
114
114
|
return (
|
|
115
115
|
text +
|
|
116
116
|
"\n\n---\n\nThese are only short snippets. " +
|
|
117
|
-
|
|
117
|
+
"To read a page, call web_fetch with its URL."
|
|
118
118
|
);
|
|
119
119
|
}
|