pi-unsloth-webtools 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -20
- package/engines.ts +68 -68
- package/index.ts +6 -6
- package/package.json +2 -1
- package/pdf.ts +1 -1
- package/user-agents.ts +12 -0
- package/web-access.ts +11 -9
- package/web-fetch.ts +55 -38
- package/web-search.ts +2 -2
package/README.md
CHANGED
|
@@ -26,7 +26,7 @@ Mirrors Unsloth Studio's `web_search` tool:
|
|
|
26
26
|
|
|
27
27
|
- Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
|
|
28
28
|
(duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
|
|
29
|
-
the same provider
|
|
29
|
+
the same provider deduplication, href-dedupe aggregator with frequency ordering, and the
|
|
30
30
|
same `SimpleFilterRanker` re-ranking. Formats results identically: `Title:` / `URL:` /
|
|
31
31
|
`Snippet:` blocks separated by `---`, ending with the hint to pass `{"url": "<URL>"}` to
|
|
32
32
|
read a full page.
|
|
@@ -45,7 +45,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
45
45
|
validation and fetch.
|
|
46
46
|
- GitHub repo root pages are rewritten to the unauthenticated README API
|
|
47
47
|
(`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
|
|
48
|
-
- Up to
|
|
48
|
+
- Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
|
|
49
49
|
- 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
|
|
50
50
|
(`signal` cancels mid-flight).
|
|
51
51
|
- PDF text extraction via the official MuPDF.js engine (the same C library pymupdf wraps):
|
|
@@ -67,19 +67,19 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
67
67
|
|
|
68
68
|
## Known differences from Studio
|
|
69
69
|
|
|
70
|
-
-
|
|
71
|
-
whole line instead of per-span; superscript
|
|
72
|
-
markers are not emitted. Tables use a conservative text-grid detector
|
|
73
|
-
tables are detected, drawn-rule-only tables are not.
|
|
74
|
-
-
|
|
70
|
+
- PDF styling: MuPDF.js exposes one font per line, so mixed-style lines style the
|
|
71
|
+
whole line instead of per-span; superscript, subscript, underline, strikeout, and
|
|
72
|
+
highlight markers are not emitted. Tables use a conservative text-grid detector:
|
|
73
|
+
aligned text tables are detected, drawn-rule-only tables are not.
|
|
74
|
+
- Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
|
|
75
75
|
impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
|
|
76
76
|
aggressively (a blocked engine simply contributes no results). User agents are a
|
|
77
77
|
fixed browser set plus ddgs's Android Google UA generator, not `fake_useragent`'s
|
|
78
78
|
database.
|
|
79
|
-
-
|
|
79
|
+
- Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
|
|
80
80
|
timeout whenever any engine timed out, so the timeout message is not masked by later
|
|
81
81
|
generic engine failures.
|
|
82
|
-
-
|
|
82
|
+
- Proxies: Studio routes through environment proxies; this port always connects
|
|
83
83
|
directly with DNS pinning (deliberately out of scope).
|
|
84
84
|
|
|
85
85
|
## Development
|
|
@@ -99,23 +99,23 @@ and `npm run test:smoke` runs only those.
|
|
|
99
99
|
|
|
100
100
|
The suite ports Unsloth Studio's own tests for these tools:
|
|
101
101
|
|
|
102
|
-
- `test/html-to-md.test.ts
|
|
102
|
+
- `test/html-to-md.test.ts`: hidden-element stripping and main-content scoping (from
|
|
103
103
|
`test_web_fetch_extraction.py`)
|
|
104
|
-
- `test/header-strip.test.ts
|
|
104
|
+
- `test/header-strip.test.ts`: the header link-density suite plus article-vs-main selection
|
|
105
105
|
and boilerplate cases (from `test_web_fetch_extraction.py`)
|
|
106
|
-
- `test/binary-guard.test.ts
|
|
106
|
+
- `test/binary-guard.test.ts`: the MIME/magic/charset/PDF matrix (from
|
|
107
107
|
`test_web_fetch_binary_guard.py`)
|
|
108
|
-
- `test/web-search-policy.test.ts
|
|
108
|
+
- `test/web-search-policy.test.ts`: policy filtering, overfetch, and failure messages
|
|
109
109
|
(from `test_web_access_policy.py`)
|
|
110
|
-
- `test/fetch-flow.test.ts
|
|
110
|
+
- `test/fetch-flow.test.ts`: GitHub README rewrite, deadline/cancellation, HTML sniffing
|
|
111
111
|
(from `test_web_fetch_extraction.py`; the fetch client is injected via seams)
|
|
112
|
-
- `test/engines.test.ts
|
|
112
|
+
- `test/engines.test.ts`: the ddgs engine port, normalizers, the XPath subset, the
|
|
113
113
|
aggregator, the ranker, and the Wikipedia engine with a stubbed fetch
|
|
114
|
-
- `test/pdf-parity.test.ts
|
|
114
|
+
- `test/pdf-parity.test.ts`: MuPDF engine capabilities, PDF 1.5 object streams,
|
|
115
115
|
ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
|
|
116
|
-
- `test/entities.test.ts
|
|
117
|
-
refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
|
|
118
|
-
- `test/smoke.test.ts
|
|
116
|
+
- `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
|
|
117
|
+
legacy refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
|
|
118
|
+
- `test/smoke.test.ts`: live network checks against real hosts
|
|
119
119
|
|
|
120
120
|
The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
|
|
121
121
|
with fakes, mirroring how the Studio suite monkeypatches `_validate_and_resolve_host`
|
|
@@ -125,4 +125,4 @@ and `build_opener`.
|
|
|
125
125
|
|
|
126
126
|
The ported logic derives from Unsloth Studio
|
|
127
127
|
([AGPL-3.0-only](https://github.com/unslothai/unsloth/blob/main/studio/LICENSE.AGPL-3.0)), so this
|
|
128
|
-
package is released under the same
|
|
128
|
+
package is released under the same AGPL-3.0-only license.
|
package/engines.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { randomBytes } from "node:crypto";
|
|
2
2
|
import { decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
|
|
3
3
|
import type { AttrDict } from "./html-to-md.ts";
|
|
4
|
+
import { randomUserAgent } from "./user-agents.ts";
|
|
4
5
|
export class EmptySweepError extends Error {
|
|
5
6
|
constructor() {
|
|
6
7
|
super("No results found");
|
|
@@ -110,6 +111,32 @@ interface XStep {
|
|
|
110
111
|
preds: Pred[];
|
|
111
112
|
terminal?: "text" | string;
|
|
112
113
|
}
|
|
114
|
+
function parsePredicateBlocks(input: string, start: number): { preds: Pred[]; next: number } {
|
|
115
|
+
const preds: Pred[] = [];
|
|
116
|
+
let pos = start;
|
|
117
|
+
while (pos < input.length && input[pos] === "[") {
|
|
118
|
+
const innerStart = pos + 1;
|
|
119
|
+
let depth = 1;
|
|
120
|
+
let quote: string | null = null;
|
|
121
|
+
let j = innerStart;
|
|
122
|
+
while (j < input.length && depth) {
|
|
123
|
+
const c = input[j];
|
|
124
|
+
if (quote !== null) {
|
|
125
|
+
if (c === quote) quote = null;
|
|
126
|
+
} else if (c === "'" || c === '"') {
|
|
127
|
+
quote = c;
|
|
128
|
+
} else if (c === "[") {
|
|
129
|
+
depth++;
|
|
130
|
+
} else if (c === "]") {
|
|
131
|
+
depth--;
|
|
132
|
+
}
|
|
133
|
+
j++;
|
|
134
|
+
}
|
|
135
|
+
preds.push(parsePredExpr(input.slice(innerStart, j - 1)));
|
|
136
|
+
pos = j;
|
|
137
|
+
}
|
|
138
|
+
return { preds, next: pos };
|
|
139
|
+
}
|
|
113
140
|
|
|
114
141
|
function parsePredExpr(input: string): Pred {
|
|
115
142
|
let pos = 0;
|
|
@@ -173,35 +200,10 @@ function parsePredExpr(input: string): Pred {
|
|
|
173
200
|
return { op: "desc", tag: name };
|
|
174
201
|
}
|
|
175
202
|
const name = word();
|
|
176
|
-
const preds =
|
|
203
|
+
const { preds, next } = parsePredicateBlocks(input, pos);
|
|
204
|
+
pos = next;
|
|
177
205
|
return { op: "child", tag: name, preds };
|
|
178
206
|
};
|
|
179
|
-
const parsePredBlocks = (): Pred[] => {
|
|
180
|
-
const preds: Pred[] = [];
|
|
181
|
-
while (pos < input.length && input[pos] === "[") {
|
|
182
|
-
const start = pos + 1;
|
|
183
|
-
let depth = 1;
|
|
184
|
-
let quote: string | null = null;
|
|
185
|
-
let i = start;
|
|
186
|
-
while (i < input.length && depth) {
|
|
187
|
-
const c = input[i];
|
|
188
|
-
if (quote !== null) {
|
|
189
|
-
if (c === quote) quote = null;
|
|
190
|
-
} else if (c === "'" || c === '"') {
|
|
191
|
-
quote = c;
|
|
192
|
-
} else if (c === "[") {
|
|
193
|
-
depth++;
|
|
194
|
-
} else if (c === "]") {
|
|
195
|
-
depth--;
|
|
196
|
-
}
|
|
197
|
-
i++;
|
|
198
|
-
}
|
|
199
|
-
const inner = input.slice(start, i - 1);
|
|
200
|
-
preds.push(parsePredExpr(inner));
|
|
201
|
-
pos = i;
|
|
202
|
-
}
|
|
203
|
-
return preds;
|
|
204
|
-
};
|
|
205
207
|
const parseAnd = (): Pred => {
|
|
206
208
|
let left = atom();
|
|
207
209
|
while (true) {
|
|
@@ -274,28 +276,8 @@ function parsePath(expr: string): XStep[] {
|
|
|
274
276
|
if (!m) break;
|
|
275
277
|
const name = m[0];
|
|
276
278
|
i += m[0].length;
|
|
277
|
-
const preds
|
|
278
|
-
|
|
279
|
-
const start = i + 1;
|
|
280
|
-
let depth = 1;
|
|
281
|
-
let quote: string | null = null;
|
|
282
|
-
let j = start;
|
|
283
|
-
while (j < expr.length && depth) {
|
|
284
|
-
const c = expr[j];
|
|
285
|
-
if (quote !== null) {
|
|
286
|
-
if (c === quote) quote = null;
|
|
287
|
-
} else if (c === "'" || c === '"') {
|
|
288
|
-
quote = c;
|
|
289
|
-
} else if (c === "[") {
|
|
290
|
-
depth++;
|
|
291
|
-
} else if (c === "]") {
|
|
292
|
-
depth--;
|
|
293
|
-
}
|
|
294
|
-
j++;
|
|
295
|
-
}
|
|
296
|
-
preds.push(parsePredExpr(expr.slice(start, j - 1)));
|
|
297
|
-
i = j;
|
|
298
|
-
}
|
|
279
|
+
const { preds, next } = parsePredicateBlocks(expr, i);
|
|
280
|
+
i = next;
|
|
299
281
|
steps.push({ axis, name, preds });
|
|
300
282
|
}
|
|
301
283
|
return steps;
|
|
@@ -409,19 +391,6 @@ export function extractResults(
|
|
|
409
391
|
return results;
|
|
410
392
|
}
|
|
411
393
|
|
|
412
|
-
const USER_AGENTS = [
|
|
413
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
414
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
415
|
-
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
416
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
417
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
418
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
|
|
419
|
-
];
|
|
420
|
-
|
|
421
|
-
function randomUserAgent(): string {
|
|
422
|
-
return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
|
|
423
|
-
}
|
|
424
|
-
|
|
425
394
|
function googleUserAgent(): string {
|
|
426
395
|
const devices: [string, string, number, number][] = [
|
|
427
396
|
["5.0", "SM-G900P Build/LRX21T", 39, 60],
|
|
@@ -503,6 +472,28 @@ async function httpPost(
|
|
|
503
472
|
return httpFetch(url, { ...options, method: "POST", body: new URLSearchParams(data).toString() });
|
|
504
473
|
}
|
|
505
474
|
|
|
475
|
+
const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
|
|
476
|
+
|
|
477
|
+
async function readBodyCapped(response: Response): Promise<string | null> {
|
|
478
|
+
const declared = Number(response.headers.get("content-length") ?? "0");
|
|
479
|
+
if (declared > MAX_ENGINE_RESPONSE_BYTES) return null;
|
|
480
|
+
if (!response.body) return "";
|
|
481
|
+
const reader = response.body.getReader();
|
|
482
|
+
const chunks: Uint8Array[] = [];
|
|
483
|
+
let total = 0;
|
|
484
|
+
while (true) {
|
|
485
|
+
const { done, value } = await reader.read();
|
|
486
|
+
if (done) break;
|
|
487
|
+
total += value.length;
|
|
488
|
+
if (total > MAX_ENGINE_RESPONSE_BYTES) {
|
|
489
|
+
await reader.cancel();
|
|
490
|
+
return null;
|
|
491
|
+
}
|
|
492
|
+
chunks.push(value);
|
|
493
|
+
}
|
|
494
|
+
return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
|
|
495
|
+
}
|
|
496
|
+
|
|
506
497
|
async function httpFetch(
|
|
507
498
|
url: string,
|
|
508
499
|
options: {
|
|
@@ -536,13 +527,18 @@ async function httpFetch(
|
|
|
536
527
|
signal: AbortSignal.any(signals),
|
|
537
528
|
});
|
|
538
529
|
} catch (err) {
|
|
539
|
-
if (err instanceof DOMException && err.name === "TimeoutError")
|
|
540
|
-
|
|
541
|
-
}
|
|
530
|
+
if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
|
|
531
|
+
if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
|
|
542
532
|
throw err;
|
|
543
533
|
}
|
|
544
534
|
if (response.status !== 200) return null;
|
|
545
|
-
|
|
535
|
+
try {
|
|
536
|
+
return await readBodyCapped(response);
|
|
537
|
+
} catch (err) {
|
|
538
|
+
if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
|
|
539
|
+
if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
|
|
540
|
+
throw err;
|
|
541
|
+
}
|
|
546
542
|
}
|
|
547
543
|
|
|
548
544
|
const DUCKDUCKGO: Engine = {
|
|
@@ -823,6 +819,7 @@ export async function autoTextSearch(
|
|
|
823
819
|
const aggregator = new ResultsAggregator();
|
|
824
820
|
const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
|
|
825
821
|
let timedOut = false;
|
|
822
|
+
let cancelled = false;
|
|
826
823
|
const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
|
|
827
824
|
const maxWorkers = Math.min(uniqueProviders, Math.ceil(maxResults / 10) + 1);
|
|
828
825
|
let i = 0;
|
|
@@ -835,7 +832,8 @@ export async function autoTextSearch(
|
|
|
835
832
|
seenProviders.add(engine.provider);
|
|
836
833
|
}
|
|
837
834
|
} catch (e) {
|
|
838
|
-
if (e instanceof
|
|
835
|
+
if (e instanceof SearchCancelled) cancelled = true;
|
|
836
|
+
if (e instanceof SearchTimeoutError) timedOut = true;
|
|
839
837
|
}
|
|
840
838
|
};
|
|
841
839
|
while (i < engines.length) {
|
|
@@ -843,11 +841,13 @@ export async function autoTextSearch(
|
|
|
843
841
|
const engine = engines[i++];
|
|
844
842
|
if (seenProviders.has(engine.provider)) continue;
|
|
845
843
|
pending.push(run(engine));
|
|
846
|
-
if (pending.length >= maxWorkers
|
|
844
|
+
if (pending.length >= maxWorkers) {
|
|
847
845
|
await Promise.allSettled(pending);
|
|
848
846
|
pending = [];
|
|
849
847
|
}
|
|
850
848
|
}
|
|
849
|
+
await Promise.allSettled(pending);
|
|
850
|
+
if (cancelled) throw new SearchCancelled();
|
|
851
851
|
const results = rankResults(aggregator.extractDicts(), query);
|
|
852
852
|
if (results.length) return results.slice(0, maxResults);
|
|
853
853
|
if (timedOut) throw new SearchTimeoutError();
|
package/index.ts
CHANGED
|
@@ -62,12 +62,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
62
62
|
name: "web_fetch",
|
|
63
63
|
label: "Web Fetch",
|
|
64
64
|
description:
|
|
65
|
-
"Fetch a URL and return readable text
|
|
66
|
-
"
|
|
67
|
-
"stripping
|
|
68
|
-
"
|
|
69
|
-
"
|
|
70
|
-
"
|
|
65
|
+
"Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
|
|
66
|
+
"main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
|
|
67
|
+
"stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
|
|
68
|
+
"README API, so the README is returned instead of the repo page's UI chrome. " +
|
|
69
|
+
"Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
|
|
70
|
+
"is capped.",
|
|
71
71
|
promptSnippet: "Fetch a web page and return readable text content",
|
|
72
72
|
parameters: WebFetchParams,
|
|
73
73
|
async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.4",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -30,6 +30,7 @@
|
|
|
30
30
|
"engines.ts",
|
|
31
31
|
"entities.ts",
|
|
32
32
|
"pdf.ts",
|
|
33
|
+
"user-agents.ts",
|
|
33
34
|
"README.md",
|
|
34
35
|
"LICENSE"
|
|
35
36
|
],
|
package/pdf.ts
CHANGED
|
@@ -527,7 +527,7 @@ function assemblePages(
|
|
|
527
527
|
return "";
|
|
528
528
|
}
|
|
529
529
|
if (pageLimitReached) {
|
|
530
|
-
text += `\n\n... (PDF extraction
|
|
530
|
+
text += `\n\n... (PDF extraction is capped at ${MAX_WEB_PDF_PAGES} pages)`;
|
|
531
531
|
}
|
|
532
532
|
return text;
|
|
533
533
|
}
|
package/user-agents.ts
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
const USER_AGENTS = [
|
|
2
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
3
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
4
|
+
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
5
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
6
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
7
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
|
|
8
|
+
];
|
|
9
|
+
|
|
10
|
+
export function randomUserAgent(): string {
|
|
11
|
+
return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
|
|
12
|
+
}
|
package/web-access.ts
CHANGED
|
@@ -173,46 +173,46 @@ export function checkUrlAccess(
|
|
|
173
173
|
policy: WebsitePolicy | null,
|
|
174
174
|
): [boolean, string, string] {
|
|
175
175
|
if (typeof url !== "string" || !url.trim()) {
|
|
176
|
-
return [false, "Blocked: URL is empty.", ""];
|
|
176
|
+
return [false, "Blocked: the URL is empty.", ""];
|
|
177
177
|
}
|
|
178
178
|
const candidate = url.trim();
|
|
179
179
|
if (
|
|
180
180
|
Array.from(candidate).some((char) => /\s/.test(char) || char.charCodeAt(0) < 32) ||
|
|
181
181
|
candidate.includes("\\")
|
|
182
182
|
) {
|
|
183
|
-
return [false, "Blocked: URL contains invalid characters.", ""];
|
|
183
|
+
return [false, "Blocked: the URL contains invalid characters.", ""];
|
|
184
184
|
}
|
|
185
185
|
let parsed: URL;
|
|
186
186
|
try {
|
|
187
187
|
parsed = new URL(candidate);
|
|
188
188
|
} catch {
|
|
189
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
189
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
190
190
|
}
|
|
191
191
|
const scheme = parsed.protocol.replace(/:$/, "").toLowerCase();
|
|
192
192
|
if (scheme !== "http" && scheme !== "https") {
|
|
193
193
|
return [false, "Blocked: only http/https URLs are allowed.", ""];
|
|
194
194
|
}
|
|
195
195
|
if (parsed.username || parsed.password || parsed.hostname.includes("%")) {
|
|
196
|
-
return [false, "Blocked:
|
|
196
|
+
return [false, "Blocked: URLs with credentials or encoded hostnames are not allowed.", ""];
|
|
197
197
|
}
|
|
198
198
|
if (!parsed.hostname) {
|
|
199
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
199
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
200
200
|
}
|
|
201
201
|
try {
|
|
202
202
|
if (parsed.port && !(PORT_RE.test(parsed.port) && Number(parsed.port) >= 1 && Number(parsed.port) <= 65535)) {
|
|
203
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
203
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
204
204
|
}
|
|
205
205
|
} catch {
|
|
206
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
206
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
207
207
|
}
|
|
208
208
|
let hostname: string;
|
|
209
209
|
try {
|
|
210
210
|
hostname = normalizeDomain(parsed.hostname);
|
|
211
211
|
} catch {
|
|
212
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
212
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
213
213
|
}
|
|
214
214
|
if (!hostnameAllowed(hostname, policy)) {
|
|
215
|
-
return [false, `Blocked: website access policy disallows ${hostname}.`, hostname];
|
|
215
|
+
return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
|
|
216
216
|
}
|
|
217
217
|
return [true, "", hostname];
|
|
218
218
|
}
|
|
@@ -368,6 +368,8 @@ export function isPublicIp(ip: string): boolean {
|
|
|
368
368
|
if (lower.startsWith("2001:db8")) return false;
|
|
369
369
|
if (lower.startsWith("64:ff9b:")) return false;
|
|
370
370
|
if (lower.startsWith("2001:10:")) return false;
|
|
371
|
+
if (lower.startsWith("2002:")) return false;
|
|
372
|
+
if (lower.startsWith("2001:0:") || lower.startsWith("2001::")) return false;
|
|
371
373
|
const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(lower);
|
|
372
374
|
if (mapped) return isPublicIp(mapped[1]);
|
|
373
375
|
if (lower.startsWith("::ffff:")) return false;
|
package/web-fetch.ts
CHANGED
|
@@ -10,21 +10,13 @@ import {
|
|
|
10
10
|
type WebsitePolicy,
|
|
11
11
|
} from "./web-access.ts";
|
|
12
12
|
import { htmlToMarkdown } from "./html-to-md.ts";
|
|
13
|
+
import { INVALID_CHARREFS } from "./entities.ts";
|
|
13
14
|
import { extractPdfText, PdfParseError } from "./pdf.ts";
|
|
15
|
+
import { randomUserAgent } from "./user-agents.ts";
|
|
14
16
|
|
|
15
|
-
const MIN_PAGE_CHARS = 2000;
|
|
16
17
|
const MAX_FETCH_BYTES = 512 * 1024;
|
|
17
18
|
const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
|
|
18
|
-
const
|
|
19
|
-
|
|
20
|
-
const USER_AGENTS = [
|
|
21
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
22
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
23
|
-
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
24
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
25
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
26
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
|
|
27
|
-
];
|
|
19
|
+
const MAX_REQUESTS = 5;
|
|
28
20
|
|
|
29
21
|
const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
|
|
30
22
|
const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
|
|
@@ -106,14 +98,17 @@ const ASCII_TEXT_BYTES = new Set<number>([
|
|
|
106
98
|
0x1b,
|
|
107
99
|
]);
|
|
108
100
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
101
|
+
export class FetchCancelledError extends Error {
|
|
102
|
+
constructor() {
|
|
103
|
+
super("cancelled");
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export class FetchTimeoutError extends Error {
|
|
108
|
+
constructor() {
|
|
109
|
+
super("timed out");
|
|
110
|
+
}
|
|
111
|
+
}
|
|
117
112
|
|
|
118
113
|
export interface FetchPageOptions {
|
|
119
114
|
timeoutMs?: number;
|
|
@@ -154,6 +149,8 @@ export interface HopOptions {
|
|
|
154
149
|
maxBytes: number;
|
|
155
150
|
maxPdfBytes: number;
|
|
156
151
|
inactivityMs: number;
|
|
152
|
+
deadlineMs?: number;
|
|
153
|
+
nowMs?: () => number;
|
|
157
154
|
signal?: AbortSignal;
|
|
158
155
|
}
|
|
159
156
|
|
|
@@ -348,8 +345,8 @@ function decodeSingleByte(bytes: Buffer, cp1252: boolean): string {
|
|
|
348
345
|
for (const byte of bytes) {
|
|
349
346
|
if (byte < 0x80) {
|
|
350
347
|
out += String.fromCharCode(byte);
|
|
351
|
-
} else if (cp1252 && byte in
|
|
352
|
-
out +=
|
|
348
|
+
} else if (cp1252 && byte in INVALID_CHARREFS) {
|
|
349
|
+
out += INVALID_CHARREFS[byte];
|
|
353
350
|
} else {
|
|
354
351
|
out += String.fromCharCode(byte);
|
|
355
352
|
}
|
|
@@ -421,7 +418,7 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
|
|
|
421
418
|
}
|
|
422
419
|
for (const entry of addresses) {
|
|
423
420
|
if (!isPublicIp(entry.address)) {
|
|
424
|
-
return { ok: false, reason: `Blocked: refusing to fetch non-public address ${entry.address}.`, ip: "", family: 0 };
|
|
421
|
+
return { ok: false, reason: `Blocked: refusing to fetch the non-public address ${entry.address}.`, ip: "", family: 0 };
|
|
425
422
|
}
|
|
426
423
|
}
|
|
427
424
|
const first = addresses[0];
|
|
@@ -440,7 +437,7 @@ function fetchBudgetExceeded(
|
|
|
440
437
|
}
|
|
441
438
|
|
|
442
439
|
|
|
443
|
-
function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
440
|
+
export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
444
441
|
return new Promise((resolve, reject) => {
|
|
445
442
|
const url = opts.url;
|
|
446
443
|
const transport = url.protocol === "https:" ? https : http;
|
|
@@ -455,6 +452,13 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
455
452
|
lookup: (_host, _opts, callback) =>
|
|
456
453
|
callback(null, [{ address: opts.pinnedIp, family: opts.family }]),
|
|
457
454
|
};
|
|
455
|
+
let settled = false;
|
|
456
|
+
const settle = (action: () => void) => {
|
|
457
|
+
if (settled) return;
|
|
458
|
+
settled = true;
|
|
459
|
+
opts.signal?.removeEventListener("abort", onAbort);
|
|
460
|
+
action();
|
|
461
|
+
};
|
|
458
462
|
const request = transport.request(options, (res: IncomingMessage) => {
|
|
459
463
|
const chunks: Buffer[] = [];
|
|
460
464
|
let total = 0;
|
|
@@ -463,17 +467,27 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
463
467
|
let extendedForPdf = false;
|
|
464
468
|
const finish = (err: string | null, body: Buffer) => {
|
|
465
469
|
settle(() => {
|
|
466
|
-
if (err)
|
|
467
|
-
|
|
470
|
+
if (err) {
|
|
471
|
+
if (err === "cancelled") reject(new FetchCancelledError());
|
|
472
|
+
else if (err === "timed out") reject(new FetchTimeoutError());
|
|
473
|
+
else reject(new Error(err));
|
|
474
|
+
} else {
|
|
468
475
|
resolve({
|
|
469
476
|
status: res.statusCode ?? 0,
|
|
470
477
|
headers: res.headers as Record<string, string | string[] | undefined>,
|
|
471
478
|
body,
|
|
472
479
|
});
|
|
480
|
+
}
|
|
473
481
|
});
|
|
474
482
|
};
|
|
475
483
|
res.on("data", (chunk: Buffer) => {
|
|
476
484
|
if (settled) return;
|
|
485
|
+
const now = opts.nowMs ?? Date.now;
|
|
486
|
+
if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
|
|
487
|
+
res.destroy();
|
|
488
|
+
finish("timed out", Buffer.concat(chunks));
|
|
489
|
+
return;
|
|
490
|
+
}
|
|
477
491
|
if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
|
|
478
492
|
if (hasPdfMagic(Buffer.concat(chunks))) {
|
|
479
493
|
limit = opts.maxPdfBytes;
|
|
@@ -497,16 +511,9 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
497
511
|
res.on("end", () => finish(null, Buffer.concat(chunks)));
|
|
498
512
|
res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
|
|
499
513
|
});
|
|
500
|
-
const onAbort = () => request.destroy(new
|
|
514
|
+
const onAbort = () => request.destroy(new FetchCancelledError());
|
|
501
515
|
opts.signal?.addEventListener("abort", onAbort, { once: true });
|
|
502
|
-
|
|
503
|
-
const settle = (action: () => void) => {
|
|
504
|
-
if (settled) return;
|
|
505
|
-
settled = true;
|
|
506
|
-
opts.signal?.removeEventListener("abort", onAbort);
|
|
507
|
-
action();
|
|
508
|
-
};
|
|
509
|
-
request.on("timeout", () => request.destroy(new Error("timed out")));
|
|
516
|
+
request.on("timeout", () => request.destroy(new FetchTimeoutError()));
|
|
510
517
|
request.on("error", (err) => settle(() => reject(err)));
|
|
511
518
|
request.end();
|
|
512
519
|
});
|
|
@@ -541,9 +548,9 @@ export async function fetchUrlRaw(
|
|
|
541
548
|
let currentUrl = url;
|
|
542
549
|
let pinnedIp = resolved.ip;
|
|
543
550
|
let pinnedFamily = resolved.family;
|
|
544
|
-
const userAgent =
|
|
551
|
+
const userAgent = randomUserAgent();
|
|
545
552
|
|
|
546
|
-
for (let hop = 0; hop <
|
|
553
|
+
for (let hop = 0; hop < MAX_REQUESTS; hop++) {
|
|
547
554
|
budgetError = fetchBudgetExceeded(deadline, signal, now);
|
|
548
555
|
if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
|
|
549
556
|
const parsed = new URL(currentUrl);
|
|
@@ -566,9 +573,15 @@ export async function fetchUrlRaw(
|
|
|
566
573
|
maxBytes,
|
|
567
574
|
maxPdfBytes,
|
|
568
575
|
inactivityMs: inactivity,
|
|
576
|
+
deadlineMs: deadline,
|
|
577
|
+
nowMs: now,
|
|
569
578
|
signal,
|
|
570
579
|
});
|
|
571
580
|
} catch (err) {
|
|
581
|
+
if (err instanceof FetchCancelledError)
|
|
582
|
+
return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
|
|
583
|
+
if (err instanceof FetchTimeoutError)
|
|
584
|
+
return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
|
|
572
585
|
const message = err instanceof Error ? err.message : String(err);
|
|
573
586
|
if (message === "cancelled")
|
|
574
587
|
return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
|
|
@@ -590,12 +603,16 @@ export async function fetchUrlRaw(
|
|
|
590
603
|
const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
|
|
591
604
|
if (!location) {
|
|
592
605
|
return {
|
|
593
|
-
error: "Failed to fetch URL: redirect missing Location header.",
|
|
606
|
+
error: "Failed to fetch URL: the redirect is missing a Location header.",
|
|
594
607
|
body: "",
|
|
595
608
|
contentType: "",
|
|
596
609
|
};
|
|
597
610
|
}
|
|
598
|
-
|
|
611
|
+
try {
|
|
612
|
+
currentUrl = new URL(location, currentUrl).toString();
|
|
613
|
+
} catch {
|
|
614
|
+
return { error: "Failed to fetch URL: the redirect has an invalid Location.", body: "", contentType: "" };
|
|
615
|
+
}
|
|
599
616
|
const [redirectAllowed, redirectReason, redirectHost] = checkUrlAccess(
|
|
600
617
|
currentUrl,
|
|
601
618
|
policy,
|
package/web-search.ts
CHANGED
|
@@ -82,7 +82,7 @@ export async function webSearch(
|
|
|
82
82
|
export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS): string {
|
|
83
83
|
if (exc instanceof SearchCancelled) return "Search cancelled.";
|
|
84
84
|
if (exc instanceof SearchTimeoutError) {
|
|
85
|
-
return `Search failed: the search engines did not respond within ${Math.round(timeoutMs / 1000)}
|
|
85
|
+
return `Search failed: the search engines did not respond within ${Math.round(timeoutMs / 1000)} seconds.`;
|
|
86
86
|
}
|
|
87
87
|
if (exc instanceof EmptySweepError || (exc instanceof Error && exc.message.includes("No results found"))) {
|
|
88
88
|
return EMPTY_SEARCH_RESULTS[0];
|
|
@@ -100,7 +100,7 @@ export function formatSearchResults(results: SearchResult[]): string {
|
|
|
100
100
|
const text = parts.join("\n\n---\n\n");
|
|
101
101
|
return (
|
|
102
102
|
text +
|
|
103
|
-
"\n\n---\n\
|
|
103
|
+
"\n\n---\n\nThese are only short snippets. " +
|
|
104
104
|
'To get the full page content, call web_search with the url parameter (e.g. {"url": "<URL>"}).'
|
|
105
105
|
);
|
|
106
106
|
}
|