pi-unsloth-webtools 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -26,7 +26,7 @@ Mirrors Unsloth Studio's `web_search` tool:
26
26
 
27
27
  - Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
28
28
  (duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
29
- the same provider-deduplication, href-dedupe aggregator with frequency ordering, and the
29
+ the same provider deduplication, href-dedupe aggregator with frequency ordering, and the
30
30
  same `SimpleFilterRanker` re-ranking. Formats results identically: `Title:` / `URL:` /
31
31
  `Snippet:` blocks separated by `---`, ending with the hint to pass `{"url": "<URL>"}` to
32
32
  read a full page.
@@ -45,7 +45,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
45
45
  validation and fetch.
46
46
  - GitHub repo root pages are rewritten to the unauthenticated README API
47
47
  (`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
48
- - Up to 5 redirect hops, each re-validated and re-resolved against the same rules.
48
+ - Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
49
49
  - 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
50
50
  (`signal` cancels mid-flight).
51
51
  - PDF text extraction via the official MuPDF.js engine (the same C library pymupdf wraps):
@@ -67,19 +67,19 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
67
67
 
68
68
  ## Known differences from Studio
69
69
 
70
- - **PDF styling**: MuPDF.js exposes one font per line, so mixed-style lines style the
71
- whole line instead of per-span; superscript/subscript/underline/strikeout/highlight
72
- markers are not emitted. Tables use a conservative text-grid detector — aligned text
73
- tables are detected, drawn-rule-only tables are not.
74
- - **Search engines**: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
70
+ - PDF styling: MuPDF.js exposes one font per line, so mixed-style lines style the
71
+ whole line instead of per-span; superscript, subscript, underline, strikeout, and
72
+ highlight markers are not emitted. Tables use a conservative text-grid detector:
73
+ aligned text tables are detected, drawn-rule-only tables are not.
74
+ - Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
75
75
  impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
76
76
  aggressively (a blocked engine simply contributes no results). User agents are a
77
77
  fixed browser set plus ddgs's Android Google UA generator, not `fake_useragent`'s
78
78
  database.
79
- - **Empty sweeps**: ddgs 9.14.4 raises the last engine exception; this port reports a
79
+ - Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
80
80
  timeout whenever any engine timed out, so the timeout message is not masked by later
81
81
  generic engine failures.
82
- - **Proxies**: Studio routes through environment proxies; this port always connects
82
+ - Proxies: Studio routes through environment proxies; this port always connects
83
83
  directly with DNS pinning (deliberately out of scope).
84
84
 
85
85
  ## Development
@@ -99,23 +99,23 @@ and `npm run test:smoke` runs only those.
99
99
 
100
100
  The suite ports Unsloth Studio's own tests for these tools:
101
101
 
102
- - `test/html-to-md.test.ts` — hidden-element stripping and main-content scoping (from
102
+ - `test/html-to-md.test.ts`: hidden-element stripping and main-content scoping (from
103
103
  `test_web_fetch_extraction.py`)
104
- - `test/header-strip.test.ts` — the header link-density suite plus article-vs-main selection
104
+ - `test/header-strip.test.ts`: the header link-density suite plus article-vs-main selection
105
105
  and boilerplate cases (from `test_web_fetch_extraction.py`)
106
- - `test/binary-guard.test.ts` — the MIME/magic/charset/PDF matrix (from
106
+ - `test/binary-guard.test.ts`: the MIME/magic/charset/PDF matrix (from
107
107
  `test_web_fetch_binary_guard.py`)
108
- - `test/web-search-policy.test.ts` — policy filtering, overfetch, and failure messages
108
+ - `test/web-search-policy.test.ts`: policy filtering, overfetch, and failure messages
109
109
  (from `test_web_access_policy.py`)
110
- - `test/fetch-flow.test.ts` — GitHub README rewrite, deadline/cancellation, HTML sniffing
110
+ - `test/fetch-flow.test.ts`: GitHub README rewrite, deadline/cancellation, HTML sniffing
111
111
  (from `test_web_fetch_extraction.py`; the fetch client is injected via seams)
112
- - `test/engines.test.ts` — the ddgs engine port: normalizers, the XPath subset, the
112
+ - `test/engines.test.ts`: the ddgs engine port, normalizers, the XPath subset, the
113
113
  aggregator, the ranker, and the Wikipedia engine with a stubbed fetch
114
- - `test/pdf-parity.test.ts` — MuPDF engine capabilities: PDF 1.5 object streams,
114
+ - `test/pdf-parity.test.ts`: MuPDF engine capabilities, PDF 1.5 object streams,
115
115
  ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
116
- - `test/entities.test.ts` — `decodeHtmlEntities` parity with CPython `html.unescape`: legacy
117
- refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
118
- - `test/smoke.test.ts` — live network checks against real hosts
116
+ - `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
117
+ legacy refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
118
+ - `test/smoke.test.ts`: live network checks against real hosts
119
119
 
120
120
  The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
121
121
  with fakes, mirroring how the Studio suite monkeypatches `_validate_and_resolve_host`
@@ -125,4 +125,4 @@ and `build_opener`.
125
125
 
126
126
  The ported logic derives from Unsloth Studio
127
127
  ([AGPL-3.0-only](https://github.com/unslothai/unsloth/blob/main/studio/LICENSE.AGPL-3.0)), so this
128
- package is released under the same **AGPL-3.0-only** license.
128
+ package is released under the same AGPL-3.0-only license.
package/engines.ts CHANGED
@@ -1,6 +1,7 @@
1
1
  import { randomBytes } from "node:crypto";
2
2
  import { decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
3
3
  import type { AttrDict } from "./html-to-md.ts";
4
+ import { randomUserAgent } from "./user-agents.ts";
4
5
  export class EmptySweepError extends Error {
5
6
  constructor() {
6
7
  super("No results found");
@@ -110,6 +111,32 @@ interface XStep {
110
111
  preds: Pred[];
111
112
  terminal?: "text" | string;
112
113
  }
114
+ function parsePredicateBlocks(input: string, start: number): { preds: Pred[]; next: number } {
115
+ const preds: Pred[] = [];
116
+ let pos = start;
117
+ while (pos < input.length && input[pos] === "[") {
118
+ const innerStart = pos + 1;
119
+ let depth = 1;
120
+ let quote: string | null = null;
121
+ let j = innerStart;
122
+ while (j < input.length && depth) {
123
+ const c = input[j];
124
+ if (quote !== null) {
125
+ if (c === quote) quote = null;
126
+ } else if (c === "'" || c === '"') {
127
+ quote = c;
128
+ } else if (c === "[") {
129
+ depth++;
130
+ } else if (c === "]") {
131
+ depth--;
132
+ }
133
+ j++;
134
+ }
135
+ preds.push(parsePredExpr(input.slice(innerStart, j - 1)));
136
+ pos = j;
137
+ }
138
+ return { preds, next: pos };
139
+ }
113
140
 
114
141
  function parsePredExpr(input: string): Pred {
115
142
  let pos = 0;
@@ -173,35 +200,10 @@ function parsePredExpr(input: string): Pred {
173
200
  return { op: "desc", tag: name };
174
201
  }
175
202
  const name = word();
176
- const preds = parsePredBlocks();
203
+ const { preds, next } = parsePredicateBlocks(input, pos);
204
+ pos = next;
177
205
  return { op: "child", tag: name, preds };
178
206
  };
179
- const parsePredBlocks = (): Pred[] => {
180
- const preds: Pred[] = [];
181
- while (pos < input.length && input[pos] === "[") {
182
- const start = pos + 1;
183
- let depth = 1;
184
- let quote: string | null = null;
185
- let i = start;
186
- while (i < input.length && depth) {
187
- const c = input[i];
188
- if (quote !== null) {
189
- if (c === quote) quote = null;
190
- } else if (c === "'" || c === '"') {
191
- quote = c;
192
- } else if (c === "[") {
193
- depth++;
194
- } else if (c === "]") {
195
- depth--;
196
- }
197
- i++;
198
- }
199
- const inner = input.slice(start, i - 1);
200
- preds.push(parsePredExpr(inner));
201
- pos = i;
202
- }
203
- return preds;
204
- };
205
207
  const parseAnd = (): Pred => {
206
208
  let left = atom();
207
209
  while (true) {
@@ -274,28 +276,8 @@ function parsePath(expr: string): XStep[] {
274
276
  if (!m) break;
275
277
  const name = m[0];
276
278
  i += m[0].length;
277
- const preds: Pred[] = [];
278
- while (i < expr.length && expr[i] === "[") {
279
- const start = i + 1;
280
- let depth = 1;
281
- let quote: string | null = null;
282
- let j = start;
283
- while (j < expr.length && depth) {
284
- const c = expr[j];
285
- if (quote !== null) {
286
- if (c === quote) quote = null;
287
- } else if (c === "'" || c === '"') {
288
- quote = c;
289
- } else if (c === "[") {
290
- depth++;
291
- } else if (c === "]") {
292
- depth--;
293
- }
294
- j++;
295
- }
296
- preds.push(parsePredExpr(expr.slice(start, j - 1)));
297
- i = j;
298
- }
279
+ const { preds, next } = parsePredicateBlocks(expr, i);
280
+ i = next;
299
281
  steps.push({ axis, name, preds });
300
282
  }
301
283
  return steps;
@@ -409,19 +391,6 @@ export function extractResults(
409
391
  return results;
410
392
  }
411
393
 
412
- const USER_AGENTS = [
413
- "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
414
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
415
- "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
416
- "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
417
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
418
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
419
- ];
420
-
421
- function randomUserAgent(): string {
422
- return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
423
- }
424
-
425
394
  function googleUserAgent(): string {
426
395
  const devices: [string, string, number, number][] = [
427
396
  ["5.0", "SM-G900P Build/LRX21T", 39, 60],
@@ -503,6 +472,28 @@ async function httpPost(
503
472
  return httpFetch(url, { ...options, method: "POST", body: new URLSearchParams(data).toString() });
504
473
  }
505
474
 
475
+ const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
476
+
477
+ async function readBodyCapped(response: Response): Promise<string | null> {
478
+ const declared = Number(response.headers.get("content-length") ?? "0");
479
+ if (declared > MAX_ENGINE_RESPONSE_BYTES) return null;
480
+ if (!response.body) return "";
481
+ const reader = response.body.getReader();
482
+ const chunks: Uint8Array[] = [];
483
+ let total = 0;
484
+ while (true) {
485
+ const { done, value } = await reader.read();
486
+ if (done) break;
487
+ total += value.length;
488
+ if (total > MAX_ENGINE_RESPONSE_BYTES) {
489
+ await reader.cancel();
490
+ return null;
491
+ }
492
+ chunks.push(value);
493
+ }
494
+ return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
495
+ }
496
+
506
497
  async function httpFetch(
507
498
  url: string,
508
499
  options: {
@@ -536,13 +527,18 @@ async function httpFetch(
536
527
  signal: AbortSignal.any(signals),
537
528
  });
538
529
  } catch (err) {
539
- if (err instanceof DOMException && err.name === "TimeoutError") {
540
- throw new Error("timed out");
541
- }
530
+ if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
531
+ if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
542
532
  throw err;
543
533
  }
544
534
  if (response.status !== 200) return null;
545
- return response.text();
535
+ try {
536
+ return await readBodyCapped(response);
537
+ } catch (err) {
538
+ if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
539
+ if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
540
+ throw err;
541
+ }
546
542
  }
547
543
 
548
544
  const DUCKDUCKGO: Engine = {
@@ -823,6 +819,7 @@ export async function autoTextSearch(
823
819
  const aggregator = new ResultsAggregator();
824
820
  const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
825
821
  let timedOut = false;
822
+ let cancelled = false;
826
823
  const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
827
824
  const maxWorkers = Math.min(uniqueProviders, Math.ceil(maxResults / 10) + 1);
828
825
  let i = 0;
@@ -835,7 +832,8 @@ export async function autoTextSearch(
835
832
  seenProviders.add(engine.provider);
836
833
  }
837
834
  } catch (e) {
838
- if (e instanceof Error && e.message.includes("timed out")) timedOut = true;
835
+ if (e instanceof SearchCancelled) cancelled = true;
836
+ if (e instanceof SearchTimeoutError) timedOut = true;
839
837
  }
840
838
  };
841
839
  while (i < engines.length) {
@@ -843,11 +841,13 @@ export async function autoTextSearch(
843
841
  const engine = engines[i++];
844
842
  if (seenProviders.has(engine.provider)) continue;
845
843
  pending.push(run(engine));
846
- if (pending.length >= maxWorkers || i >= maxWorkers) {
844
+ if (pending.length >= maxWorkers) {
847
845
  await Promise.allSettled(pending);
848
846
  pending = [];
849
847
  }
850
848
  }
849
+ await Promise.allSettled(pending);
850
+ if (cancelled) throw new SearchCancelled();
851
851
  const results = rankResults(aggregator.extractDicts(), query);
852
852
  if (results.length) return results.slice(0, maxResults);
853
853
  if (timedOut) throw new SearchTimeoutError();
package/index.ts CHANGED
@@ -62,12 +62,12 @@ export default function (pi: ExtensionAPI) {
62
62
  name: "web_fetch",
63
63
  label: "Web Fetch",
64
64
  description:
65
- "Fetch a URL and return readable text content. HTML responses are converted to Markdown " +
66
- "with a main-content heuristic (article/main scoping, hidden-element and boilerplate " +
67
- "stripping); non-HTML text responses are returned as-is. GitHub repo root pages are " +
68
- "rewritten to the README API so the model reads the README instead of the repo page's UI " +
69
- "chrome. Blocks private/loopback/link-local targets (SSRF protection) and caps the " +
70
- "download size.",
65
+ "Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
66
+ "main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
67
+ "stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
68
+ "README API, so the README is returned instead of the repo page's UI chrome. " +
69
+ "Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
70
+ "is capped.",
71
71
  promptSnippet: "Fetch a web page and return readable text content",
72
72
  parameters: WebFetchParams,
73
73
  async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-unsloth-webtools",
3
- "version": "0.2.2",
3
+ "version": "0.2.4",
4
4
  "type": "module",
5
5
  "description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
6
6
  "main": "index.ts",
@@ -30,6 +30,7 @@
30
30
  "engines.ts",
31
31
  "entities.ts",
32
32
  "pdf.ts",
33
+ "user-agents.ts",
33
34
  "README.md",
34
35
  "LICENSE"
35
36
  ],
package/pdf.ts CHANGED
@@ -527,7 +527,7 @@ function assemblePages(
527
527
  return "";
528
528
  }
529
529
  if (pageLimitReached) {
530
- text += `\n\n... (PDF extraction page processing capped at ${MAX_WEB_PDF_PAGES} pages)`;
530
+ text += `\n\n... (PDF extraction is capped at ${MAX_WEB_PDF_PAGES} pages)`;
531
531
  }
532
532
  return text;
533
533
  }
package/user-agents.ts ADDED
@@ -0,0 +1,12 @@
1
+ const USER_AGENTS = [
2
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
3
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
4
+ "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
5
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
6
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
7
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
8
+ ];
9
+
10
+ export function randomUserAgent(): string {
11
+ return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
12
+ }
package/web-access.ts CHANGED
@@ -173,46 +173,46 @@ export function checkUrlAccess(
173
173
  policy: WebsitePolicy | null,
174
174
  ): [boolean, string, string] {
175
175
  if (typeof url !== "string" || !url.trim()) {
176
- return [false, "Blocked: URL is empty.", ""];
176
+ return [false, "Blocked: the URL is empty.", ""];
177
177
  }
178
178
  const candidate = url.trim();
179
179
  if (
180
180
  Array.from(candidate).some((char) => /\s/.test(char) || char.charCodeAt(0) < 32) ||
181
181
  candidate.includes("\\")
182
182
  ) {
183
- return [false, "Blocked: URL contains invalid characters.", ""];
183
+ return [false, "Blocked: the URL contains invalid characters.", ""];
184
184
  }
185
185
  let parsed: URL;
186
186
  try {
187
187
  parsed = new URL(candidate);
188
188
  } catch {
189
- return [false, "Blocked: URL has an invalid hostname or port.", ""];
189
+ return [false, "Blocked: the URL has an invalid hostname or port.", ""];
190
190
  }
191
191
  const scheme = parsed.protocol.replace(/:$/, "").toLowerCase();
192
192
  if (scheme !== "http" && scheme !== "https") {
193
193
  return [false, "Blocked: only http/https URLs are allowed.", ""];
194
194
  }
195
195
  if (parsed.username || parsed.password || parsed.hostname.includes("%")) {
196
- return [false, "Blocked: URL credentials or encoded hostnames are not allowed.", ""];
196
+ return [false, "Blocked: URLs with credentials or encoded hostnames are not allowed.", ""];
197
197
  }
198
198
  if (!parsed.hostname) {
199
- return [false, "Blocked: URL has an invalid hostname or port.", ""];
199
+ return [false, "Blocked: the URL has an invalid hostname or port.", ""];
200
200
  }
201
201
  try {
202
202
  if (parsed.port && !(PORT_RE.test(parsed.port) && Number(parsed.port) >= 1 && Number(parsed.port) <= 65535)) {
203
- return [false, "Blocked: URL has an invalid hostname or port.", ""];
203
+ return [false, "Blocked: the URL has an invalid hostname or port.", ""];
204
204
  }
205
205
  } catch {
206
- return [false, "Blocked: URL has an invalid hostname or port.", ""];
206
+ return [false, "Blocked: the URL has an invalid hostname or port.", ""];
207
207
  }
208
208
  let hostname: string;
209
209
  try {
210
210
  hostname = normalizeDomain(parsed.hostname);
211
211
  } catch {
212
- return [false, "Blocked: URL has an invalid hostname or port.", ""];
212
+ return [false, "Blocked: the URL has an invalid hostname or port.", ""];
213
213
  }
214
214
  if (!hostnameAllowed(hostname, policy)) {
215
- return [false, `Blocked: website access policy disallows ${hostname}.`, hostname];
215
+ return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
216
216
  }
217
217
  return [true, "", hostname];
218
218
  }
@@ -368,6 +368,8 @@ export function isPublicIp(ip: string): boolean {
368
368
  if (lower.startsWith("2001:db8")) return false;
369
369
  if (lower.startsWith("64:ff9b:")) return false;
370
370
  if (lower.startsWith("2001:10:")) return false;
371
+ if (lower.startsWith("2002:")) return false;
372
+ if (lower.startsWith("2001:0:") || lower.startsWith("2001::")) return false;
371
373
  const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(lower);
372
374
  if (mapped) return isPublicIp(mapped[1]);
373
375
  if (lower.startsWith("::ffff:")) return false;
package/web-fetch.ts CHANGED
@@ -10,21 +10,13 @@ import {
10
10
  type WebsitePolicy,
11
11
  } from "./web-access.ts";
12
12
  import { htmlToMarkdown } from "./html-to-md.ts";
13
+ import { INVALID_CHARREFS } from "./entities.ts";
13
14
  import { extractPdfText, PdfParseError } from "./pdf.ts";
15
+ import { randomUserAgent } from "./user-agents.ts";
14
16
 
15
- const MIN_PAGE_CHARS = 2000;
16
17
  const MAX_FETCH_BYTES = 512 * 1024;
17
18
  const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
18
- const MAX_REDIRECTS = 5;
19
-
20
- const USER_AGENTS = [
21
- "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
22
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
23
- "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
24
- "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
25
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
26
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
27
- ];
19
+ const MAX_REQUESTS = 5;
28
20
 
29
21
  const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
30
22
  const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
@@ -106,14 +98,17 @@ const ASCII_TEXT_BYTES = new Set<number>([
106
98
  0x1b,
107
99
  ]);
108
100
 
109
- const CP1252_HIGH: Record<number, number> = {
110
- 0x80: 0x20ac, 0x82: 0x201a, 0x83: 0x0192, 0x84: 0x201e, 0x85: 0x2026,
111
- 0x86: 0x2020, 0x87: 0x2021, 0x88: 0x02c6, 0x89: 0x2030, 0x8a: 0x0160,
112
- 0x8b: 0x2039, 0x8c: 0x0152, 0x8e: 0x017d, 0x91: 0x2018, 0x92: 0x2019,
113
- 0x93: 0x201c, 0x94: 0x201d, 0x95: 0x2022, 0x96: 0x2013, 0x97: 0x2014,
114
- 0x98: 0x02dc, 0x99: 0x2122, 0x9a: 0x0161, 0x9b: 0x203a, 0x9c: 0x0153,
115
- 0x9e: 0x017e, 0x9f: 0x0178,
116
- };
101
+ export class FetchCancelledError extends Error {
102
+ constructor() {
103
+ super("cancelled");
104
+ }
105
+ }
106
+
107
+ export class FetchTimeoutError extends Error {
108
+ constructor() {
109
+ super("timed out");
110
+ }
111
+ }
117
112
 
118
113
  export interface FetchPageOptions {
119
114
  timeoutMs?: number;
@@ -154,6 +149,8 @@ export interface HopOptions {
154
149
  maxBytes: number;
155
150
  maxPdfBytes: number;
156
151
  inactivityMs: number;
152
+ deadlineMs?: number;
153
+ nowMs?: () => number;
157
154
  signal?: AbortSignal;
158
155
  }
159
156
 
@@ -348,8 +345,8 @@ function decodeSingleByte(bytes: Buffer, cp1252: boolean): string {
348
345
  for (const byte of bytes) {
349
346
  if (byte < 0x80) {
350
347
  out += String.fromCharCode(byte);
351
- } else if (cp1252 && byte in CP1252_HIGH) {
352
- out += String.fromCodePoint(CP1252_HIGH[byte]);
348
+ } else if (cp1252 && byte in INVALID_CHARREFS) {
349
+ out += INVALID_CHARREFS[byte];
353
350
  } else {
354
351
  out += String.fromCharCode(byte);
355
352
  }
@@ -421,7 +418,7 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
421
418
  }
422
419
  for (const entry of addresses) {
423
420
  if (!isPublicIp(entry.address)) {
424
- return { ok: false, reason: `Blocked: refusing to fetch non-public address ${entry.address}.`, ip: "", family: 0 };
421
+ return { ok: false, reason: `Blocked: refusing to fetch the non-public address ${entry.address}.`, ip: "", family: 0 };
425
422
  }
426
423
  }
427
424
  const first = addresses[0];
@@ -440,7 +437,7 @@ function fetchBudgetExceeded(
440
437
  }
441
438
 
442
439
 
443
- function requestHop(opts: HopOptions): Promise<HopResponse> {
440
+ export function requestHop(opts: HopOptions): Promise<HopResponse> {
444
441
  return new Promise((resolve, reject) => {
445
442
  const url = opts.url;
446
443
  const transport = url.protocol === "https:" ? https : http;
@@ -455,6 +452,13 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
455
452
  lookup: (_host, _opts, callback) =>
456
453
  callback(null, [{ address: opts.pinnedIp, family: opts.family }]),
457
454
  };
455
+ let settled = false;
456
+ const settle = (action: () => void) => {
457
+ if (settled) return;
458
+ settled = true;
459
+ opts.signal?.removeEventListener("abort", onAbort);
460
+ action();
461
+ };
458
462
  const request = transport.request(options, (res: IncomingMessage) => {
459
463
  const chunks: Buffer[] = [];
460
464
  let total = 0;
@@ -463,17 +467,27 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
463
467
  let extendedForPdf = false;
464
468
  const finish = (err: string | null, body: Buffer) => {
465
469
  settle(() => {
466
- if (err) reject(new Error(err));
467
- else
470
+ if (err) {
471
+ if (err === "cancelled") reject(new FetchCancelledError());
472
+ else if (err === "timed out") reject(new FetchTimeoutError());
473
+ else reject(new Error(err));
474
+ } else {
468
475
  resolve({
469
476
  status: res.statusCode ?? 0,
470
477
  headers: res.headers as Record<string, string | string[] | undefined>,
471
478
  body,
472
479
  });
480
+ }
473
481
  });
474
482
  };
475
483
  res.on("data", (chunk: Buffer) => {
476
484
  if (settled) return;
485
+ const now = opts.nowMs ?? Date.now;
486
+ if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
487
+ res.destroy();
488
+ finish("timed out", Buffer.concat(chunks));
489
+ return;
490
+ }
477
491
  if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
478
492
  if (hasPdfMagic(Buffer.concat(chunks))) {
479
493
  limit = opts.maxPdfBytes;
@@ -497,16 +511,9 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
497
511
  res.on("end", () => finish(null, Buffer.concat(chunks)));
498
512
  res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
499
513
  });
500
- const onAbort = () => request.destroy(new Error("cancelled"));
514
+ const onAbort = () => request.destroy(new FetchCancelledError());
501
515
  opts.signal?.addEventListener("abort", onAbort, { once: true });
502
- let settled = false;
503
- const settle = (action: () => void) => {
504
- if (settled) return;
505
- settled = true;
506
- opts.signal?.removeEventListener("abort", onAbort);
507
- action();
508
- };
509
- request.on("timeout", () => request.destroy(new Error("timed out")));
516
+ request.on("timeout", () => request.destroy(new FetchTimeoutError()));
510
517
  request.on("error", (err) => settle(() => reject(err)));
511
518
  request.end();
512
519
  });
@@ -541,9 +548,9 @@ export async function fetchUrlRaw(
541
548
  let currentUrl = url;
542
549
  let pinnedIp = resolved.ip;
543
550
  let pinnedFamily = resolved.family;
544
- const userAgent = USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
551
+ const userAgent = randomUserAgent();
545
552
 
546
- for (let hop = 0; hop < MAX_REDIRECTS; hop++) {
553
+ for (let hop = 0; hop < MAX_REQUESTS; hop++) {
547
554
  budgetError = fetchBudgetExceeded(deadline, signal, now);
548
555
  if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
549
556
  const parsed = new URL(currentUrl);
@@ -566,9 +573,15 @@ export async function fetchUrlRaw(
566
573
  maxBytes,
567
574
  maxPdfBytes,
568
575
  inactivityMs: inactivity,
576
+ deadlineMs: deadline,
577
+ nowMs: now,
569
578
  signal,
570
579
  });
571
580
  } catch (err) {
581
+ if (err instanceof FetchCancelledError)
582
+ return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
583
+ if (err instanceof FetchTimeoutError)
584
+ return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
572
585
  const message = err instanceof Error ? err.message : String(err);
573
586
  if (message === "cancelled")
574
587
  return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
@@ -590,12 +603,16 @@ export async function fetchUrlRaw(
590
603
  const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
591
604
  if (!location) {
592
605
  return {
593
- error: "Failed to fetch URL: redirect missing Location header.",
606
+ error: "Failed to fetch URL: the redirect is missing a Location header.",
594
607
  body: "",
595
608
  contentType: "",
596
609
  };
597
610
  }
598
- currentUrl = new URL(location, currentUrl).toString();
611
+ try {
612
+ currentUrl = new URL(location, currentUrl).toString();
613
+ } catch {
614
+ return { error: "Failed to fetch URL: the redirect has an invalid Location.", body: "", contentType: "" };
615
+ }
599
616
  const [redirectAllowed, redirectReason, redirectHost] = checkUrlAccess(
600
617
  currentUrl,
601
618
  policy,
package/web-search.ts CHANGED
@@ -82,7 +82,7 @@ export async function webSearch(
82
82
  export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS): string {
83
83
  if (exc instanceof SearchCancelled) return "Search cancelled.";
84
84
  if (exc instanceof SearchTimeoutError) {
85
- return `Search failed: the search engines did not respond within ${Math.round(timeoutMs / 1000)}s.`;
85
+ return `Search failed: the search engines did not respond within ${Math.round(timeoutMs / 1000)} seconds.`;
86
86
  }
87
87
  if (exc instanceof EmptySweepError || (exc instanceof Error && exc.message.includes("No results found"))) {
88
88
  return EMPTY_SEARCH_RESULTS[0];
@@ -100,7 +100,7 @@ export function formatSearchResults(results: SearchResult[]): string {
100
100
  const text = parts.join("\n\n---\n\n");
101
101
  return (
102
102
  text +
103
- "\n\n---\n\nIMPORTANT: These are only short snippets. " +
103
+ "\n\n---\n\nThese are only short snippets. " +
104
104
  'To get the full page content, call web_search with the url parameter (e.g. {"url": "<URL>"}).'
105
105
  );
106
106
  }