pi-unsloth-webtools 0.2.2 → 0.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -20
- package/engines.ts +33 -63
- package/index.ts +6 -6
- package/package.json +2 -1
- package/pdf.ts +1 -1
- package/user-agents.ts +12 -0
- package/web-access.ts +11 -9
- package/web-fetch.ts +11 -16
- package/web-search.ts +2 -2
package/README.md
CHANGED
|
@@ -26,7 +26,7 @@ Mirrors Unsloth Studio's `web_search` tool:
|
|
|
26
26
|
|
|
27
27
|
- Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
|
|
28
28
|
(duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
|
|
29
|
-
the same provider
|
|
29
|
+
the same provider deduplication, href-dedupe aggregator with frequency ordering, and the
|
|
30
30
|
same `SimpleFilterRanker` re-ranking. Formats results identically: `Title:` / `URL:` /
|
|
31
31
|
`Snippet:` blocks separated by `---`, ending with the hint to pass `{"url": "<URL>"}` to
|
|
32
32
|
read a full page.
|
|
@@ -45,7 +45,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
45
45
|
validation and fetch.
|
|
46
46
|
- GitHub repo root pages are rewritten to the unauthenticated README API
|
|
47
47
|
(`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
|
|
48
|
-
- Up to
|
|
48
|
+
- Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
|
|
49
49
|
- 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
|
|
50
50
|
(`signal` cancels mid-flight).
|
|
51
51
|
- PDF text extraction via the official MuPDF.js engine (the same C library pymupdf wraps):
|
|
@@ -67,19 +67,19 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
67
67
|
|
|
68
68
|
## Known differences from Studio
|
|
69
69
|
|
|
70
|
-
-
|
|
71
|
-
whole line instead of per-span; superscript
|
|
72
|
-
markers are not emitted. Tables use a conservative text-grid detector
|
|
73
|
-
tables are detected, drawn-rule-only tables are not.
|
|
74
|
-
-
|
|
70
|
+
- PDF styling: MuPDF.js exposes one font per line, so mixed-style lines style the
|
|
71
|
+
whole line instead of per-span; superscript, subscript, underline, strikeout, and
|
|
72
|
+
highlight markers are not emitted. Tables use a conservative text-grid detector:
|
|
73
|
+
aligned text tables are detected, drawn-rule-only tables are not.
|
|
74
|
+
- Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
|
|
75
75
|
impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
|
|
76
76
|
aggressively (a blocked engine simply contributes no results). User agents are a
|
|
77
77
|
fixed browser set plus ddgs's Android Google UA generator, not `fake_useragent`'s
|
|
78
78
|
database.
|
|
79
|
-
-
|
|
79
|
+
- Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
|
|
80
80
|
timeout whenever any engine timed out, so the timeout message is not masked by later
|
|
81
81
|
generic engine failures.
|
|
82
|
-
-
|
|
82
|
+
- Proxies: Studio routes through environment proxies; this port always connects
|
|
83
83
|
directly with DNS pinning (deliberately out of scope).
|
|
84
84
|
|
|
85
85
|
## Development
|
|
@@ -99,23 +99,23 @@ and `npm run test:smoke` runs only those.
|
|
|
99
99
|
|
|
100
100
|
The suite ports Unsloth Studio's own tests for these tools:
|
|
101
101
|
|
|
102
|
-
- `test/html-to-md.test.ts
|
|
102
|
+
- `test/html-to-md.test.ts`: hidden-element stripping and main-content scoping (from
|
|
103
103
|
`test_web_fetch_extraction.py`)
|
|
104
|
-
- `test/header-strip.test.ts
|
|
104
|
+
- `test/header-strip.test.ts`: the header link-density suite plus article-vs-main selection
|
|
105
105
|
and boilerplate cases (from `test_web_fetch_extraction.py`)
|
|
106
|
-
- `test/binary-guard.test.ts
|
|
106
|
+
- `test/binary-guard.test.ts`: the MIME/magic/charset/PDF matrix (from
|
|
107
107
|
`test_web_fetch_binary_guard.py`)
|
|
108
|
-
- `test/web-search-policy.test.ts
|
|
108
|
+
- `test/web-search-policy.test.ts`: policy filtering, overfetch, and failure messages
|
|
109
109
|
(from `test_web_access_policy.py`)
|
|
110
|
-
- `test/fetch-flow.test.ts
|
|
110
|
+
- `test/fetch-flow.test.ts`: GitHub README rewrite, deadline/cancellation, HTML sniffing
|
|
111
111
|
(from `test_web_fetch_extraction.py`; the fetch client is injected via seams)
|
|
112
|
-
- `test/engines.test.ts
|
|
112
|
+
- `test/engines.test.ts`: the ddgs engine port, normalizers, the XPath subset, the
|
|
113
113
|
aggregator, the ranker, and the Wikipedia engine with a stubbed fetch
|
|
114
|
-
- `test/pdf-parity.test.ts
|
|
114
|
+
- `test/pdf-parity.test.ts`: MuPDF engine capabilities, PDF 1.5 object streams,
|
|
115
115
|
ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
|
|
116
|
-
- `test/entities.test.ts
|
|
117
|
-
refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
|
|
118
|
-
- `test/smoke.test.ts
|
|
116
|
+
- `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
|
|
117
|
+
legacy refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
|
|
118
|
+
- `test/smoke.test.ts`: live network checks against real hosts
|
|
119
119
|
|
|
120
120
|
The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
|
|
121
121
|
with fakes, mirroring how the Studio suite monkeypatches `_validate_and_resolve_host`
|
|
@@ -125,4 +125,4 @@ and `build_opener`.
|
|
|
125
125
|
|
|
126
126
|
The ported logic derives from Unsloth Studio
|
|
127
127
|
([AGPL-3.0-only](https://github.com/unslothai/unsloth/blob/main/studio/LICENSE.AGPL-3.0)), so this
|
|
128
|
-
package is released under the same
|
|
128
|
+
package is released under the same AGPL-3.0-only license.
|
package/engines.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { randomBytes } from "node:crypto";
|
|
2
2
|
import { decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
|
|
3
3
|
import type { AttrDict } from "./html-to-md.ts";
|
|
4
|
+
import { randomUserAgent } from "./user-agents.ts";
|
|
4
5
|
export class EmptySweepError extends Error {
|
|
5
6
|
constructor() {
|
|
6
7
|
super("No results found");
|
|
@@ -110,6 +111,32 @@ interface XStep {
|
|
|
110
111
|
preds: Pred[];
|
|
111
112
|
terminal?: "text" | string;
|
|
112
113
|
}
|
|
114
|
+
function parsePredicateBlocks(input: string, start: number): { preds: Pred[]; next: number } {
|
|
115
|
+
const preds: Pred[] = [];
|
|
116
|
+
let pos = start;
|
|
117
|
+
while (pos < input.length && input[pos] === "[") {
|
|
118
|
+
const innerStart = pos + 1;
|
|
119
|
+
let depth = 1;
|
|
120
|
+
let quote: string | null = null;
|
|
121
|
+
let j = innerStart;
|
|
122
|
+
while (j < input.length && depth) {
|
|
123
|
+
const c = input[j];
|
|
124
|
+
if (quote !== null) {
|
|
125
|
+
if (c === quote) quote = null;
|
|
126
|
+
} else if (c === "'" || c === '"') {
|
|
127
|
+
quote = c;
|
|
128
|
+
} else if (c === "[") {
|
|
129
|
+
depth++;
|
|
130
|
+
} else if (c === "]") {
|
|
131
|
+
depth--;
|
|
132
|
+
}
|
|
133
|
+
j++;
|
|
134
|
+
}
|
|
135
|
+
preds.push(parsePredExpr(input.slice(innerStart, j - 1)));
|
|
136
|
+
pos = j;
|
|
137
|
+
}
|
|
138
|
+
return { preds, next: pos };
|
|
139
|
+
}
|
|
113
140
|
|
|
114
141
|
function parsePredExpr(input: string): Pred {
|
|
115
142
|
let pos = 0;
|
|
@@ -173,35 +200,10 @@ function parsePredExpr(input: string): Pred {
|
|
|
173
200
|
return { op: "desc", tag: name };
|
|
174
201
|
}
|
|
175
202
|
const name = word();
|
|
176
|
-
const preds =
|
|
203
|
+
const { preds, next } = parsePredicateBlocks(input, pos);
|
|
204
|
+
pos = next;
|
|
177
205
|
return { op: "child", tag: name, preds };
|
|
178
206
|
};
|
|
179
|
-
const parsePredBlocks = (): Pred[] => {
|
|
180
|
-
const preds: Pred[] = [];
|
|
181
|
-
while (pos < input.length && input[pos] === "[") {
|
|
182
|
-
const start = pos + 1;
|
|
183
|
-
let depth = 1;
|
|
184
|
-
let quote: string | null = null;
|
|
185
|
-
let i = start;
|
|
186
|
-
while (i < input.length && depth) {
|
|
187
|
-
const c = input[i];
|
|
188
|
-
if (quote !== null) {
|
|
189
|
-
if (c === quote) quote = null;
|
|
190
|
-
} else if (c === "'" || c === '"') {
|
|
191
|
-
quote = c;
|
|
192
|
-
} else if (c === "[") {
|
|
193
|
-
depth++;
|
|
194
|
-
} else if (c === "]") {
|
|
195
|
-
depth--;
|
|
196
|
-
}
|
|
197
|
-
i++;
|
|
198
|
-
}
|
|
199
|
-
const inner = input.slice(start, i - 1);
|
|
200
|
-
preds.push(parsePredExpr(inner));
|
|
201
|
-
pos = i;
|
|
202
|
-
}
|
|
203
|
-
return preds;
|
|
204
|
-
};
|
|
205
207
|
const parseAnd = (): Pred => {
|
|
206
208
|
let left = atom();
|
|
207
209
|
while (true) {
|
|
@@ -274,28 +276,8 @@ function parsePath(expr: string): XStep[] {
|
|
|
274
276
|
if (!m) break;
|
|
275
277
|
const name = m[0];
|
|
276
278
|
i += m[0].length;
|
|
277
|
-
const preds
|
|
278
|
-
|
|
279
|
-
const start = i + 1;
|
|
280
|
-
let depth = 1;
|
|
281
|
-
let quote: string | null = null;
|
|
282
|
-
let j = start;
|
|
283
|
-
while (j < expr.length && depth) {
|
|
284
|
-
const c = expr[j];
|
|
285
|
-
if (quote !== null) {
|
|
286
|
-
if (c === quote) quote = null;
|
|
287
|
-
} else if (c === "'" || c === '"') {
|
|
288
|
-
quote = c;
|
|
289
|
-
} else if (c === "[") {
|
|
290
|
-
depth++;
|
|
291
|
-
} else if (c === "]") {
|
|
292
|
-
depth--;
|
|
293
|
-
}
|
|
294
|
-
j++;
|
|
295
|
-
}
|
|
296
|
-
preds.push(parsePredExpr(expr.slice(start, j - 1)));
|
|
297
|
-
i = j;
|
|
298
|
-
}
|
|
279
|
+
const { preds, next } = parsePredicateBlocks(expr, i);
|
|
280
|
+
i = next;
|
|
299
281
|
steps.push({ axis, name, preds });
|
|
300
282
|
}
|
|
301
283
|
return steps;
|
|
@@ -409,19 +391,6 @@ export function extractResults(
|
|
|
409
391
|
return results;
|
|
410
392
|
}
|
|
411
393
|
|
|
412
|
-
const USER_AGENTS = [
|
|
413
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
414
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
415
|
-
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
416
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
417
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
418
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
|
|
419
|
-
];
|
|
420
|
-
|
|
421
|
-
function randomUserAgent(): string {
|
|
422
|
-
return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
|
|
423
|
-
}
|
|
424
|
-
|
|
425
394
|
function googleUserAgent(): string {
|
|
426
395
|
const devices: [string, string, number, number][] = [
|
|
427
396
|
["5.0", "SM-G900P Build/LRX21T", 39, 60],
|
|
@@ -843,11 +812,12 @@ export async function autoTextSearch(
|
|
|
843
812
|
const engine = engines[i++];
|
|
844
813
|
if (seenProviders.has(engine.provider)) continue;
|
|
845
814
|
pending.push(run(engine));
|
|
846
|
-
if (pending.length >= maxWorkers
|
|
815
|
+
if (pending.length >= maxWorkers) {
|
|
847
816
|
await Promise.allSettled(pending);
|
|
848
817
|
pending = [];
|
|
849
818
|
}
|
|
850
819
|
}
|
|
820
|
+
await Promise.allSettled(pending);
|
|
851
821
|
const results = rankResults(aggregator.extractDicts(), query);
|
|
852
822
|
if (results.length) return results.slice(0, maxResults);
|
|
853
823
|
if (timedOut) throw new SearchTimeoutError();
|
package/index.ts
CHANGED
|
@@ -62,12 +62,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
62
62
|
name: "web_fetch",
|
|
63
63
|
label: "Web Fetch",
|
|
64
64
|
description:
|
|
65
|
-
"Fetch a URL and return readable text
|
|
66
|
-
"
|
|
67
|
-
"stripping
|
|
68
|
-
"
|
|
69
|
-
"
|
|
70
|
-
"
|
|
65
|
+
"Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
|
|
66
|
+
"main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
|
|
67
|
+
"stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
|
|
68
|
+
"README API, so the README is returned instead of the repo page's UI chrome. " +
|
|
69
|
+
"Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
|
|
70
|
+
"is capped.",
|
|
71
71
|
promptSnippet: "Fetch a web page and return readable text content",
|
|
72
72
|
parameters: WebFetchParams,
|
|
73
73
|
async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.3",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -30,6 +30,7 @@
|
|
|
30
30
|
"engines.ts",
|
|
31
31
|
"entities.ts",
|
|
32
32
|
"pdf.ts",
|
|
33
|
+
"user-agents.ts",
|
|
33
34
|
"README.md",
|
|
34
35
|
"LICENSE"
|
|
35
36
|
],
|
package/pdf.ts
CHANGED
|
@@ -527,7 +527,7 @@ function assemblePages(
|
|
|
527
527
|
return "";
|
|
528
528
|
}
|
|
529
529
|
if (pageLimitReached) {
|
|
530
|
-
text += `\n\n... (PDF extraction
|
|
530
|
+
text += `\n\n... (PDF extraction is capped at ${MAX_WEB_PDF_PAGES} pages)`;
|
|
531
531
|
}
|
|
532
532
|
return text;
|
|
533
533
|
}
|
package/user-agents.ts
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
const USER_AGENTS = [
|
|
2
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
3
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
4
|
+
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
5
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
6
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
7
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
|
|
8
|
+
];
|
|
9
|
+
|
|
10
|
+
export function randomUserAgent(): string {
|
|
11
|
+
return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
|
|
12
|
+
}
|
package/web-access.ts
CHANGED
|
@@ -173,46 +173,46 @@ export function checkUrlAccess(
|
|
|
173
173
|
policy: WebsitePolicy | null,
|
|
174
174
|
): [boolean, string, string] {
|
|
175
175
|
if (typeof url !== "string" || !url.trim()) {
|
|
176
|
-
return [false, "Blocked: URL is empty.", ""];
|
|
176
|
+
return [false, "Blocked: the URL is empty.", ""];
|
|
177
177
|
}
|
|
178
178
|
const candidate = url.trim();
|
|
179
179
|
if (
|
|
180
180
|
Array.from(candidate).some((char) => /\s/.test(char) || char.charCodeAt(0) < 32) ||
|
|
181
181
|
candidate.includes("\\")
|
|
182
182
|
) {
|
|
183
|
-
return [false, "Blocked: URL contains invalid characters.", ""];
|
|
183
|
+
return [false, "Blocked: the URL contains invalid characters.", ""];
|
|
184
184
|
}
|
|
185
185
|
let parsed: URL;
|
|
186
186
|
try {
|
|
187
187
|
parsed = new URL(candidate);
|
|
188
188
|
} catch {
|
|
189
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
189
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
190
190
|
}
|
|
191
191
|
const scheme = parsed.protocol.replace(/:$/, "").toLowerCase();
|
|
192
192
|
if (scheme !== "http" && scheme !== "https") {
|
|
193
193
|
return [false, "Blocked: only http/https URLs are allowed.", ""];
|
|
194
194
|
}
|
|
195
195
|
if (parsed.username || parsed.password || parsed.hostname.includes("%")) {
|
|
196
|
-
return [false, "Blocked:
|
|
196
|
+
return [false, "Blocked: URLs with credentials or encoded hostnames are not allowed.", ""];
|
|
197
197
|
}
|
|
198
198
|
if (!parsed.hostname) {
|
|
199
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
199
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
200
200
|
}
|
|
201
201
|
try {
|
|
202
202
|
if (parsed.port && !(PORT_RE.test(parsed.port) && Number(parsed.port) >= 1 && Number(parsed.port) <= 65535)) {
|
|
203
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
203
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
204
204
|
}
|
|
205
205
|
} catch {
|
|
206
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
206
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
207
207
|
}
|
|
208
208
|
let hostname: string;
|
|
209
209
|
try {
|
|
210
210
|
hostname = normalizeDomain(parsed.hostname);
|
|
211
211
|
} catch {
|
|
212
|
-
return [false, "Blocked: URL has an invalid hostname or port.", ""];
|
|
212
|
+
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
213
213
|
}
|
|
214
214
|
if (!hostnameAllowed(hostname, policy)) {
|
|
215
|
-
return [false, `Blocked: website access policy disallows ${hostname}.`, hostname];
|
|
215
|
+
return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
|
|
216
216
|
}
|
|
217
217
|
return [true, "", hostname];
|
|
218
218
|
}
|
|
@@ -368,6 +368,8 @@ export function isPublicIp(ip: string): boolean {
|
|
|
368
368
|
if (lower.startsWith("2001:db8")) return false;
|
|
369
369
|
if (lower.startsWith("64:ff9b:")) return false;
|
|
370
370
|
if (lower.startsWith("2001:10:")) return false;
|
|
371
|
+
if (lower.startsWith("2002:")) return false;
|
|
372
|
+
if (lower.startsWith("2001:0:") || lower.startsWith("2001::")) return false;
|
|
371
373
|
const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(lower);
|
|
372
374
|
if (mapped) return isPublicIp(mapped[1]);
|
|
373
375
|
if (lower.startsWith("::ffff:")) return false;
|
package/web-fetch.ts
CHANGED
|
@@ -11,20 +11,11 @@ import {
|
|
|
11
11
|
} from "./web-access.ts";
|
|
12
12
|
import { htmlToMarkdown } from "./html-to-md.ts";
|
|
13
13
|
import { extractPdfText, PdfParseError } from "./pdf.ts";
|
|
14
|
+
import { randomUserAgent } from "./user-agents.ts";
|
|
14
15
|
|
|
15
|
-
const MIN_PAGE_CHARS = 2000;
|
|
16
16
|
const MAX_FETCH_BYTES = 512 * 1024;
|
|
17
17
|
const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
|
|
18
|
-
const
|
|
19
|
-
|
|
20
|
-
const USER_AGENTS = [
|
|
21
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
22
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
23
|
-
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
|
|
24
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
25
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:133.0) Gecko/20100101 Firefox/133.0",
|
|
26
|
-
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15",
|
|
27
|
-
];
|
|
18
|
+
const MAX_REQUESTS = 5;
|
|
28
19
|
|
|
29
20
|
const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
|
|
30
21
|
const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
|
|
@@ -421,7 +412,7 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
|
|
|
421
412
|
}
|
|
422
413
|
for (const entry of addresses) {
|
|
423
414
|
if (!isPublicIp(entry.address)) {
|
|
424
|
-
return { ok: false, reason: `Blocked: refusing to fetch non-public address ${entry.address}.`, ip: "", family: 0 };
|
|
415
|
+
return { ok: false, reason: `Blocked: refusing to fetch the non-public address ${entry.address}.`, ip: "", family: 0 };
|
|
425
416
|
}
|
|
426
417
|
}
|
|
427
418
|
const first = addresses[0];
|
|
@@ -541,9 +532,9 @@ export async function fetchUrlRaw(
|
|
|
541
532
|
let currentUrl = url;
|
|
542
533
|
let pinnedIp = resolved.ip;
|
|
543
534
|
let pinnedFamily = resolved.family;
|
|
544
|
-
const userAgent =
|
|
535
|
+
const userAgent = randomUserAgent();
|
|
545
536
|
|
|
546
|
-
for (let hop = 0; hop <
|
|
537
|
+
for (let hop = 0; hop < MAX_REQUESTS; hop++) {
|
|
547
538
|
budgetError = fetchBudgetExceeded(deadline, signal, now);
|
|
548
539
|
if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
|
|
549
540
|
const parsed = new URL(currentUrl);
|
|
@@ -590,12 +581,16 @@ export async function fetchUrlRaw(
|
|
|
590
581
|
const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
|
|
591
582
|
if (!location) {
|
|
592
583
|
return {
|
|
593
|
-
error: "Failed to fetch URL: redirect missing Location header.",
|
|
584
|
+
error: "Failed to fetch URL: the redirect is missing a Location header.",
|
|
594
585
|
body: "",
|
|
595
586
|
contentType: "",
|
|
596
587
|
};
|
|
597
588
|
}
|
|
598
|
-
|
|
589
|
+
try {
|
|
590
|
+
currentUrl = new URL(location, currentUrl).toString();
|
|
591
|
+
} catch {
|
|
592
|
+
return { error: "Failed to fetch URL: the redirect has an invalid Location.", body: "", contentType: "" };
|
|
593
|
+
}
|
|
599
594
|
const [redirectAllowed, redirectReason, redirectHost] = checkUrlAccess(
|
|
600
595
|
currentUrl,
|
|
601
596
|
policy,
|
package/web-search.ts
CHANGED
|
@@ -82,7 +82,7 @@ export async function webSearch(
|
|
|
82
82
|
export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS): string {
|
|
83
83
|
if (exc instanceof SearchCancelled) return "Search cancelled.";
|
|
84
84
|
if (exc instanceof SearchTimeoutError) {
|
|
85
|
-
return `Search failed: the search engines did not respond within ${Math.round(timeoutMs / 1000)}
|
|
85
|
+
return `Search failed: the search engines did not respond within ${Math.round(timeoutMs / 1000)} seconds.`;
|
|
86
86
|
}
|
|
87
87
|
if (exc instanceof EmptySweepError || (exc instanceof Error && exc.message.includes("No results found"))) {
|
|
88
88
|
return EMPTY_SEARCH_RESULTS[0];
|
|
@@ -100,7 +100,7 @@ export function formatSearchResults(results: SearchResult[]): string {
|
|
|
100
100
|
const text = parts.join("\n\n---\n\n");
|
|
101
101
|
return (
|
|
102
102
|
text +
|
|
103
|
-
"\n\n---\n\
|
|
103
|
+
"\n\n---\n\nThese are only short snippets. " +
|
|
104
104
|
'To get the full page content, call web_search with the url parameter (e.g. {"url": "<URL>"}).'
|
|
105
105
|
);
|
|
106
106
|
}
|