pi-unsloth-webtools 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -4
- package/engines.ts +3 -3
- package/html-to-md.ts +32 -3
- package/package.json +4 -3
- package/web-fetch.ts +103 -24
- package/web-search.ts +9 -15
- package/ROADMAP.md +0 -77
package/README.md
CHANGED
|
@@ -53,8 +53,8 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
53
53
|
pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
|
|
54
54
|
with Studio's corrupted/incomplete fallback to plain text.
|
|
55
55
|
- Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
|
|
56
|
-
decoding (declared charset, BOM sniffing for UTF-8/16/32,
|
|
57
|
-
single-byte pages).
|
|
56
|
+
decoding (declared charset, BOM sniffing for UTF-8/16/32, `<meta charset>` sniffing for
|
|
57
|
+
CJK and Windows/ISO encodings, cp1252 rescue for mislabeled single-byte pages).
|
|
58
58
|
- HTML → Markdown conversion ported from Studio's dependency-free `_html_to_md.py`: headings,
|
|
59
59
|
links, emphasis, lists, tables, blockquotes, code fences, entity decoding; hidden-element
|
|
60
60
|
stripping (`hidden`, `aria-hidden`, inline styles); `<article>`/`<main>` main-content scoping
|
|
@@ -65,8 +65,22 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
65
65
|
- HTML entity decoding replicates CPython's `html.unescape` (full 2,231-entry HTML5 table,
|
|
66
66
|
longest-prefix rule, Windows-1252 numeric mappings), matching Studio byte-for-byte.
|
|
67
67
|
|
|
68
|
-
|
|
69
|
-
|
|
68
|
+
## Known differences from Studio
|
|
69
|
+
|
|
70
|
+
- **PDF styling**: MuPDF.js exposes one font per line, so mixed-style lines style the
|
|
71
|
+
whole line instead of per-span; superscript/subscript/underline/strikeout/highlight
|
|
72
|
+
markers are not emitted. Tables use a conservative text-grid detector — aligned text
|
|
73
|
+
tables are detected, drawn-rule-only tables are not.
|
|
74
|
+
- **Search engines**: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
|
|
75
|
+
impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
|
|
76
|
+
aggressively (a blocked engine simply contributes no results). User agents are a
|
|
77
|
+
fixed browser set plus ddgs's Android Google UA generator, not `fake_useragent`'s
|
|
78
|
+
database.
|
|
79
|
+
- **Empty sweeps**: ddgs 9.14.4 raises the last engine exception; this port reports a
|
|
80
|
+
timeout whenever any engine timed out, so the timeout message is not masked by later
|
|
81
|
+
generic engine failures.
|
|
82
|
+
- **Proxies**: Studio routes through environment proxies; this port always connects
|
|
83
|
+
directly with DNS pinning (deliberately out of scope).
|
|
70
84
|
|
|
71
85
|
## Development
|
|
72
86
|
|
|
@@ -74,10 +88,15 @@ fingerprinting, proxies).
|
|
|
74
88
|
npm install
|
|
75
89
|
npm run typecheck
|
|
76
90
|
npm test
|
|
91
|
+
npm run test:unit
|
|
92
|
+
npm run test:smoke
|
|
77
93
|
```
|
|
78
94
|
|
|
79
95
|
## Tests
|
|
80
96
|
|
|
97
|
+
`npm test` runs the full suite. `npm run test:unit` skips the live-network smoke tests,
|
|
98
|
+
and `npm run test:smoke` runs only those.
|
|
99
|
+
|
|
81
100
|
The suite ports Unsloth Studio's own tests for these tools:
|
|
82
101
|
|
|
83
102
|
- `test/html-to-md.test.ts` — hidden-element stripping and main-content scoping (from
|
|
@@ -94,6 +113,8 @@ The suite ports Unsloth Studio's own tests for these tools:
|
|
|
94
113
|
aggregator, the ranker, and the Wikipedia engine with a stubbed fetch
|
|
95
114
|
- `test/pdf-parity.test.ts` — MuPDF engine capabilities: PDF 1.5 object streams,
|
|
96
115
|
ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
|
|
116
|
+
- `test/entities.test.ts` — `decodeHtmlEntities` parity with CPython `html.unescape`: legacy
|
|
117
|
+
refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
|
|
97
118
|
- `test/smoke.test.ts` — live network checks against real hosts
|
|
98
119
|
|
|
99
120
|
The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
|
package/engines.ts
CHANGED
|
@@ -822,7 +822,7 @@ export async function autoTextSearch(
|
|
|
822
822
|
const seenProviders = new Set<string>();
|
|
823
823
|
const aggregator = new ResultsAggregator();
|
|
824
824
|
const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
|
|
825
|
-
let
|
|
825
|
+
let timedOut = false;
|
|
826
826
|
const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
|
|
827
827
|
const maxWorkers = Math.min(uniqueProviders, Math.ceil(maxResults / 10) + 1);
|
|
828
828
|
let i = 0;
|
|
@@ -835,7 +835,7 @@ export async function autoTextSearch(
|
|
|
835
835
|
seenProviders.add(engine.provider);
|
|
836
836
|
}
|
|
837
837
|
} catch (e) {
|
|
838
|
-
|
|
838
|
+
if (e instanceof Error && e.message.includes("timed out")) timedOut = true;
|
|
839
839
|
}
|
|
840
840
|
};
|
|
841
841
|
while (i < engines.length) {
|
|
@@ -850,6 +850,6 @@ export async function autoTextSearch(
|
|
|
850
850
|
}
|
|
851
851
|
const results = rankResults(aggregator.extractDicts(), query);
|
|
852
852
|
if (results.length) return results.slice(0, maxResults);
|
|
853
|
-
if (
|
|
853
|
+
if (timedOut) throw new SearchTimeoutError();
|
|
854
854
|
throw new EmptySweepError();
|
|
855
855
|
}
|
package/html-to-md.ts
CHANGED
|
@@ -197,7 +197,7 @@ export function decodeHtmlEntities(text: string): string {
|
|
|
197
197
|
return text.replace(CHARREF_RE, (whole, s: string) => {
|
|
198
198
|
if (s[0] === "#") {
|
|
199
199
|
const hex = s[1] === "x" || s[1] === "X";
|
|
200
|
-
const num = parseInt(s.slice(2).replace(/;+$/, ""), hex ? 16 : 10);
|
|
200
|
+
const num = parseInt(s.slice(hex ? 2 : 1).replace(/;+$/, ""), hex ? 16 : 10);
|
|
201
201
|
const mapped = INVALID_CHARREFS[num];
|
|
202
202
|
if (mapped !== undefined) return mapped;
|
|
203
203
|
if ((num >= 0xd800 && num <= 0xdfff) || num > 0x10ffff) return "\ufffd";
|
|
@@ -227,6 +227,21 @@ interface HtmlHandlers {
|
|
|
227
227
|
const START_TAG_NAME_RE = /^[a-zA-Z][^\s/>]*/;
|
|
228
228
|
const ATTR_NAME_RE = /^[^\s=/>]+/;
|
|
229
229
|
|
|
230
|
+
const RAW_TEXT_TAGS = [
|
|
231
|
+
"script",
|
|
232
|
+
"style",
|
|
233
|
+
"textarea",
|
|
234
|
+
"title",
|
|
235
|
+
"xmp",
|
|
236
|
+
"iframe",
|
|
237
|
+
"noembed",
|
|
238
|
+
"noframes",
|
|
239
|
+
];
|
|
240
|
+
|
|
241
|
+
const RAW_TEXT_CLOSERS: Record<string, RegExp> = Object.fromEntries(
|
|
242
|
+
RAW_TEXT_TAGS.map((name) => [name, new RegExp(`</${name}\\s*>`, "i")]),
|
|
243
|
+
);
|
|
244
|
+
|
|
230
245
|
function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, boolean] {
|
|
231
246
|
const attrs: AttrDict = {};
|
|
232
247
|
while (pos < input.length) {
|
|
@@ -346,8 +361,22 @@ export function feedHtml(input: string, handlers: HtmlHandlers): void {
|
|
|
346
361
|
continue;
|
|
347
362
|
}
|
|
348
363
|
emitText(input.slice(textStart, i));
|
|
349
|
-
if (tag.kind === "start")
|
|
350
|
-
|
|
364
|
+
if (tag.kind === "start") {
|
|
365
|
+
const rawCloser = RAW_TEXT_CLOSERS[tag.name!];
|
|
366
|
+
if (rawCloser) {
|
|
367
|
+
const after = input.slice(tag.end);
|
|
368
|
+
const closeMatch = rawCloser.exec(after);
|
|
369
|
+
if (closeMatch) {
|
|
370
|
+
handlers.handleStartTag(tag.name!, tag.attrs!);
|
|
371
|
+
emitText(after.slice(0, closeMatch.index));
|
|
372
|
+
handlers.handleEndTag(tag.name!);
|
|
373
|
+
textStart = tag.end + closeMatch.index + closeMatch[0].length;
|
|
374
|
+
i = textStart;
|
|
375
|
+
continue;
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
handlers.handleStartTag(tag.name!, tag.attrs!);
|
|
379
|
+
} else if (tag.kind === "startend") handlers.handleStartEndTag(tag.name!, tag.attrs!);
|
|
351
380
|
else if (tag.kind === "end") handlers.handleEndTag(tag.name!);
|
|
352
381
|
textStart = tag.end;
|
|
353
382
|
i = tag.end;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.2",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -31,7 +31,6 @@
|
|
|
31
31
|
"entities.ts",
|
|
32
32
|
"pdf.ts",
|
|
33
33
|
"README.md",
|
|
34
|
-
"ROADMAP.md",
|
|
35
34
|
"LICENSE"
|
|
36
35
|
],
|
|
37
36
|
"pi": {
|
|
@@ -48,8 +47,10 @@
|
|
|
48
47
|
},
|
|
49
48
|
"scripts": {
|
|
50
49
|
"test": "vitest run",
|
|
50
|
+
"test:unit": "vitest run --exclude **/smoke.test.ts",
|
|
51
|
+
"test:smoke": "vitest run test/smoke.test.ts",
|
|
51
52
|
"typecheck": "tsc --noEmit",
|
|
52
|
-
"prepublishOnly": "npm run typecheck"
|
|
53
|
+
"prepublishOnly": "npm run typecheck && npm run test:unit"
|
|
53
54
|
},
|
|
54
55
|
"devDependencies": {
|
|
55
56
|
"@earendil-works/pi-coding-agent": "^0.84.0",
|
package/web-fetch.ts
CHANGED
|
@@ -237,6 +237,34 @@ function hasSingleByteTextEvidence(data: Buffer): boolean {
|
|
|
237
237
|
return ascii / data.length >= MIN_SINGLE_BYTE_ASCII_RATIO;
|
|
238
238
|
}
|
|
239
239
|
|
|
240
|
+
const CHARSET_ALIASES: Record<string, string> = {
|
|
241
|
+
gbk: "gbk",
|
|
242
|
+
gb2312: "gbk",
|
|
243
|
+
"gb-2312": "gbk",
|
|
244
|
+
cp936: "gbk",
|
|
245
|
+
"x-gbk": "gbk",
|
|
246
|
+
gb18030: "gb18030",
|
|
247
|
+
big5: "big5",
|
|
248
|
+
"big5-hkscs": "big5",
|
|
249
|
+
sjis: "shift_jis",
|
|
250
|
+
"x-sjis": "shift_jis",
|
|
251
|
+
cp932: "shift_jis",
|
|
252
|
+
"shift-jis": "shift_jis",
|
|
253
|
+
"euc-jp": "euc-jp",
|
|
254
|
+
"euc-kr": "euc-kr",
|
|
255
|
+
ksc5601: "euc-kr",
|
|
256
|
+
"ks_c_5601-1987": "euc-kr",
|
|
257
|
+
"ks_c_5601-1989": "euc-kr",
|
|
258
|
+
"iso-2022-jp": "iso-2022-jp",
|
|
259
|
+
"koi8-r": "koi8-r",
|
|
260
|
+
"koi8-u": "koi8-u",
|
|
261
|
+
cp866: "cp866",
|
|
262
|
+
"x-mac-cyrillic": "x-mac-cyrillic",
|
|
263
|
+
"windows-874": "windows-874",
|
|
264
|
+
cp874: "windows-874",
|
|
265
|
+
"tis-620": "tis-620",
|
|
266
|
+
};
|
|
267
|
+
|
|
240
268
|
function normalizeCharset(name: string): string | null {
|
|
241
269
|
const n = name.trim().replace(/["']/g, "").toLowerCase();
|
|
242
270
|
switch (n) {
|
|
@@ -270,8 +298,28 @@ function normalizeCharset(name: string): string | null {
|
|
|
270
298
|
case "utf-32be":
|
|
271
299
|
return "utf-32be";
|
|
272
300
|
default:
|
|
273
|
-
|
|
301
|
+
break;
|
|
302
|
+
}
|
|
303
|
+
const alias = CHARSET_ALIASES[n];
|
|
304
|
+
if (alias !== undefined) return alias;
|
|
305
|
+
if (/^windows-125[0-8]$/.test(n) || /^iso-8859-(?:[2-9]|1[0-6])$/.test(n)) return n;
|
|
306
|
+
return null;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
function sniffMetaCharset(bytes: Buffer): string | null {
|
|
310
|
+
const head = bytes.subarray(0, 1024).toString("latin1").toLowerCase();
|
|
311
|
+
const match =
|
|
312
|
+
/<meta\b[^>]*\bcharset\s*=\s*["']?\s*([a-z0-9_.\-]+)/.exec(head) ??
|
|
313
|
+
/<meta\b[^>]*\bhttp-equiv\s*=\s*["']?content-type["']?[^>]*\bcharset\s*=\s*["']?\s*([a-z0-9_.\-]+)/.exec(head);
|
|
314
|
+
return match ? normalizeCharset(match[1]) : null;
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
function sniffMetaCharsetForHtml(bytes: Buffer, contentType: string): string | null {
|
|
318
|
+
if (!contentType.includes("html")) {
|
|
319
|
+
const probe = bytes.subarray(0, 256).toString("latin1");
|
|
320
|
+
if (!looksLikeHtmlDocument(probe)) return null;
|
|
274
321
|
}
|
|
322
|
+
return sniffMetaCharset(bytes);
|
|
275
323
|
}
|
|
276
324
|
|
|
277
325
|
function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
|
|
@@ -328,11 +376,38 @@ function decodeWithCodec(bytes: Buffer, codec: string | null): string {
|
|
|
328
376
|
case "cp1252":
|
|
329
377
|
return decodeSingleByte(bytes, true);
|
|
330
378
|
case "iso8859-1":
|
|
331
|
-
default:
|
|
332
379
|
return decodeSingleByte(bytes, false);
|
|
380
|
+
default:
|
|
381
|
+
return decodeWithLabel(bytes, codec);
|
|
333
382
|
}
|
|
334
383
|
}
|
|
335
384
|
|
|
385
|
+
function decodeWithLabel(bytes: Buffer, label: string | null): string {
|
|
386
|
+
if (label === null) return decodeSingleByte(bytes, false);
|
|
387
|
+
if (label === "tis-620") return decodeTis620(bytes);
|
|
388
|
+
try {
|
|
389
|
+
return new TextDecoder(label, { fatal: false }).decode(bytes);
|
|
390
|
+
} catch {
|
|
391
|
+
return decodeSingleByte(bytes, false);
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
function decodeTis620(bytes: Buffer): string {
|
|
396
|
+
let out = "";
|
|
397
|
+
for (const byte of bytes) {
|
|
398
|
+
if (byte < 0x80) {
|
|
399
|
+
out += String.fromCharCode(byte);
|
|
400
|
+
} else if (byte >= 0xa1 && byte <= 0xfb) {
|
|
401
|
+
out += String.fromCodePoint(0x0e01 + byte - 0xa1);
|
|
402
|
+
} else if (byte === 0xa0) {
|
|
403
|
+
out += "\u00a0";
|
|
404
|
+
} else {
|
|
405
|
+
out += "\ufffd";
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
return out;
|
|
409
|
+
}
|
|
410
|
+
|
|
336
411
|
|
|
337
412
|
async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promise<ResolvedHost> {
|
|
338
413
|
let addresses: { address: string; family: number }[];
|
|
@@ -386,20 +461,19 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
386
461
|
const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
|
|
387
462
|
let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
|
|
388
463
|
let extendedForPdf = false;
|
|
389
|
-
let done = false;
|
|
390
464
|
const finish = (err: string | null, body: Buffer) => {
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
465
|
+
settle(() => {
|
|
466
|
+
if (err) reject(new Error(err));
|
|
467
|
+
else
|
|
468
|
+
resolve({
|
|
469
|
+
status: res.statusCode ?? 0,
|
|
470
|
+
headers: res.headers as Record<string, string | string[] | undefined>,
|
|
471
|
+
body,
|
|
472
|
+
});
|
|
473
|
+
});
|
|
400
474
|
};
|
|
401
475
|
res.on("data", (chunk: Buffer) => {
|
|
402
|
-
if (
|
|
476
|
+
if (settled) return;
|
|
403
477
|
if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
|
|
404
478
|
if (hasPdfMagic(Buffer.concat(chunks))) {
|
|
405
479
|
limit = opts.maxPdfBytes;
|
|
@@ -423,10 +497,17 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
423
497
|
res.on("end", () => finish(null, Buffer.concat(chunks)));
|
|
424
498
|
res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
|
|
425
499
|
});
|
|
426
|
-
request.on("timeout", () => request.destroy(new Error("timed out")));
|
|
427
|
-
request.on("error", (err) => reject(err));
|
|
428
500
|
const onAbort = () => request.destroy(new Error("cancelled"));
|
|
429
501
|
opts.signal?.addEventListener("abort", onAbort, { once: true });
|
|
502
|
+
let settled = false;
|
|
503
|
+
const settle = (action: () => void) => {
|
|
504
|
+
if (settled) return;
|
|
505
|
+
settled = true;
|
|
506
|
+
opts.signal?.removeEventListener("abort", onAbort);
|
|
507
|
+
action();
|
|
508
|
+
};
|
|
509
|
+
request.on("timeout", () => request.destroy(new Error("timed out")));
|
|
510
|
+
request.on("error", (err) => settle(() => reject(err)));
|
|
430
511
|
request.end();
|
|
431
512
|
});
|
|
432
513
|
}
|
|
@@ -471,6 +552,7 @@ export async function fetchUrlRaw(
|
|
|
471
552
|
"User-Agent": userAgent,
|
|
472
553
|
Host: hostHeader,
|
|
473
554
|
Accept: "text/html,application/xhtml+xml,text/plain;q=0.9,*/*;q=0.5",
|
|
555
|
+
"Accept-Encoding": "identity",
|
|
474
556
|
};
|
|
475
557
|
if (options.extraHeaders) Object.assign(headers, options.extraHeaders);
|
|
476
558
|
const inactivity = Math.max(1, deadline - now());
|
|
@@ -497,8 +579,9 @@ export async function fetchUrlRaw(
|
|
|
497
579
|
|
|
498
580
|
if (response.status >= 300 && response.status < 400) {
|
|
499
581
|
if (![301, 302, 303, 307, 308].includes(response.status)) {
|
|
582
|
+
const reason = http.STATUS_CODES[response.status] ?? "";
|
|
500
583
|
return {
|
|
501
|
-
error: `Failed to fetch URL: HTTP ${response.status} ${
|
|
584
|
+
error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
|
|
502
585
|
body: "",
|
|
503
586
|
contentType: "",
|
|
504
587
|
};
|
|
@@ -577,7 +660,10 @@ export async function fetchUrlRaw(
|
|
|
577
660
|
|
|
578
661
|
const declaredCodec = declaredCharset ? normalizeCharset(declaredCharset) : null;
|
|
579
662
|
const bomCodec = bomCodecFor(response.body);
|
|
580
|
-
const rawHtml = decodeWithCodec(
|
|
663
|
+
const rawHtml = decodeWithCodec(
|
|
664
|
+
response.body,
|
|
665
|
+
declaredCodec ?? bomCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
|
|
666
|
+
);
|
|
581
667
|
|
|
582
668
|
if (looksBinary(rawHtml)) {
|
|
583
669
|
let alt: string | null = null;
|
|
@@ -612,13 +698,6 @@ function bomCodecFor(bytes: Buffer): string | null {
|
|
|
612
698
|
return null;
|
|
613
699
|
}
|
|
614
700
|
|
|
615
|
-
function statusReason(status: number): string {
|
|
616
|
-
const reasons: Record<number, string> = {
|
|
617
|
-
301: "Moved Permanently", 302: "Found", 303: "See Other", 307: "Temporary Redirect", 308: "Permanent Redirect",
|
|
618
|
-
};
|
|
619
|
-
return reasons[status] ?? "";
|
|
620
|
-
}
|
|
621
|
-
|
|
622
701
|
export function truncatePageText(text: string, maxChars?: number): string {
|
|
623
702
|
if (!text) return "(page returned no readable text)";
|
|
624
703
|
if (typeof maxChars === "number" && maxChars > 0 && text.length > maxChars) {
|
package/web-search.ts
CHANGED
|
@@ -64,23 +64,17 @@ export async function webSearch(
|
|
|
64
64
|
const results = await client(effectiveQuery, wanted, signal);
|
|
65
65
|
if (signal?.aborted) return "Search cancelled.";
|
|
66
66
|
if (!results.length) return EMPTY_SEARCH_RESULTS[0];
|
|
67
|
-
const
|
|
67
|
+
const allowed: SearchResult[] = [];
|
|
68
68
|
for (const result of results) {
|
|
69
|
-
if (
|
|
69
|
+
if (allowed.length >= maxResults) break;
|
|
70
70
|
const href = String(result.href ?? "").trim();
|
|
71
71
|
if (href && !checkUrlAccess(href, policy)[0]) continue;
|
|
72
|
-
|
|
73
|
-
const snippet = String(result.body ?? "").replace(/\s+/g, " ");
|
|
74
|
-
parts.push(`Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`);
|
|
72
|
+
allowed.push(result);
|
|
75
73
|
}
|
|
76
|
-
if (!
|
|
77
|
-
|
|
78
|
-
return (
|
|
79
|
-
text +
|
|
80
|
-
"\n\n---\n\nIMPORTANT: These are only short snippets. " +
|
|
81
|
-
'To get the full page content, call web_search with the url parameter (e.g. {"url": "<URL>"}).'
|
|
82
|
-
);
|
|
74
|
+
if (!allowed.length) return EMPTY_SEARCH_RESULTS[1];
|
|
75
|
+
return formatSearchResults(allowed);
|
|
83
76
|
} catch (err) {
|
|
77
|
+
if (signal?.aborted) return "Search cancelled.";
|
|
84
78
|
return searchFailureMessage(err, timeoutMs);
|
|
85
79
|
}
|
|
86
80
|
}
|
|
@@ -98,9 +92,9 @@ export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS
|
|
|
98
92
|
|
|
99
93
|
export function formatSearchResults(results: SearchResult[]): string {
|
|
100
94
|
const parts = results.map((result) => {
|
|
101
|
-
const title = result.title.replace(/\s+/g, " ");
|
|
102
|
-
const href = result.href.trim();
|
|
103
|
-
const snippet = result.body.replace(/\s+/g, " ");
|
|
95
|
+
const title = String(result.title ?? "").replace(/\s+/g, " ");
|
|
96
|
+
const href = String(result.href ?? "").trim();
|
|
97
|
+
const snippet = String(result.body ?? "").replace(/\s+/g, " ");
|
|
104
98
|
return `Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`;
|
|
105
99
|
});
|
|
106
100
|
const text = parts.join("\n\n---\n\n");
|
package/ROADMAP.md
DELETED
|
@@ -1,77 +0,0 @@
|
|
|
1
|
-
# Roadmap
|
|
2
|
-
|
|
3
|
-
Tracking the remaining gaps between this port and Unsloth Studio's web tools, and the
|
|
4
|
-
planned work to close them.
|
|
5
|
-
|
|
6
|
-
## Implemented
|
|
7
|
-
|
|
8
|
-
### PDF text extraction (MuPDF engine)
|
|
9
|
-
|
|
10
|
-
`pdf.ts` uses the official **MuPDF.js** (`mupdf` npm package) — the same C engine that
|
|
11
|
-
PyMuPDF wraps — replacing the earlier minimal built-in extractor:
|
|
12
|
-
|
|
13
|
-
- Full xref handling: tables, cross-reference streams, and PDF 1.5+ object streams
|
|
14
|
-
- All standard stream filters (FlateDecode, ASCII85Decode, LZW, RunLength, DCT, JPX, ...)
|
|
15
|
-
- Font encodings and ToUnicode mapping (non-Latin text extracts correctly)
|
|
16
|
-
- Encryption detection via `needsPassword()` (reported as unreadable, matching pymupdf's
|
|
17
|
-
no-password behavior)
|
|
18
|
-
- The markdown layer replicates pymupdf4llm's algorithm: `IdentifyHeaders` font-size
|
|
19
|
-
heading detection, `get_raw_lines` line reconstruction (tolerance 3, 10% span-join
|
|
20
|
-
delta), `write_text` styling (bold/italic/mono, code fences, bullets, link
|
|
21
|
-
resolution with `%0x`-escaped URIs), and Studio's corrupted/incomplete fallback to
|
|
22
|
-
plain MuPDF text with the exact thresholds from `backend/core/rag/parsers.py`
|
|
23
|
-
- Table detection and pipe-markdown rendering matching pymupdf's `Table.to_markdown`
|
|
24
|
-
shape (`|header|`, `|---|`, detail rows, `Col{i}` fill for empty headers)
|
|
25
|
-
|
|
26
|
-
Known deltas vs pymupdf4llm:
|
|
27
|
-
|
|
28
|
-
- Span-level styling: MuPDF.js's structured-text JSON exposes one font per line, so
|
|
29
|
-
mixed-style lines (one bold word inside a body line) style the whole line instead of
|
|
30
|
-
per-span. Line-level styling matches for homogeneous lines.
|
|
31
|
-
- Superscript/subscript/underline/strikeout/highlight markers are not emitted (the
|
|
32
|
-
JSON does not expose char-level flags).
|
|
33
|
-
- Table detection is a conservative text-grid detector (column-start clustering with
|
|
34
|
-
a 5 pt tolerance, contiguous multi-row bands) instead of PyMuPDF's
|
|
35
|
-
`find_tables()` vector-graphics analysis. Aligned text tables are detected; tables
|
|
36
|
-
defined only by drawn rules without aligned text are not.
|
|
37
|
-
|
|
38
|
-
The minimal extractor remains as an automatic fallback when the `mupdf` package cannot
|
|
39
|
-
be loaded (for example a stripped install).
|
|
40
|
-
|
|
41
|
-
## Not planned (explicit decisions)
|
|
42
|
-
|
|
43
|
-
### Proxy support
|
|
44
|
-
|
|
45
|
-
Studio routes requests through urllib's environment proxies and honors
|
|
46
|
-
`UNSLOTH_STUDIO_DISABLE_DNS_PINNING` for enterprise proxies. This port always connects
|
|
47
|
-
directly with DNS pinning. Deliberately out of scope.
|
|
48
|
-
|
|
49
|
-
### Page size budgets
|
|
50
|
-
|
|
51
|
-
Studio's `_page_char_budget()` sizes fetched pages to the serving model's context window.
|
|
52
|
-
This port deliberately has **no character budget at all**: fetched pages and PDFs are
|
|
53
|
-
returned in full (the PDF page-count cap of 50 pages remains, matching Studio). The raw
|
|
54
|
-
download caps (512 KiB text / 10 MiB PDF) still bound what is fetched. A `maxChars`
|
|
55
|
-
parameter remains available on the tool for callers that want to truncate.
|
|
56
|
-
|
|
57
|
-
## Known behavioral differences
|
|
58
|
-
|
|
59
|
-
### Search engine set
|
|
60
|
-
|
|
61
|
-
The port implements ddgs 9.14.4's `DDGS.text()` exactly: the same seven engines
|
|
62
|
-
(duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia — bing is `disabled` in
|
|
63
|
-
ddgs upstream), the same provider-deduplication, href-dedupe aggregator with
|
|
64
|
-
frequency ordering, and the same `SimpleFilterRanker` re-ranking. Remaining deltas:
|
|
65
|
-
|
|
66
|
-
- **TLS fingerprinting**: ddgs uses `primp` with browser TLS impersonation. Node's
|
|
67
|
-
`fetch` has a different fingerprint, so Google/Brave/Yahoo/Yandex may block or serve
|
|
68
|
-
consent pages more aggressively. When an engine is blocked it simply contributes no
|
|
69
|
-
results, exactly as when ddgs is blocked.
|
|
70
|
-
- **User agents**: ddgs uses `fake_useragent`'s database; the port uses a fixed set of
|
|
71
|
-
browser UAs plus ddgs's own Android Google UA generator.
|
|
72
|
-
|
|
73
|
-
### Entity decoding
|
|
74
|
-
|
|
75
|
-
`decodeHtmlEntities` replicates CPython's `html.unescape` exactly (full HTML5 table,
|
|
76
|
-
longest-prefix rule, Windows-1252 numeric mappings, invalid-codepoint handling), so the
|
|
77
|
-
HTML-to-Markdown converter and search-result normalization match Studio's output.
|