pi-unsloth-webtools 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -53,8 +53,8 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
53
53
  pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
54
54
  with Studio's corrupted/incomplete fallback to plain text.
55
55
  - Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
56
- decoding (declared charset, BOM sniffing for UTF-8/16/32, cp1252 rescue for mislabeled
57
- single-byte pages).
56
+ decoding (declared charset, BOM sniffing for UTF-8/16/32, `<meta charset>` sniffing for
57
+ CJK and Windows/ISO encodings, cp1252 rescue for mislabeled single-byte pages).
58
58
  - HTML → Markdown conversion ported from Studio's dependency-free `_html_to_md.py`: headings,
59
59
  links, emphasis, lists, tables, blockquotes, code fences, entity decoding; hidden-element
60
60
  stripping (`hidden`, `aria-hidden`, inline styles); `<article>`/`<main>` main-content scoping
@@ -65,8 +65,22 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
65
65
  - HTML entity decoding replicates CPython's `html.unescape` (full 2,231-entry HTML5 table,
66
66
  longest-prefix rule, Windows-1252 numeric mappings), matching Studio byte-for-byte.
67
67
 
68
- See `ROADMAP.md` for the remaining gaps (per-line styling, table detection, TLS
69
- fingerprinting, proxies).
68
+ ## Known differences from Studio
69
+
70
+ - **PDF styling**: MuPDF.js exposes one font per line, so mixed-style lines style the
71
+ whole line instead of per-span; superscript/subscript/underline/strikeout/highlight
72
+ markers are not emitted. Tables use a conservative text-grid detector — aligned text
73
+ tables are detected, drawn-rule-only tables are not.
74
+ - **Search engines**: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
75
+ impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
76
+ aggressively (a blocked engine simply contributes no results). User agents are a
77
+ fixed browser set plus ddgs's Android Google UA generator, not `fake_useragent`'s
78
+ database.
79
+ - **Empty sweeps**: ddgs 9.14.4 raises the last engine exception; this port reports a
80
+ timeout whenever any engine timed out, so the timeout message is not masked by later
81
+ generic engine failures.
82
+ - **Proxies**: Studio routes through environment proxies; this port always connects
83
+ directly with DNS pinning (deliberately out of scope).
70
84
 
71
85
  ## Development
72
86
 
@@ -74,10 +88,15 @@ fingerprinting, proxies).
74
88
  npm install
75
89
  npm run typecheck
76
90
  npm test
91
+ npm run test:unit
92
+ npm run test:smoke
77
93
  ```
78
94
 
79
95
  ## Tests
80
96
 
97
+ `npm test` runs the full suite. `npm run test:unit` skips the live-network smoke tests,
98
+ and `npm run test:smoke` runs only those.
99
+
81
100
  The suite ports Unsloth Studio's own tests for these tools:
82
101
 
83
102
  - `test/html-to-md.test.ts` — hidden-element stripping and main-content scoping (from
@@ -94,6 +113,8 @@ The suite ports Unsloth Studio's own tests for these tools:
94
113
  aggregator, the ranker, and the Wikipedia engine with a stubbed fetch
95
114
  - `test/pdf-parity.test.ts` — MuPDF engine capabilities: PDF 1.5 object streams,
96
115
  ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
116
+ - `test/entities.test.ts` — `decodeHtmlEntities` parity with CPython `html.unescape`: legacy
117
+ refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
97
118
  - `test/smoke.test.ts` — live network checks against real hosts
98
119
 
99
120
  The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
package/engines.ts CHANGED
@@ -822,7 +822,7 @@ export async function autoTextSearch(
822
822
  const seenProviders = new Set<string>();
823
823
  const aggregator = new ResultsAggregator();
824
824
  const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
825
- let err: unknown = null;
825
+ let timedOut = false;
826
826
  const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
827
827
  const maxWorkers = Math.min(uniqueProviders, Math.ceil(maxResults / 10) + 1);
828
828
  let i = 0;
@@ -835,7 +835,7 @@ export async function autoTextSearch(
835
835
  seenProviders.add(engine.provider);
836
836
  }
837
837
  } catch (e) {
838
- err = e;
838
+ if (e instanceof Error && e.message.includes("timed out")) timedOut = true;
839
839
  }
840
840
  };
841
841
  while (i < engines.length) {
@@ -850,6 +850,6 @@ export async function autoTextSearch(
850
850
  }
851
851
  const results = rankResults(aggregator.extractDicts(), query);
852
852
  if (results.length) return results.slice(0, maxResults);
853
- if (err instanceof Error && err.message.includes("timed out")) throw new SearchTimeoutError();
853
+ if (timedOut) throw new SearchTimeoutError();
854
854
  throw new EmptySweepError();
855
855
  }
package/html-to-md.ts CHANGED
@@ -197,7 +197,7 @@ export function decodeHtmlEntities(text: string): string {
197
197
  return text.replace(CHARREF_RE, (whole, s: string) => {
198
198
  if (s[0] === "#") {
199
199
  const hex = s[1] === "x" || s[1] === "X";
200
- const num = parseInt(s.slice(2).replace(/;+$/, ""), hex ? 16 : 10);
200
+ const num = parseInt(s.slice(hex ? 2 : 1).replace(/;+$/, ""), hex ? 16 : 10);
201
201
  const mapped = INVALID_CHARREFS[num];
202
202
  if (mapped !== undefined) return mapped;
203
203
  if ((num >= 0xd800 && num <= 0xdfff) || num > 0x10ffff) return "\ufffd";
@@ -227,6 +227,21 @@ interface HtmlHandlers {
227
227
  const START_TAG_NAME_RE = /^[a-zA-Z][^\s/>]*/;
228
228
  const ATTR_NAME_RE = /^[^\s=/>]+/;
229
229
 
230
+ const RAW_TEXT_TAGS = [
231
+ "script",
232
+ "style",
233
+ "textarea",
234
+ "title",
235
+ "xmp",
236
+ "iframe",
237
+ "noembed",
238
+ "noframes",
239
+ ];
240
+
241
+ const RAW_TEXT_CLOSERS: Record<string, RegExp> = Object.fromEntries(
242
+ RAW_TEXT_TAGS.map((name) => [name, new RegExp(`</${name}\\s*>`, "i")]),
243
+ );
244
+
230
245
  function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, boolean] {
231
246
  const attrs: AttrDict = {};
232
247
  while (pos < input.length) {
@@ -346,8 +361,22 @@ export function feedHtml(input: string, handlers: HtmlHandlers): void {
346
361
  continue;
347
362
  }
348
363
  emitText(input.slice(textStart, i));
349
- if (tag.kind === "start") handlers.handleStartTag(tag.name!, tag.attrs!);
350
- else if (tag.kind === "startend") handlers.handleStartEndTag(tag.name!, tag.attrs!);
364
+ if (tag.kind === "start") {
365
+ const rawCloser = RAW_TEXT_CLOSERS[tag.name!];
366
+ if (rawCloser) {
367
+ const after = input.slice(tag.end);
368
+ const closeMatch = rawCloser.exec(after);
369
+ if (closeMatch) {
370
+ handlers.handleStartTag(tag.name!, tag.attrs!);
371
+ emitText(after.slice(0, closeMatch.index));
372
+ handlers.handleEndTag(tag.name!);
373
+ textStart = tag.end + closeMatch.index + closeMatch[0].length;
374
+ i = textStart;
375
+ continue;
376
+ }
377
+ }
378
+ handlers.handleStartTag(tag.name!, tag.attrs!);
379
+ } else if (tag.kind === "startend") handlers.handleStartEndTag(tag.name!, tag.attrs!);
351
380
  else if (tag.kind === "end") handlers.handleEndTag(tag.name!);
352
381
  textStart = tag.end;
353
382
  i = tag.end;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-unsloth-webtools",
3
- "version": "0.2.0",
3
+ "version": "0.2.2",
4
4
  "type": "module",
5
5
  "description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
6
6
  "main": "index.ts",
@@ -31,7 +31,6 @@
31
31
  "entities.ts",
32
32
  "pdf.ts",
33
33
  "README.md",
34
- "ROADMAP.md",
35
34
  "LICENSE"
36
35
  ],
37
36
  "pi": {
@@ -48,8 +47,10 @@
48
47
  },
49
48
  "scripts": {
50
49
  "test": "vitest run",
50
+ "test:unit": "vitest run --exclude **/smoke.test.ts",
51
+ "test:smoke": "vitest run test/smoke.test.ts",
51
52
  "typecheck": "tsc --noEmit",
52
- "prepublishOnly": "npm run typecheck"
53
+ "prepublishOnly": "npm run typecheck && npm run test:unit"
53
54
  },
54
55
  "devDependencies": {
55
56
  "@earendil-works/pi-coding-agent": "^0.84.0",
package/web-fetch.ts CHANGED
@@ -237,6 +237,34 @@ function hasSingleByteTextEvidence(data: Buffer): boolean {
237
237
  return ascii / data.length >= MIN_SINGLE_BYTE_ASCII_RATIO;
238
238
  }
239
239
 
240
+ const CHARSET_ALIASES: Record<string, string> = {
241
+ gbk: "gbk",
242
+ gb2312: "gbk",
243
+ "gb-2312": "gbk",
244
+ cp936: "gbk",
245
+ "x-gbk": "gbk",
246
+ gb18030: "gb18030",
247
+ big5: "big5",
248
+ "big5-hkscs": "big5",
249
+ sjis: "shift_jis",
250
+ "x-sjis": "shift_jis",
251
+ cp932: "shift_jis",
252
+ "shift-jis": "shift_jis",
253
+ "euc-jp": "euc-jp",
254
+ "euc-kr": "euc-kr",
255
+ ksc5601: "euc-kr",
256
+ "ks_c_5601-1987": "euc-kr",
257
+ "ks_c_5601-1989": "euc-kr",
258
+ "iso-2022-jp": "iso-2022-jp",
259
+ "koi8-r": "koi8-r",
260
+ "koi8-u": "koi8-u",
261
+ cp866: "cp866",
262
+ "x-mac-cyrillic": "x-mac-cyrillic",
263
+ "windows-874": "windows-874",
264
+ cp874: "windows-874",
265
+ "tis-620": "tis-620",
266
+ };
267
+
240
268
  function normalizeCharset(name: string): string | null {
241
269
  const n = name.trim().replace(/["']/g, "").toLowerCase();
242
270
  switch (n) {
@@ -270,8 +298,28 @@ function normalizeCharset(name: string): string | null {
270
298
  case "utf-32be":
271
299
  return "utf-32be";
272
300
  default:
273
- return null;
301
+ break;
302
+ }
303
+ const alias = CHARSET_ALIASES[n];
304
+ if (alias !== undefined) return alias;
305
+ if (/^windows-125[0-8]$/.test(n) || /^iso-8859-(?:[2-9]|1[0-6])$/.test(n)) return n;
306
+ return null;
307
+ }
308
+
309
+ function sniffMetaCharset(bytes: Buffer): string | null {
310
+ const head = bytes.subarray(0, 1024).toString("latin1").toLowerCase();
311
+ const match =
312
+ /<meta\b[^>]*\bcharset\s*=\s*["']?\s*([a-z0-9_.\-]+)/.exec(head) ??
313
+ /<meta\b[^>]*\bhttp-equiv\s*=\s*["']?content-type["']?[^>]*\bcharset\s*=\s*["']?\s*([a-z0-9_.\-]+)/.exec(head);
314
+ return match ? normalizeCharset(match[1]) : null;
315
+ }
316
+
317
+ function sniffMetaCharsetForHtml(bytes: Buffer, contentType: string): string | null {
318
+ if (!contentType.includes("html")) {
319
+ const probe = bytes.subarray(0, 256).toString("latin1");
320
+ if (!looksLikeHtmlDocument(probe)) return null;
274
321
  }
322
+ return sniffMetaCharset(bytes);
275
323
  }
276
324
 
277
325
  function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
@@ -328,11 +376,38 @@ function decodeWithCodec(bytes: Buffer, codec: string | null): string {
328
376
  case "cp1252":
329
377
  return decodeSingleByte(bytes, true);
330
378
  case "iso8859-1":
331
- default:
332
379
  return decodeSingleByte(bytes, false);
380
+ default:
381
+ return decodeWithLabel(bytes, codec);
333
382
  }
334
383
  }
335
384
 
385
+ function decodeWithLabel(bytes: Buffer, label: string | null): string {
386
+ if (label === null) return decodeSingleByte(bytes, false);
387
+ if (label === "tis-620") return decodeTis620(bytes);
388
+ try {
389
+ return new TextDecoder(label, { fatal: false }).decode(bytes);
390
+ } catch {
391
+ return decodeSingleByte(bytes, false);
392
+ }
393
+ }
394
+
395
+ function decodeTis620(bytes: Buffer): string {
396
+ let out = "";
397
+ for (const byte of bytes) {
398
+ if (byte < 0x80) {
399
+ out += String.fromCharCode(byte);
400
+ } else if (byte >= 0xa1 && byte <= 0xfb) {
401
+ out += String.fromCodePoint(0x0e01 + byte - 0xa1);
402
+ } else if (byte === 0xa0) {
403
+ out += "\u00a0";
404
+ } else {
405
+ out += "\ufffd";
406
+ }
407
+ }
408
+ return out;
409
+ }
410
+
336
411
 
337
412
  async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promise<ResolvedHost> {
338
413
  let addresses: { address: string; family: number }[];
@@ -386,20 +461,19 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
386
461
  const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
387
462
  let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
388
463
  let extendedForPdf = false;
389
- let done = false;
390
464
  const finish = (err: string | null, body: Buffer) => {
391
- if (done) return;
392
- done = true;
393
- if (err) reject(new Error(err));
394
- else
395
- resolve({
396
- status: res.statusCode ?? 0,
397
- headers: res.headers as Record<string, string | string[] | undefined>,
398
- body,
399
- });
465
+ settle(() => {
466
+ if (err) reject(new Error(err));
467
+ else
468
+ resolve({
469
+ status: res.statusCode ?? 0,
470
+ headers: res.headers as Record<string, string | string[] | undefined>,
471
+ body,
472
+ });
473
+ });
400
474
  };
401
475
  res.on("data", (chunk: Buffer) => {
402
- if (done) return;
476
+ if (settled) return;
403
477
  if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
404
478
  if (hasPdfMagic(Buffer.concat(chunks))) {
405
479
  limit = opts.maxPdfBytes;
@@ -423,10 +497,17 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
423
497
  res.on("end", () => finish(null, Buffer.concat(chunks)));
424
498
  res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
425
499
  });
426
- request.on("timeout", () => request.destroy(new Error("timed out")));
427
- request.on("error", (err) => reject(err));
428
500
  const onAbort = () => request.destroy(new Error("cancelled"));
429
501
  opts.signal?.addEventListener("abort", onAbort, { once: true });
502
+ let settled = false;
503
+ const settle = (action: () => void) => {
504
+ if (settled) return;
505
+ settled = true;
506
+ opts.signal?.removeEventListener("abort", onAbort);
507
+ action();
508
+ };
509
+ request.on("timeout", () => request.destroy(new Error("timed out")));
510
+ request.on("error", (err) => settle(() => reject(err)));
430
511
  request.end();
431
512
  });
432
513
  }
@@ -471,6 +552,7 @@ export async function fetchUrlRaw(
471
552
  "User-Agent": userAgent,
472
553
  Host: hostHeader,
473
554
  Accept: "text/html,application/xhtml+xml,text/plain;q=0.9,*/*;q=0.5",
555
+ "Accept-Encoding": "identity",
474
556
  };
475
557
  if (options.extraHeaders) Object.assign(headers, options.extraHeaders);
476
558
  const inactivity = Math.max(1, deadline - now());
@@ -497,8 +579,9 @@ export async function fetchUrlRaw(
497
579
 
498
580
  if (response.status >= 300 && response.status < 400) {
499
581
  if (![301, 302, 303, 307, 308].includes(response.status)) {
582
+ const reason = http.STATUS_CODES[response.status] ?? "";
500
583
  return {
501
- error: `Failed to fetch URL: HTTP ${response.status} ${statusReason(response.status)}`,
584
+ error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
502
585
  body: "",
503
586
  contentType: "",
504
587
  };
@@ -577,7 +660,10 @@ export async function fetchUrlRaw(
577
660
 
578
661
  const declaredCodec = declaredCharset ? normalizeCharset(declaredCharset) : null;
579
662
  const bomCodec = bomCodecFor(response.body);
580
- const rawHtml = decodeWithCodec(response.body, declaredCodec ?? bomCodec ?? "utf-8");
663
+ const rawHtml = decodeWithCodec(
664
+ response.body,
665
+ declaredCodec ?? bomCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
666
+ );
581
667
 
582
668
  if (looksBinary(rawHtml)) {
583
669
  let alt: string | null = null;
@@ -612,13 +698,6 @@ function bomCodecFor(bytes: Buffer): string | null {
612
698
  return null;
613
699
  }
614
700
 
615
- function statusReason(status: number): string {
616
- const reasons: Record<number, string> = {
617
- 301: "Moved Permanently", 302: "Found", 303: "See Other", 307: "Temporary Redirect", 308: "Permanent Redirect",
618
- };
619
- return reasons[status] ?? "";
620
- }
621
-
622
701
  export function truncatePageText(text: string, maxChars?: number): string {
623
702
  if (!text) return "(page returned no readable text)";
624
703
  if (typeof maxChars === "number" && maxChars > 0 && text.length > maxChars) {
package/web-search.ts CHANGED
@@ -64,23 +64,17 @@ export async function webSearch(
64
64
  const results = await client(effectiveQuery, wanted, signal);
65
65
  if (signal?.aborted) return "Search cancelled.";
66
66
  if (!results.length) return EMPTY_SEARCH_RESULTS[0];
67
- const parts: string[] = [];
67
+ const allowed: SearchResult[] = [];
68
68
  for (const result of results) {
69
- if (parts.length >= maxResults) break;
69
+ if (allowed.length >= maxResults) break;
70
70
  const href = String(result.href ?? "").trim();
71
71
  if (href && !checkUrlAccess(href, policy)[0]) continue;
72
- const title = String(result.title ?? "").replace(/\s+/g, " ");
73
- const snippet = String(result.body ?? "").replace(/\s+/g, " ");
74
- parts.push(`Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`);
72
+ allowed.push(result);
75
73
  }
76
- if (!parts.length) return EMPTY_SEARCH_RESULTS[1];
77
- const text = parts.join("\n\n---\n\n");
78
- return (
79
- text +
80
- "\n\n---\n\nIMPORTANT: These are only short snippets. " +
81
- 'To get the full page content, call web_search with the url parameter (e.g. {"url": "<URL>"}).'
82
- );
74
+ if (!allowed.length) return EMPTY_SEARCH_RESULTS[1];
75
+ return formatSearchResults(allowed);
83
76
  } catch (err) {
77
+ if (signal?.aborted) return "Search cancelled.";
84
78
  return searchFailureMessage(err, timeoutMs);
85
79
  }
86
80
  }
@@ -98,9 +92,9 @@ export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS
98
92
 
99
93
  export function formatSearchResults(results: SearchResult[]): string {
100
94
  const parts = results.map((result) => {
101
- const title = result.title.replace(/\s+/g, " ");
102
- const href = result.href.trim();
103
- const snippet = result.body.replace(/\s+/g, " ");
95
+ const title = String(result.title ?? "").replace(/\s+/g, " ");
96
+ const href = String(result.href ?? "").trim();
97
+ const snippet = String(result.body ?? "").replace(/\s+/g, " ");
104
98
  return `Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`;
105
99
  });
106
100
  const text = parts.join("\n\n---\n\n");
package/ROADMAP.md DELETED
@@ -1,77 +0,0 @@
1
- # Roadmap
2
-
3
- Tracking the remaining gaps between this port and Unsloth Studio's web tools, and the
4
- planned work to close them.
5
-
6
- ## Implemented
7
-
8
- ### PDF text extraction (MuPDF engine)
9
-
10
- `pdf.ts` uses the official **MuPDF.js** (`mupdf` npm package) — the same C engine that
11
- PyMuPDF wraps — replacing the earlier minimal built-in extractor:
12
-
13
- - Full xref handling: tables, cross-reference streams, and PDF 1.5+ object streams
14
- - All standard stream filters (FlateDecode, ASCII85Decode, LZW, RunLength, DCT, JPX, ...)
15
- - Font encodings and ToUnicode mapping (non-Latin text extracts correctly)
16
- - Encryption detection via `needsPassword()` (reported as unreadable, matching pymupdf's
17
- no-password behavior)
18
- - The markdown layer replicates pymupdf4llm's algorithm: `IdentifyHeaders` font-size
19
- heading detection, `get_raw_lines` line reconstruction (tolerance 3, 10% span-join
20
- delta), `write_text` styling (bold/italic/mono, code fences, bullets, link
21
- resolution with `%0x`-escaped URIs), and Studio's corrupted/incomplete fallback to
22
- plain MuPDF text with the exact thresholds from `backend/core/rag/parsers.py`
23
- - Table detection and pipe-markdown rendering matching pymupdf's `Table.to_markdown`
24
- shape (`|header|`, `|---|`, detail rows, `Col{i}` fill for empty headers)
25
-
26
- Known deltas vs pymupdf4llm:
27
-
28
- - Span-level styling: MuPDF.js's structured-text JSON exposes one font per line, so
29
- mixed-style lines (one bold word inside a body line) style the whole line instead of
30
- per-span. Line-level styling matches for homogeneous lines.
31
- - Superscript/subscript/underline/strikeout/highlight markers are not emitted (the
32
- JSON does not expose char-level flags).
33
- - Table detection is a conservative text-grid detector (column-start clustering with
34
- a 5 pt tolerance, contiguous multi-row bands) instead of PyMuPDF's
35
- `find_tables()` vector-graphics analysis. Aligned text tables are detected; tables
36
- defined only by drawn rules without aligned text are not.
37
-
38
- The minimal extractor remains as an automatic fallback when the `mupdf` package cannot
39
- be loaded (for example a stripped install).
40
-
41
- ## Not planned (explicit decisions)
42
-
43
- ### Proxy support
44
-
45
- Studio routes requests through urllib's environment proxies and honors
46
- `UNSLOTH_STUDIO_DISABLE_DNS_PINNING` for enterprise proxies. This port always connects
47
- directly with DNS pinning. Deliberately out of scope.
48
-
49
- ### Page size budgets
50
-
51
- Studio's `_page_char_budget()` sizes fetched pages to the serving model's context window.
52
- This port deliberately has **no character budget at all**: fetched pages and PDFs are
53
- returned in full (the PDF page-count cap of 50 pages remains, matching Studio). The raw
54
- download caps (512 KiB text / 10 MiB PDF) still bound what is fetched. A `maxChars`
55
- parameter remains available on the tool for callers that want to truncate.
56
-
57
- ## Known behavioral differences
58
-
59
- ### Search engine set
60
-
61
- The port implements ddgs 9.14.4's `DDGS.text()` exactly: the same seven engines
62
- (duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia — bing is `disabled` in
63
- ddgs upstream), the same provider-deduplication, href-dedupe aggregator with
64
- frequency ordering, and the same `SimpleFilterRanker` re-ranking. Remaining deltas:
65
-
66
- - **TLS fingerprinting**: ddgs uses `primp` with browser TLS impersonation. Node's
67
- `fetch` has a different fingerprint, so Google/Brave/Yahoo/Yandex may block or serve
68
- consent pages more aggressively. When an engine is blocked it simply contributes no
69
- results, exactly as when ddgs is blocked.
70
- - **User agents**: ddgs uses `fake_useragent`'s database; the port uses a fixed set of
71
- browser UAs plus ddgs's own Android Google UA generator.
72
-
73
- ### Entity decoding
74
-
75
- `decodeHtmlEntities` replicates CPython's `html.unescape` exactly (full HTML5 table,
76
- longest-prefix rule, Windows-1252 numeric mappings, invalid-codepoint handling), so the
77
- HTML-to-Markdown converter and search-result normalization match Studio's output.