pi-unsloth-webtools 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -26,44 +26,72 @@ Mirrors Unsloth Studio's `web_search` tool:
26
26
 
27
27
  - Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
28
28
  (duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
29
- the same provider deduplication, href-dedupe aggregator with frequency ordering, and the
30
- same `SimpleFilterRanker` re-ranking. Formats results identically: `Title:` / `URL:` /
29
+ the same provider deduplication, href-dedupe aggregator with frequency ordering (hrefs are
30
+ canonicalized first — `utm_*`/tracking parameters and fragments are dropped and the URL is
31
+ re-serialized, collapsing host-case, default-port, and trailing-slash variants — so the
32
+ same page found via different tracking links collapses), and the same `SimpleFilterRanker`
33
+ re-ranking. Formats results identically: `Title:` / `URL:` /
31
34
  `Snippet:` blocks separated by `---`, ending with the hint to pass `{"url": "<URL>"}` to
32
35
  read a full page.
33
- - Accepts an optional `url` parameter; when given, fetches that page's text instead of searching.
36
+ - Accepts an optional `url` parameter; when given, fetches that page's text instead of
37
+ searching (optionally truncated with `maxChars`).
34
38
  - Rate-limit, timeout, and empty-result messages mirror Studio's `_search_failure_message`.
39
+ - Transient engine failures (network errors or null responses) are retried once with a short
40
+ backoff inside the same timeout budget (a retry that cannot fit in the remaining budget is
41
+ skipped); timeouts and cancellations are never retried.
35
42
 
36
43
  ### web_fetch
37
44
 
38
45
  Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
39
46
 
40
47
  - URL scheme normalization (bare hosts like `google.com` become `https://google.com`).
41
- - URL validation: http/https only, no credentials or encoded hostnames, hostname/port checks.
48
+ - URL validation: http/https only, no credentials or encoded hostnames, hostname/port checks (any
49
+ port 1–65535 is permitted; SSRF protection is enforced at the resolved-IP layer, not by port
50
+ allowlists).
42
51
  - DNS resolution with SSRF protection: every resolved address is validated against
43
52
  private/loopback/link-local/CGNAT/documentation/multicast/reserved ranges, then the validated IP
44
53
  is pinned for the connection (custom `lookup` + SNI `servername`), so DNS cannot rebind between
45
- validation and fetch.
54
+ validation and fetch; resolution shares the caller's abort signal and the overall deadline,
55
+ so a stuck resolver cannot outlive the fetch.
46
56
  - GitHub repo root pages are rewritten to the unauthenticated README API
47
- (`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
57
+ (`Accept: application/vnd.github.raw+json`), falling back to the raw README URL
58
+ (`raw.githubusercontent.com`, no API rate limit) and then to the HTML page on failure.
59
+ returning error-page content.
48
60
  - Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
49
61
  - 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
50
- (`signal` cancels mid-flight).
62
+ (`signal` cancels mid-flight). Fetches cut off by a download cap are marked with a trailing
63
+ truncation notice, so a partial page is not mistaken for a complete one.
64
+ - Responses sent with a `Content-Encoding` of gzip, deflate, or brotli (servers that ignore the
65
+ `Accept-Encoding: identity` request) are decompressed while streaming, so the download caps
66
+ bound the decoded page text (a gzip'd PDF still gets the 10 MiB PDF budget via magic sniffing on
67
+ the decoded head) and a stream cut by the cap or a mid-stream decode failure returns the readable
68
+ decoded prefix with the truncation notice instead of a binary-content error. A 64 MiB
69
+ decoded-output cap bounds a compressed bomb.
70
+ - Long fetches and searches report a short progress note to the session before they start, so
71
+ slow tool calls are not silent.
51
72
  - PDF text extraction via the official MuPDF.js engine (the same C library pymupdf wraps):
52
73
  object streams, all filters, ToUnicode fonts, encryption detection, and a
53
- pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
74
+ pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables),
75
+ running header/footer and page-number stripping,
54
76
  with Studio's corrupted/incomplete fallback to plain text.
55
77
  - Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
56
- decoding (declared charset, BOM sniffing for UTF-8/16/32, `<meta charset>` sniffing for
78
+ decoding (BOM sniffing for UTF-8/16/32 first, then the declared charset, `<meta charset>` sniffing for
57
79
  CJK and Windows/ISO encodings, cp1252 rescue for mislabeled single-byte pages).
58
80
  - HTML → Markdown conversion ported from Studio's dependency-free `_html_to_md.py`: headings,
59
81
  links, emphasis, lists, tables, blockquotes, code fences, entity decoding; hidden-element
60
82
  stripping (`hidden`, `aria-hidden`, inline styles); `<article>`/`<main>` main-content scoping
61
83
  with link-density header stripping; boilerplate-line removal.
62
84
  - No page-size budget: fetched pages and PDFs are returned in full (Studio's window-aware
63
- cap is deliberately dropped; the optional `maxChars` parameter still truncates when given).
85
+ cap is deliberately dropped; the optional `maxChars` parameter still truncates when given,
86
+ on `web_fetch` and on `web_search`'s url mode).
64
87
  The 512 KiB / 10 MiB download caps still bound the raw fetch.
65
88
  - HTML entity decoding replicates CPython's `html.unescape` (full 2,231-entry HTML5 table,
66
89
  longest-prefix rule, Windows-1252 numeric mappings), matching Studio byte-for-byte.
90
+ - Fetched HTML pages are prefixed with the decoded document `<title>`, so the model can
91
+ see which page it is reading. `Author:` (`meta name=author` / `article:author` / `dc.creator`),
92
+ `Date:` (`article:published_time` / `dc.date` / `date`) and `Site:` (`og:site_name` /
93
+ `application-name`) lines are added when declared, so the model can judge recency and
94
+ provenance.
67
95
 
68
96
  ## Known differences from Studio
69
97
 
@@ -71,6 +99,11 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
71
99
  whole line instead of per-span; superscript, subscript, underline, strikeout, and
72
100
  highlight markers are not emitted. Tables use a conservative text-grid detector:
73
101
  aligned text tables are detected, drawn-rule-only tables are not.
102
+ - Running headers and footers: lines repeated at the same page-edge position on at
103
+ least half the pages (two pages minimum) are dropped from the markdown layer, as is
104
+ any numeric-only line at a fixed edge position where page numbers appear on at least
105
+ half the pages (so a one-off number sharing that position is dropped too, while fused
106
+ labels like `Page 3 of 12` survive). Studio and pymupdf4llm return them verbatim.
74
107
  - Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
75
108
  impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
76
109
  aggressively (a blocked engine simply contributes no results). User agents are a
@@ -78,9 +111,18 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
78
111
  database.
79
112
  - Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
80
113
  timeout whenever any engine timed out, so the timeout message is not masked by later
81
- generic engine failures.
114
+ generic engine failures. The timeout budget bounds the entire sweep: per-engine
115
+ timeouts shrink as the budget is consumed, so the reported timeout matches the
116
+ worst-case wall time.
82
117
  - Proxies: Studio routes through environment proxies; this port always connects
83
118
  directly with DNS pinning (deliberately out of scope).
119
+ - Dedup and titles: the aggregator keys on canonicalized hrefs (`utm_*`/tracking parameters
120
+ and fragments stripped, then the URL re-serialized); fetched HTML pages are prefixed with
121
+ the document `<title>`. Studio keys on raw hrefs and returns the converted body alone.
122
+ - Upstream drift: current ddgs ships ten backends (adding bing, startpage, grokipedia),
123
+ requires a `vqd` token for DuckDuckGo, and exposes an `extract()` mode. This port
124
+ deliberately pins the Studio snapshot — seven engines, bing disabled upstream, no vqd,
125
+ no pagination — so engine behavior matches Studio rather than ddgs head.
84
126
 
85
127
  ## Development
86
128
 
@@ -115,7 +157,8 @@ The suite ports Unsloth Studio's own tests for these tools:
115
157
  ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
116
158
  - `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
117
159
  legacy refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
118
- - `test/smoke.test.ts`: live network checks against real hosts
160
+ - `test/smoke.test.ts`: live network checks against real hosts, including a per-engine
161
+ result-health sweep (at least three engines must return well-formed results)
119
162
 
120
163
  The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
121
164
  with fakes, mirroring how the Studio suite monkeypatches `_validate_and_resolve_host`
package/engines.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { randomBytes } from "node:crypto";
2
- import { decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
2
+ import { collapseWhitespace, decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
3
3
  import type { AttrDict } from "./html-to-md.ts";
4
4
  import { randomUserAgent } from "./user-agents.ts";
5
5
  export class EmptySweepError extends Error {
@@ -35,7 +35,7 @@ export function normalizeText(raw: string): string {
35
35
  text = decodeHtmlEntities(text);
36
36
  text = text.normalize("NFC");
37
37
  text = text.replace(/[\p{Cc}\p{Cf}\p{Co}\p{Cs}\p{Cn}]/gu, "");
38
- return text.trim().split(/\s+/).join(" ");
38
+ return collapseWhitespace(text);
39
39
  }
40
40
 
41
41
  export function normalizeUrl(url: string): string {
@@ -47,6 +47,41 @@ export function normalizeUrl(url: string): string {
47
47
  }
48
48
  }
49
49
 
50
+ const TRACKING_PARAM_NAMES = new Set([
51
+ "_hsenc",
52
+ "_hsmi",
53
+ "dclid",
54
+ "fbclid",
55
+ "gbraid",
56
+ "gclid",
57
+ "gclsrc",
58
+ "igshid",
59
+ "mc_cid",
60
+ "mc_eid",
61
+ "msclkid",
62
+ "srsltid",
63
+ "twclid",
64
+ "wbraid",
65
+ "yclid",
66
+ ]);
67
+
68
+ export function canonicalizeHref(href: string): string {
69
+ if (!href) return "";
70
+ try {
71
+ const url = new URL(href);
72
+ for (const key of [...url.searchParams.keys()]) {
73
+ const lower = key.toLowerCase();
74
+ if (lower.startsWith("utm_") || TRACKING_PARAM_NAMES.has(lower)) {
75
+ url.searchParams.delete(key);
76
+ }
77
+ }
78
+ url.hash = "";
79
+ return url.toString();
80
+ } catch {
81
+ return href;
82
+ }
83
+ }
84
+
50
85
  export interface DomNode {
51
86
  tag: string;
52
87
  attrs: Record<string, string>;
@@ -382,7 +417,7 @@ export function extractResults(
382
417
  ["body", elementsXpath.body],
383
418
  ] as const;
384
419
  for (const [key, value] of entries) {
385
- const data = xpathText(value, item).join("").trim().split(/\s+/).join(" ");
420
+ const data = collapseWhitespace(xpathText(value, item).join(""));
386
421
  if (!data) continue;
387
422
  result[key] = key === "href" ? normalizeUrl(data) : normalizeText(data);
388
423
  }
@@ -444,15 +479,22 @@ export interface Engine {
444
479
  ): Promise<SearchResult[] | null>;
445
480
  }
446
481
 
482
+ interface HttpRequestOptions {
483
+ headers?: Record<string, string>;
484
+ cookies?: Record<string, string>;
485
+ timeoutMs: number;
486
+ signal?: AbortSignal;
487
+ }
488
+
489
+ interface HttpOptions extends HttpRequestOptions {
490
+ method?: string;
491
+ body?: string;
492
+ }
493
+
447
494
  async function httpGet(
448
495
  url: string,
449
496
  params: Record<string, string>,
450
- options: {
451
- headers?: Record<string, string>;
452
- cookies?: Record<string, string>;
453
- timeoutMs: number;
454
- signal?: AbortSignal;
455
- },
497
+ options: HttpRequestOptions,
456
498
  ): Promise<string | null> {
457
499
  const target = new URL(url);
458
500
  for (const [key, value] of Object.entries(params)) target.searchParams.set(key, value);
@@ -462,17 +504,15 @@ async function httpGet(
462
504
  async function httpPost(
463
505
  url: string,
464
506
  data: Record<string, string>,
465
- options: {
466
- headers?: Record<string, string>;
467
- cookies?: Record<string, string>;
468
- timeoutMs: number;
469
- signal?: AbortSignal;
470
- },
507
+ options: HttpRequestOptions,
471
508
  ): Promise<string | null> {
472
509
  return httpFetch(url, { ...options, method: "POST", body: new URLSearchParams(data).toString() });
473
510
  }
474
511
 
475
512
  const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
513
+ const ENGINE_RETRY_BACKOFF_MS = 250;
514
+
515
+ const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
476
516
 
477
517
  async function readBodyCapped(response: Response): Promise<string | null> {
478
518
  const declared = Number(response.headers.get("content-length") ?? "0");
@@ -494,22 +534,22 @@ async function readBodyCapped(response: Response): Promise<string | null> {
494
534
  return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
495
535
  }
496
536
 
537
+ function mapFetchError(err: unknown): never {
538
+ if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
539
+ if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
540
+ throw err;
541
+ }
542
+
497
543
  async function httpFetch(
498
544
  url: string,
499
- options: {
500
- method?: string;
501
- body?: string;
502
- headers?: Record<string, string>;
503
- cookies?: Record<string, string>;
504
- timeoutMs: number;
505
- signal?: AbortSignal;
506
- },
545
+ options: HttpOptions,
507
546
  ): Promise<string | null> {
508
547
  const headers: Record<string, string> = {
509
548
  "User-Agent": options.headers?.["User-Agent"] ?? randomUserAgent(),
510
549
  Accept: "*/*",
511
550
  ...options.headers,
512
551
  };
552
+ if (options.method === "POST") headers["Content-Type"] = "application/x-www-form-urlencoded";
513
553
  const cookie = options.cookies
514
554
  ? Object.entries(options.cookies)
515
555
  .map(([key, value]) => `${key}=${value}`)
@@ -527,17 +567,20 @@ async function httpFetch(
527
567
  signal: AbortSignal.any(signals),
528
568
  });
529
569
  } catch (err) {
530
- if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
531
- if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
532
- throw err;
570
+ throw mapFetchError(err);
571
+ }
572
+ if (response.status !== 200) {
573
+ try {
574
+ await response.body?.cancel();
575
+ } catch {
576
+ return null;
577
+ }
578
+ return null;
533
579
  }
534
- if (response.status !== 200) return null;
535
580
  try {
536
581
  return await readBodyCapped(response);
537
582
  } catch (err) {
538
- if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
539
- if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
540
- throw err;
583
+ throw mapFetchError(err);
541
584
  }
542
585
  }
543
586
 
@@ -648,7 +691,7 @@ const MOJEEK: Engine = {
648
691
  const YAHOO: Engine = {
649
692
  name: "yahoo",
650
693
  provider: "bing",
651
- async search(query, ctx, timeoutMs, signal) {
694
+ async search(query, _ctx, timeoutMs, signal) {
652
695
  const ylt = tokenUrlSafe(18);
653
696
  const ylu = tokenUrlSafe(35);
654
697
  const html = await httpGet(
@@ -674,7 +717,7 @@ const YAHOO: Engine = {
674
717
  const YANDEX: Engine = {
675
718
  name: "yandex",
676
719
  provider: "yandex",
677
- async search(query, ctx, timeoutMs, signal) {
720
+ async search(query, _ctx, timeoutMs, signal) {
678
721
  const searchid = 1000000 + Math.floor(Math.random() * 9000000);
679
722
  const html = await httpGet(
680
723
  "https://yandex.com/search/site/",
@@ -695,6 +738,7 @@ const WIKIPEDIA: Engine = {
695
738
  provider: "wikipedia",
696
739
  priority: 2,
697
740
  async search(query, ctx, timeoutMs, signal) {
741
+ const started = Date.now();
698
742
  const lang = ctx.region.toLowerCase().split("-")[1] ?? "en";
699
743
  const encoded = encodeURIComponent(query);
700
744
  const opensearchUrl =
@@ -715,7 +759,7 @@ const WIKIPEDIA: Engine = {
715
759
  const extractUrl =
716
760
  `https://${lang}.wikipedia.org/w/api.php?action=query&format=json&prop=extracts` +
717
761
  `&titles=${encodeURIComponent(title)}&explaintext=0&exintro=0&redirects=1`;
718
- const extract = await httpGet(extractUrl, {}, { timeoutMs, signal });
762
+ const extract = await httpGet(extractUrl, {}, { timeoutMs: Math.max(1, timeoutMs - (Date.now() - started)), signal });
719
763
  if (extract) {
720
764
  try {
721
765
  const pageData = JSON.parse(extract) as {
@@ -732,7 +776,7 @@ const WIKIPEDIA: Engine = {
732
776
  },
733
777
  };
734
778
 
735
- const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
779
+ export const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
736
780
 
737
781
  export class ResultsAggregator {
738
782
  private cache = new Map<string, SearchResult>();
@@ -743,10 +787,10 @@ export class ResultsAggregator {
743
787
  }
744
788
 
745
789
  append(item: SearchResult): void {
746
- const key = item.href;
790
+ const key = canonicalizeHref(item.href);
747
791
  const existing = this.cache.get(key);
748
792
  if (!existing || item.body.length > existing.body.length) {
749
- this.cache.set(key, item);
793
+ this.cache.set(key, { ...item, href: key });
750
794
  }
751
795
  this.counter.set(key, (this.counter.get(key) ?? 0) + 1);
752
796
  }
@@ -815,6 +859,7 @@ export async function autoTextSearch(
815
859
  signal?: AbortSignal,
816
860
  ): Promise<SearchResult[]> {
817
861
  const engines = shuffledEngines();
862
+ const deadline = Date.now() + timeoutMs;
818
863
  const seenProviders = new Set<string>();
819
864
  const aggregator = new ResultsAggregator();
820
865
  const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
@@ -825,19 +870,40 @@ export async function autoTextSearch(
825
870
  let i = 0;
826
871
  let pending: Promise<void>[] = [];
827
872
  const run = async (engine: Engine) => {
828
- try {
829
- const results = await engine.search(query, ctx, timeoutMs, signal);
830
- if (results && results.length) {
831
- aggregator.extend(results);
832
- seenProviders.add(engine.provider);
873
+ let results: SearchResult[] | null = null;
874
+ for (let attempt = 0; attempt < 2 && results === null; attempt++) {
875
+ const budgetLeft = deadline - Date.now();
876
+ if (budgetLeft <= 0) return;
877
+ if (attempt > 0 && budgetLeft < ENGINE_RETRY_BACKOFF_MS) return;
878
+ const remaining = Math.max(1, budgetLeft);
879
+ try {
880
+ results = await engine.search(query, ctx, remaining, signal);
881
+ } catch (e) {
882
+ if (e instanceof SearchCancelled) {
883
+ cancelled = true;
884
+ return;
885
+ }
886
+ if (e instanceof SearchTimeoutError) {
887
+ timedOut = true;
888
+ return;
889
+ }
833
890
  }
834
- } catch (e) {
835
- if (e instanceof SearchCancelled) cancelled = true;
836
- if (e instanceof SearchTimeoutError) timedOut = true;
891
+ if (results === null && attempt === 0) {
892
+ const backoff = Math.min(ENGINE_RETRY_BACKOFF_MS, Math.max(0, deadline - Date.now()));
893
+ if (backoff > 0) await sleep(backoff);
894
+ if (signal?.aborted) {
895
+ cancelled = true;
896
+ return;
897
+ }
898
+ }
899
+ }
900
+ if (results && results.length) {
901
+ aggregator.extend(results);
902
+ seenProviders.add(engine.provider);
837
903
  }
838
904
  };
839
905
  while (i < engines.length) {
840
- if (aggregator.size >= maxResults) break;
906
+ if (aggregator.size >= maxResults || cancelled) break;
841
907
  const engine = engines[i++];
842
908
  if (seenProviders.has(engine.provider)) continue;
843
909
  pending.push(run(engine));
package/html-to-md.ts CHANGED
@@ -194,7 +194,7 @@ class HeaderFrame {
194
194
  const CHARREF_RE = /&(#[0-9]+;?|#[xX][0-9a-fA-F]+;?|[^\t\n\f <&#;]{1,32};?)/g;
195
195
 
196
196
  export function decodeHtmlEntities(text: string): string {
197
- return text.replace(CHARREF_RE, (whole, s: string) => {
197
+ return text.replace(CHARREF_RE, (_whole, s: string) => {
198
198
  if (s[0] === "#") {
199
199
  const hex = s[1] === "x" || s[1] === "X";
200
200
  const num = parseInt(s.slice(hex ? 2 : 1).replace(/;+$/, ""), hex ? 16 : 10);
@@ -214,6 +214,10 @@ export function decodeHtmlEntities(text: string): string {
214
214
  });
215
215
  }
216
216
 
217
+ export function collapseWhitespace(text: string): string {
218
+ return text.replace(/\s+/g, " ").trim();
219
+ }
220
+
217
221
 
218
222
  interface HtmlHandlers {
219
223
  handleStartTag(name: string, attrs: AttrDict): void;
@@ -331,9 +335,9 @@ export function feedHtml(input: string, handlers: HtmlHandlers): void {
331
335
  pos = amp + named[0].length;
332
336
  continue;
333
337
  }
334
- const numeric = /^&#(?:[xX]([0-9a-fA-F]+)|([0-9]+));/.exec(text.slice(amp));
338
+ const numeric = /^&#([xX][0-9a-fA-F]+|[0-9]+);?/.exec(text.slice(amp));
335
339
  if (numeric) {
336
- handlers.handleCharRef(numeric[1] ?? numeric[2]);
340
+ handlers.handleCharRef(numeric[1]);
337
341
  pos = amp + numeric[0].length;
338
342
  continue;
339
343
  }
@@ -384,6 +388,10 @@ export function feedHtml(input: string, handlers: HtmlHandlers): void {
384
388
  emitText(input.slice(textStart));
385
389
  }
386
390
 
391
+ function popMarksAbove(marks: number[], index: number): void {
392
+ while (marks.length && marks[marks.length - 1] >= index) marks.pop();
393
+ }
394
+
387
395
  class MarkdownRenderer {
388
396
  out: string[] = [];
389
397
  private skipDepth = 0;
@@ -524,8 +532,8 @@ class MarkdownRenderer {
524
532
  }
525
533
 
526
534
  private finishLink(): void {
527
- const text = this.linkTextParts.join("").replace(/\s+/g, " ").trim();
528
- const headingText = this.linkHeadingParts.join("").replace(/\s+/g, " ").trim();
535
+ const text = collapseWhitespace(this.linkTextParts.join(""));
536
+ const headingText = collapseWhitespace(this.linkHeadingParts.join(""));
529
537
  const href = this.linkHref ?? "";
530
538
  this.inLink = false;
531
539
  this.linkTextParts = [];
@@ -571,12 +579,8 @@ class MarkdownRenderer {
571
579
  }
572
580
  if (closeAt === null) break;
573
581
  this.truncateOpenTags(closeAt);
574
- while (this.hiddenMarks.length && this.hiddenMarks[this.hiddenMarks.length - 1] >= closeAt) {
575
- this.hiddenMarks.pop();
576
- }
577
- while (this.headingMarks.length && this.headingMarks[this.headingMarks.length - 1] >= closeAt) {
578
- this.headingMarks.pop();
579
- }
582
+ popMarksAbove(this.hiddenMarks, closeAt);
583
+ popMarksAbove(this.headingMarks, closeAt);
580
584
  this.closeHeaderFrames(closeAt);
581
585
  }
582
586
  }
@@ -692,12 +696,8 @@ class MarkdownRenderer {
692
696
  for (let i = this.openTags.length - 1; i >= 0; i--) {
693
697
  if (this.openTags[i] === tag) {
694
698
  this.truncateOpenTags(i);
695
- while (this.hiddenMarks.length && this.hiddenMarks[this.hiddenMarks.length - 1] >= i) {
696
- this.hiddenMarks.pop();
697
- }
698
- while (this.headingMarks.length && this.headingMarks[this.headingMarks.length - 1] >= i) {
699
- this.headingMarks.pop();
700
- }
699
+ popMarksAbove(this.hiddenMarks, i);
700
+ popMarksAbove(this.headingMarks, i);
701
701
  this.closeHeaderFrames(i, tag === "header");
702
702
  break;
703
703
  }
@@ -715,7 +715,11 @@ class MarkdownRenderer {
715
715
  return !suppressed;
716
716
  }
717
717
 
718
- handleStartEndTag(_name: string, _attrs: AttrDict): void {}
718
+ handleStartEndTag(name: string, attrs: AttrDict): void {
719
+ if (!VOID_TAGS.has(name)) return;
720
+ this.handleStartTag(name, attrs);
721
+ this.handleEndTag(name);
722
+ }
719
723
  handleStartTag(tag: string, attrs: AttrDict): void {
720
724
  if (this.skipDepth) {
721
725
  if (SKIP_TAGS.has(tag)) this.skipDepth++;
package/index.ts CHANGED
@@ -1,9 +1,11 @@
1
1
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
2
2
  import { Type } from "typebox";
3
3
  import { webSearch } from "./web-search.ts";
4
- import { fetchPageText } from "./web-fetch.ts";
4
+ import { DEFAULT_FETCH_TIMEOUT_MS, fetchPageText } from "./web-fetch.ts";
5
5
 
6
- const FETCH_TIMEOUT_MS = 60_000;
6
+ function positiveMaxChars(value: unknown): number | undefined {
7
+ return typeof value === "number" && value > 0 ? value : undefined;
8
+ }
7
9
 
8
10
  const WebSearchParams = Type.Object({
9
11
  query: Type.Optional(
@@ -15,6 +17,12 @@ const WebSearchParams = Type.Object({
15
17
  "A URL to fetch full page content from (instead of searching). Use this to read a page found in search results.",
16
18
  }),
17
19
  ),
20
+ maxChars: Type.Optional(
21
+ Type.Number({
22
+ description:
23
+ "Truncate the fetched page to this many characters (only used with the url parameter)",
24
+ }),
25
+ ),
18
26
  });
19
27
 
20
28
  const WebFetchParams = Type.Object({
@@ -38,21 +46,25 @@ export default function (pi: ExtensionAPI) {
38
46
  "Use web_search with the url parameter (e.g. {\"url\": \"<URL>\"}) to read the full text of a page found in search results.",
39
47
  ],
40
48
  parameters: WebSearchParams,
41
- async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
49
+ async execute(_toolCallId, params, signal, onUpdate, _ctx) {
42
50
  if (params.url?.trim()) {
51
+ const url = params.url.trim();
52
+ onUpdate?.({ content: [{ type: "text", text: `Fetching ${url}...` }], details: {} });
43
53
  return {
44
54
  content: [
45
55
  {
46
56
  type: "text",
47
- text: await fetchPageText(params.url.trim(), {
48
- timeoutMs: FETCH_TIMEOUT_MS,
57
+ text: await fetchPageText(url, {
58
+ timeoutMs: DEFAULT_FETCH_TIMEOUT_MS,
49
59
  signal: signal ?? undefined,
60
+ maxChars: positiveMaxChars(params.maxChars),
50
61
  }),
51
62
  },
52
63
  ],
53
64
  details: {},
54
65
  };
55
66
  }
67
+ onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
56
68
  const text = await webSearch(params.query, { signal: signal ?? undefined });
57
69
  return { content: [{ type: "text", text }], details: {} };
58
70
  },
@@ -70,13 +82,11 @@ export default function (pi: ExtensionAPI) {
70
82
  "is capped.",
71
83
  promptSnippet: "Fetch a web page and return readable text content",
72
84
  parameters: WebFetchParams,
73
- async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
74
- const maxChars =
75
- typeof params.maxChars === "number" && params.maxChars > 0
76
- ? params.maxChars
77
- : undefined;
85
+ async execute(_toolCallId, params, signal, onUpdate, _ctx) {
86
+ const maxChars = positiveMaxChars(params.maxChars);
87
+ onUpdate?.({ content: [{ type: "text", text: `Fetching ${params.url}...` }], details: {} });
78
88
  const text = await fetchPageText(params.url, {
79
- timeoutMs: FETCH_TIMEOUT_MS,
89
+ timeoutMs: DEFAULT_FETCH_TIMEOUT_MS,
80
90
  signal: signal ?? undefined,
81
91
  maxChars,
82
92
  });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-unsloth-webtools",
3
- "version": "0.2.4",
3
+ "version": "0.3.0",
4
4
  "type": "module",
5
5
  "description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
6
6
  "main": "index.ts",
@@ -51,7 +51,8 @@
51
51
  "test:unit": "vitest run --exclude **/smoke.test.ts",
52
52
  "test:smoke": "vitest run test/smoke.test.ts",
53
53
  "typecheck": "tsc --noEmit",
54
- "prepublishOnly": "npm run typecheck && npm run test:unit"
54
+ "check:package": "node -e \"const fs=require('fs');const pkg=require('./package.json');const listed=new Set(pkg.files);const bad=[];for(const f of pkg.files){if(!fs.existsSync(f))bad.push('missing file: '+f)}for(const f of fs.readdirSync('.').filter(f=>f.endsWith('.ts'))){if(!listed.has(f))bad.push('unlisted source: '+f)}if(bad.length){console.error(bad.join('\\n'));process.exit(1)}\"",
55
+ "prepublishOnly": "npm run typecheck && npm run test:unit && npm run check:package"
55
56
  },
56
57
  "devDependencies": {
57
58
  "@earendil-works/pi-coding-agent": "^0.84.0",