pi-unsloth-webtools 0.2.5 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -26,33 +26,60 @@ Mirrors Unsloth Studio's `web_search` tool:
26
26
 
27
27
  - Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
28
28
  (duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
29
- the same provider deduplication, href-dedupe aggregator with frequency ordering, and the
30
- same `SimpleFilterRanker` re-ranking. Formats results identically: `Title:` / `URL:` /
29
+ the same provider deduplication, href-dedupe aggregator with frequency ordering (hrefs are
30
+ canonicalized first — `utm_*`/tracking parameters and fragments are dropped and the URL is
31
+ re-serialized, collapsing host-case, default-port, and trailing-slash variants — so the
32
+ same page found via different tracking links collapses), and the same `SimpleFilterRanker`
33
+ re-ranking. Formats results identically: `Title:` / `URL:` /
31
34
  `Snippet:` blocks separated by `---`, ending with the hint to pass `{"url": "<URL>"}` to
32
35
  read a full page.
33
- - Accepts an optional `url` parameter; when given, fetches that page's text instead of searching.
36
+ - Accepts an optional `url` parameter; when given, fetches that page's text instead of
37
+ searching (optionally truncated with `maxChars`).
34
38
  - Rate-limit, timeout, and empty-result messages mirror Studio's `_search_failure_message`.
39
+ - Transient engine failures (network errors or null responses) are retried once with a short
40
+ backoff inside the same timeout budget (a retry that cannot fit in the remaining budget is
41
+ skipped); timeouts and cancellations are never retried.
42
+ - Sweeps stop as soon as enough results are gathered: engines still in flight are aborted
43
+ instead of being allowed to run to their timeout.
35
44
 
36
45
  ### web_fetch
37
46
 
38
47
  Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
39
48
 
40
49
  - URL scheme normalization (bare hosts like `google.com` become `https://google.com`).
41
- - URL validation: http/https only, no credentials or encoded hostnames, hostname/port checks.
50
+ - URL validation: http/https only, no credentials or encoded hostnames, hostname/port checks (any
51
+ port 1–65535 is permitted; SSRF protection is enforced at the resolved-IP layer, not by port
52
+ allowlists).
53
+ Canonical public IPv4 literals are accepted like IPv6 literals; private literals are still
54
+ blocked at the resolved-IP layer.
42
55
  - DNS resolution with SSRF protection: every resolved address is validated against
43
56
  private/loopback/link-local/CGNAT/documentation/multicast/reserved ranges, then the validated IP
44
57
  is pinned for the connection (custom `lookup` + SNI `servername`), so DNS cannot rebind between
45
- validation and fetch.
58
+ validation and fetch; resolution shares the caller's abort signal and the overall deadline,
59
+ so a stuck resolver cannot outlive the fetch.
60
+ When a host publishes both IPv4 and IPv6 addresses, IPv4 is preferred (broken IPv6 routes
61
+ cannot stall a fetch), and a connection failure falls back to the next validated address for
62
+ the same host before giving up.
46
63
  - GitHub repo root pages are rewritten to the unauthenticated README API
47
- (`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
48
- - HTTP 4xx/5xx responses are reported as errors with the status reason instead of
64
+ (`Accept: application/vnd.github.raw+json`), falling back to the raw README URL
65
+ (`raw.githubusercontent.com`, no API rate limit) and then to the HTML page on failure.
49
66
  returning error-page content.
50
67
  - Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
51
68
  - 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
52
- (`signal` cancels mid-flight).
69
+ (`signal` cancels mid-flight). Fetches cut off by a download cap are marked with a trailing
70
+ truncation notice, so a partial page is not mistaken for a complete one.
71
+ - Responses sent with a `Content-Encoding` of gzip, deflate, or brotli (servers that ignore the
72
+ `Accept-Encoding: identity` request) are decompressed while streaming, so the download caps
73
+ bound the decoded page text (a gzip'd PDF still gets the 10 MiB PDF budget via magic sniffing on
74
+ the decoded head) and a stream cut by the cap or a mid-stream decode failure returns the readable
75
+ decoded prefix with the truncation notice instead of a binary-content error. A 64 MiB
76
+ decoded-output cap bounds a compressed bomb.
77
+ - Long fetches and searches report a short progress note to the session before they start, so
78
+ slow tool calls are not silent.
53
79
  - PDF text extraction via the official MuPDF.js engine (the same C library pymupdf wraps):
54
80
  object streams, all filters, ToUnicode fonts, encryption detection, and a
55
- pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
81
+ pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables),
82
+ running header/footer and page-number stripping,
56
83
  with Studio's corrupted/incomplete fallback to plain text.
57
84
  - Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
58
85
  decoding (BOM sniffing for UTF-8/16/32 first, then the declared charset, `<meta charset>` sniffing for
@@ -62,10 +89,16 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
62
89
  stripping (`hidden`, `aria-hidden`, inline styles); `<article>`/`<main>` main-content scoping
63
90
  with link-density header stripping; boilerplate-line removal.
64
91
  - No page-size budget: fetched pages and PDFs are returned in full (Studio's window-aware
65
- cap is deliberately dropped; the optional `maxChars` parameter still truncates when given).
92
+ cap is deliberately dropped; the optional `maxChars` parameter still truncates when given,
93
+ on `web_fetch` and on `web_search`'s url mode).
66
94
  The 512 KiB / 10 MiB download caps still bound the raw fetch.
67
95
  - HTML entity decoding replicates CPython's `html.unescape` (full 2,231-entry HTML5 table,
68
96
  longest-prefix rule, Windows-1252 numeric mappings), matching Studio byte-for-byte.
97
+ - Fetched HTML pages are prefixed with the decoded document `<title>`, so the model can
98
+ see which page it is reading. `Author:` (`meta name=author` / `article:author` / `dc.creator`),
99
+ `Date:` (`article:published_time` / `dc.date` / `date`) and `Site:` (`og:site_name` /
100
+ `application-name`) lines are added when declared, so the model can judge recency and
101
+ provenance.
69
102
 
70
103
  ## Known differences from Studio
71
104
 
@@ -73,6 +106,11 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
73
106
  whole line instead of per-span; superscript, subscript, underline, strikeout, and
74
107
  highlight markers are not emitted. Tables use a conservative text-grid detector:
75
108
  aligned text tables are detected, drawn-rule-only tables are not.
109
+ - Running headers and footers: lines repeated at the same page-edge position on at
110
+ least half the pages (two pages minimum) are dropped from the markdown layer, as is
111
+ any numeric-only line at a fixed edge position where page numbers appear on at least
112
+ half the pages (so a one-off number sharing that position is dropped too, while fused
113
+ labels like `Page 3 of 12` survive). Studio and pymupdf4llm return them verbatim.
76
114
  - Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
77
115
  impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
78
116
  aggressively (a blocked engine simply contributes no results). User agents are a
@@ -85,6 +123,13 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
85
123
  worst-case wall time.
86
124
  - Proxies: Studio routes through environment proxies; this port always connects
87
125
  directly with DNS pinning (deliberately out of scope).
126
+ - Dedup and titles: the aggregator keys on canonicalized hrefs (`utm_*`/tracking parameters
127
+ and fragments stripped, then the URL re-serialized); fetched HTML pages are prefixed with
128
+ the document `<title>`. Studio keys on raw hrefs and returns the converted body alone.
129
+ - Upstream drift: current ddgs ships ten backends (adding bing, startpage, grokipedia),
130
+ requires a `vqd` token for DuckDuckGo, and exposes an `extract()` mode. This port
131
+ deliberately pins the Studio snapshot — seven engines, bing disabled upstream, no vqd,
132
+ no pagination — so engine behavior matches Studio rather than ddgs head.
88
133
 
89
134
  ## Development
90
135
 
@@ -119,7 +164,8 @@ The suite ports Unsloth Studio's own tests for these tools:
119
164
  ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
120
165
  - `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
121
166
  legacy refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
122
- - `test/smoke.test.ts`: live network checks against real hosts
167
+ - `test/smoke.test.ts`: live network checks against real hosts, including a per-engine
168
+ result-health sweep (at least three engines must return well-formed results)
123
169
 
124
170
  The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
125
171
  with fakes, mirroring how the Studio suite monkeypatches `_validate_and_resolve_host`
package/engines.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { randomBytes } from "node:crypto";
2
- import { decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
2
+ import { collapseWhitespace, decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
3
3
  import type { AttrDict } from "./html-to-md.ts";
4
4
  import { randomUserAgent } from "./user-agents.ts";
5
5
  export class EmptySweepError extends Error {
@@ -35,7 +35,7 @@ export function normalizeText(raw: string): string {
35
35
  text = decodeHtmlEntities(text);
36
36
  text = text.normalize("NFC");
37
37
  text = text.replace(/[\p{Cc}\p{Cf}\p{Co}\p{Cs}\p{Cn}]/gu, "");
38
- return text.trim().split(/\s+/).join(" ");
38
+ return collapseWhitespace(text);
39
39
  }
40
40
 
41
41
  export function normalizeUrl(url: string): string {
@@ -47,6 +47,41 @@ export function normalizeUrl(url: string): string {
47
47
  }
48
48
  }
49
49
 
50
+ const TRACKING_PARAM_NAMES = new Set([
51
+ "_hsenc",
52
+ "_hsmi",
53
+ "dclid",
54
+ "fbclid",
55
+ "gbraid",
56
+ "gclid",
57
+ "gclsrc",
58
+ "igshid",
59
+ "mc_cid",
60
+ "mc_eid",
61
+ "msclkid",
62
+ "srsltid",
63
+ "twclid",
64
+ "wbraid",
65
+ "yclid",
66
+ ]);
67
+
68
+ export function canonicalizeHref(href: string): string {
69
+ if (!href) return "";
70
+ try {
71
+ const url = new URL(href);
72
+ for (const key of [...url.searchParams.keys()]) {
73
+ const lower = key.toLowerCase();
74
+ if (lower.startsWith("utm_") || TRACKING_PARAM_NAMES.has(lower)) {
75
+ url.searchParams.delete(key);
76
+ }
77
+ }
78
+ url.hash = "";
79
+ return url.toString();
80
+ } catch {
81
+ return href;
82
+ }
83
+ }
84
+
50
85
  export interface DomNode {
51
86
  tag: string;
52
87
  attrs: Record<string, string>;
@@ -382,7 +417,7 @@ export function extractResults(
382
417
  ["body", elementsXpath.body],
383
418
  ] as const;
384
419
  for (const [key, value] of entries) {
385
- const data = xpathText(value, item).join("").trim().split(/\s+/).join(" ");
420
+ const data = collapseWhitespace(xpathText(value, item).join(""));
386
421
  if (!data) continue;
387
422
  result[key] = key === "href" ? normalizeUrl(data) : normalizeText(data);
388
423
  }
@@ -444,15 +479,22 @@ export interface Engine {
444
479
  ): Promise<SearchResult[] | null>;
445
480
  }
446
481
 
482
+ interface HttpRequestOptions {
483
+ headers?: Record<string, string>;
484
+ cookies?: Record<string, string>;
485
+ timeoutMs: number;
486
+ signal?: AbortSignal;
487
+ }
488
+
489
+ interface HttpOptions extends HttpRequestOptions {
490
+ method?: string;
491
+ body?: string;
492
+ }
493
+
447
494
  async function httpGet(
448
495
  url: string,
449
496
  params: Record<string, string>,
450
- options: {
451
- headers?: Record<string, string>;
452
- cookies?: Record<string, string>;
453
- timeoutMs: number;
454
- signal?: AbortSignal;
455
- },
497
+ options: HttpRequestOptions,
456
498
  ): Promise<string | null> {
457
499
  const target = new URL(url);
458
500
  for (const [key, value] of Object.entries(params)) target.searchParams.set(key, value);
@@ -462,17 +504,15 @@ async function httpGet(
462
504
  async function httpPost(
463
505
  url: string,
464
506
  data: Record<string, string>,
465
- options: {
466
- headers?: Record<string, string>;
467
- cookies?: Record<string, string>;
468
- timeoutMs: number;
469
- signal?: AbortSignal;
470
- },
507
+ options: HttpRequestOptions,
471
508
  ): Promise<string | null> {
472
509
  return httpFetch(url, { ...options, method: "POST", body: new URLSearchParams(data).toString() });
473
510
  }
474
511
 
475
512
  const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
513
+ const ENGINE_RETRY_BACKOFF_MS = 250;
514
+
515
+ const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
476
516
 
477
517
  async function readBodyCapped(response: Response): Promise<string | null> {
478
518
  const declared = Number(response.headers.get("content-length") ?? "0");
@@ -494,16 +534,15 @@ async function readBodyCapped(response: Response): Promise<string | null> {
494
534
  return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
495
535
  }
496
536
 
537
+ function mapFetchError(err: unknown): never {
538
+ if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
539
+ if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
540
+ throw err;
541
+ }
542
+
497
543
  async function httpFetch(
498
544
  url: string,
499
- options: {
500
- method?: string;
501
- body?: string;
502
- headers?: Record<string, string>;
503
- cookies?: Record<string, string>;
504
- timeoutMs: number;
505
- signal?: AbortSignal;
506
- },
545
+ options: HttpOptions,
507
546
  ): Promise<string | null> {
508
547
  const headers: Record<string, string> = {
509
548
  "User-Agent": options.headers?.["User-Agent"] ?? randomUserAgent(),
@@ -528,17 +567,20 @@ async function httpFetch(
528
567
  signal: AbortSignal.any(signals),
529
568
  });
530
569
  } catch (err) {
531
- if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
532
- if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
533
- throw err;
570
+ throw mapFetchError(err);
571
+ }
572
+ if (response.status !== 200) {
573
+ try {
574
+ await response.body?.cancel();
575
+ } catch {
576
+ return null;
577
+ }
578
+ return null;
534
579
  }
535
- if (response.status !== 200) return null;
536
580
  try {
537
581
  return await readBodyCapped(response);
538
582
  } catch (err) {
539
- if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
540
- if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
541
- throw err;
583
+ throw mapFetchError(err);
542
584
  }
543
585
  }
544
586
 
@@ -649,7 +691,7 @@ const MOJEEK: Engine = {
649
691
  const YAHOO: Engine = {
650
692
  name: "yahoo",
651
693
  provider: "bing",
652
- async search(query, ctx, timeoutMs, signal) {
694
+ async search(query, _ctx, timeoutMs, signal) {
653
695
  const ylt = tokenUrlSafe(18);
654
696
  const ylu = tokenUrlSafe(35);
655
697
  const html = await httpGet(
@@ -675,7 +717,7 @@ const YAHOO: Engine = {
675
717
  const YANDEX: Engine = {
676
718
  name: "yandex",
677
719
  provider: "yandex",
678
- async search(query, ctx, timeoutMs, signal) {
720
+ async search(query, _ctx, timeoutMs, signal) {
679
721
  const searchid = 1000000 + Math.floor(Math.random() * 9000000);
680
722
  const html = await httpGet(
681
723
  "https://yandex.com/search/site/",
@@ -734,7 +776,7 @@ const WIKIPEDIA: Engine = {
734
776
  },
735
777
  };
736
778
 
737
- const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
779
+ export const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
738
780
 
739
781
  export class ResultsAggregator {
740
782
  private cache = new Map<string, SearchResult>();
@@ -745,10 +787,10 @@ export class ResultsAggregator {
745
787
  }
746
788
 
747
789
  append(item: SearchResult): void {
748
- const key = item.href;
790
+ const key = canonicalizeHref(item.href);
749
791
  const existing = this.cache.get(key);
750
792
  if (!existing || item.body.length > existing.body.length) {
751
- this.cache.set(key, item);
793
+ this.cache.set(key, { ...item, href: key });
752
794
  }
753
795
  this.counter.set(key, (this.counter.get(key) ?? 0) + 1);
754
796
  }
@@ -821,6 +863,7 @@ export async function autoTextSearch(
821
863
  const seenProviders = new Set<string>();
822
864
  const aggregator = new ResultsAggregator();
823
865
  const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
866
+ const controller = new AbortController();
824
867
  let timedOut = false;
825
868
  let cancelled = false;
826
869
  const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
@@ -828,20 +871,50 @@ export async function autoTextSearch(
828
871
  let i = 0;
829
872
  let pending: Promise<void>[] = [];
830
873
  const run = async (engine: Engine) => {
831
- const remaining = Math.max(1, deadline - Date.now());
832
- try {
833
- const results = await engine.search(query, ctx, remaining, signal);
834
- if (results && results.length) {
835
- aggregator.extend(results);
836
- seenProviders.add(engine.provider);
874
+ let results: SearchResult[] | null = null;
875
+ for (let attempt = 0; attempt < 2 && results === null; attempt++) {
876
+ if (controller.signal.aborted) return;
877
+ const budgetLeft = deadline - Date.now();
878
+ if (budgetLeft <= 0) return;
879
+ if (attempt > 0 && budgetLeft < ENGINE_RETRY_BACKOFF_MS) return;
880
+ const remaining = Math.max(1, budgetLeft);
881
+ const engineSignal = signal
882
+ ? AbortSignal.any([signal, controller.signal])
883
+ : controller.signal;
884
+ try {
885
+ results = await engine.search(query, ctx, remaining, engineSignal);
886
+ } catch (e) {
887
+ if (e instanceof SearchCancelled) {
888
+ if (controller.signal.aborted && !signal?.aborted) return;
889
+ cancelled = true;
890
+ return;
891
+ }
892
+ if (e instanceof SearchTimeoutError) {
893
+ timedOut = true;
894
+ return;
895
+ }
896
+ }
897
+ if (results === null && attempt === 0) {
898
+ const backoff = Math.min(ENGINE_RETRY_BACKOFF_MS, Math.max(0, deadline - Date.now()));
899
+ if (backoff > 0) await sleep(backoff);
900
+ if (signal?.aborted) {
901
+ cancelled = true;
902
+ return;
903
+ }
904
+ if (controller.signal.aborted) return;
837
905
  }
838
- } catch (e) {
839
- if (e instanceof SearchCancelled) cancelled = true;
840
- if (e instanceof SearchTimeoutError) timedOut = true;
906
+ }
907
+ if (results && results.length) {
908
+ aggregator.extend(results);
909
+ seenProviders.add(engine.provider);
910
+ if (aggregator.size >= maxResults) controller.abort();
841
911
  }
842
912
  };
843
913
  while (i < engines.length) {
844
- if (aggregator.size >= maxResults) break;
914
+ if (aggregator.size >= maxResults || cancelled) {
915
+ controller.abort();
916
+ break;
917
+ }
845
918
  const engine = engines[i++];
846
919
  if (seenProviders.has(engine.provider)) continue;
847
920
  pending.push(run(engine));
package/html-to-md.ts CHANGED
@@ -194,7 +194,7 @@ class HeaderFrame {
194
194
  const CHARREF_RE = /&(#[0-9]+;?|#[xX][0-9a-fA-F]+;?|[^\t\n\f <&#;]{1,32};?)/g;
195
195
 
196
196
  export function decodeHtmlEntities(text: string): string {
197
- return text.replace(CHARREF_RE, (whole, s: string) => {
197
+ return text.replace(CHARREF_RE, (_whole, s: string) => {
198
198
  if (s[0] === "#") {
199
199
  const hex = s[1] === "x" || s[1] === "X";
200
200
  const num = parseInt(s.slice(hex ? 2 : 1).replace(/;+$/, ""), hex ? 16 : 10);
@@ -214,6 +214,10 @@ export function decodeHtmlEntities(text: string): string {
214
214
  });
215
215
  }
216
216
 
217
+ export function collapseWhitespace(text: string): string {
218
+ return text.replace(/\s+/g, " ").trim();
219
+ }
220
+
217
221
 
218
222
  interface HtmlHandlers {
219
223
  handleStartTag(name: string, attrs: AttrDict): void;
@@ -224,8 +228,11 @@ interface HtmlHandlers {
224
228
  handleCharRef(name: string): void;
225
229
  }
226
230
 
227
- const START_TAG_NAME_RE = /^[a-zA-Z][^\s/>]*/;
228
- const ATTR_NAME_RE = /^[^\s=/>]+/;
231
+ const START_TAG_NAME_RE = /[a-zA-Z][^\s/>]*/y;
232
+ const ATTR_NAME_RE = /[^\s=/>]+/y;
233
+ const ENTITY_NAMED_RE = /&([A-Za-z][A-Za-z0-9.-]*);/y;
234
+ const ENTITY_NUMERIC_RE = /&#([xX][0-9a-fA-F]+|[0-9]+);?/y;
235
+ const ENTITY_LEGACY_RE = /&([A-Za-z][A-Za-z0-9.-]*)(?=[^A-Za-z0-9]|$)/y;
229
236
 
230
237
  const RAW_TEXT_TAGS = [
231
238
  "script",
@@ -238,9 +245,23 @@ const RAW_TEXT_TAGS = [
238
245
  "noframes",
239
246
  ];
240
247
 
241
- const RAW_TEXT_CLOSERS: Record<string, RegExp> = Object.fromEntries(
242
- RAW_TEXT_TAGS.map((name) => [name, new RegExp(`</${name}\\s*>`, "i")]),
243
- );
248
+ function findRawTextClose(
249
+ input: string,
250
+ lowerInput: string,
251
+ from: number,
252
+ name: string,
253
+ ): { start: number; end: number } | null {
254
+ const needle = `</${name}`;
255
+ let pos = from;
256
+ while (true) {
257
+ const hit = lowerInput.indexOf(needle, pos);
258
+ if (hit === -1) return null;
259
+ let j = hit + needle.length;
260
+ while (j < input.length && /\s/.test(input[j])) j++;
261
+ if (j < input.length && input[j] === ">") return { start: hit, end: j + 1 };
262
+ pos = hit + 1;
263
+ }
264
+ }
244
265
 
245
266
  function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, boolean] {
246
267
  const attrs: AttrDict = {};
@@ -253,10 +274,11 @@ function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, bo
253
274
  pos++;
254
275
  continue;
255
276
  }
256
- const nameMatch = ATTR_NAME_RE.exec(input.slice(pos));
277
+ ATTR_NAME_RE.lastIndex = pos;
278
+ const nameMatch = ATTR_NAME_RE.exec(input);
257
279
  if (!nameMatch) return [attrs, -1, false];
258
280
  const name = nameMatch[0].toLowerCase();
259
- pos += nameMatch[0].length;
281
+ pos = nameMatch.index + nameMatch[0].length;
260
282
  while (pos < input.length && /\s/.test(input[pos])) pos++;
261
283
  let value: string | null = null;
262
284
  if (pos < input.length && input[pos] === "=") {
@@ -281,64 +303,69 @@ function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, bo
281
303
  }
282
304
 
283
305
  function scanTag(
284
- html: string,
306
+ input: string,
285
307
  i: number,
286
308
  ): { end: number; kind: "comment" | "decl" | "end" | "start" | "startend"; name?: string; attrs?: AttrDict } | null {
287
- const rest = html.slice(i + 1);
288
- if (rest.startsWith("!--")) {
289
- const close = html.indexOf("-->", i + 4);
309
+ if (input.startsWith("!--", i + 1)) {
310
+ const close = input.indexOf("-->", i + 4);
290
311
  if (close === -1) return null;
291
312
  return { end: close + 3, kind: "decl" };
292
313
  }
293
- if (rest.startsWith("!") || rest.startsWith("?")) {
314
+ const next = input[i + 1];
315
+ if (next === "!" || next === "?") {
294
316
  let j = i + 2;
295
- while (j < html.length && html[j] !== ">") j++;
296
- if (j >= html.length) return null;
317
+ while (j < input.length && input[j] !== ">") j++;
318
+ if (j >= input.length) return null;
297
319
  return { end: j + 1, kind: "decl" };
298
320
  }
299
- if (rest.startsWith("/")) {
321
+ if (next === "/") {
300
322
  let j = i + 2;
301
- while (j < html.length && /[\s>]/.test(html[j]) === false) j++;
302
- const name = html.slice(i + 2, j).toLowerCase();
323
+ while (j < input.length && !/[\s>]/.test(input[j])) j++;
324
+ const name = input.slice(i + 2, j).toLowerCase();
303
325
  if (!name) return null;
304
- while (j < html.length && html[j] !== ">") j++;
305
- if (j >= html.length) return null;
326
+ while (j < input.length && input[j] !== ">") j++;
327
+ if (j >= input.length) return null;
306
328
  return { end: j + 1, kind: "end", name };
307
329
  }
308
- const nameMatch = START_TAG_NAME_RE.exec(rest);
330
+ START_TAG_NAME_RE.lastIndex = i + 1;
331
+ const nameMatch = START_TAG_NAME_RE.exec(input);
309
332
  if (!nameMatch) return null;
310
333
  const name = nameMatch[0].toLowerCase();
311
- const [attrs, next, selfClosing] = parseAttrsUntilClose(html, i + 1 + nameMatch[0].length);
312
- if (next === -1) return null;
313
- return { end: next, kind: selfClosing ? "startend" : "start", name, attrs };
334
+ const [attrs, nextPos, selfClosing] = parseAttrsUntilClose(input, i + 1 + nameMatch[0].length);
335
+ if (nextPos === -1) return null;
336
+ return { end: nextPos, kind: selfClosing ? "startend" : "start", name, attrs };
314
337
  }
315
338
 
316
339
 
317
340
  export function feedHtml(input: string, handlers: HtmlHandlers): void {
318
- const emitText = (text: string) => {
319
- if (!text) return;
320
- let pos = 0;
321
- while (pos < text.length) {
322
- const amp = text.indexOf("&", pos);
323
- if (amp === -1) {
324
- handlers.handleData(text.slice(pos));
341
+ let lowerInput: string | null = null;
342
+ const emitTextRange = (start: number, end: number) => {
343
+ if (end <= start) return;
344
+ let pos = start;
345
+ while (pos < end) {
346
+ const amp = input.indexOf("&", pos);
347
+ if (amp === -1 || amp >= end) {
348
+ handlers.handleData(input.slice(pos, end));
325
349
  return;
326
350
  }
327
- if (amp > pos) handlers.handleData(text.slice(pos, amp));
328
- const named = /^&([A-Za-z][A-Za-z0-9.-]*);/.exec(text.slice(amp));
329
- if (named) {
351
+ if (amp > pos) handlers.handleData(input.slice(pos, amp));
352
+ ENTITY_NAMED_RE.lastIndex = amp;
353
+ const named = ENTITY_NAMED_RE.exec(input);
354
+ if (named && named.index === amp && amp + named[0].length <= end) {
330
355
  handlers.handleEntityRef(named[1]);
331
356
  pos = amp + named[0].length;
332
357
  continue;
333
358
  }
334
- const numeric = /^&#(?:[xX]([0-9a-fA-F]+)|([0-9]+));/.exec(text.slice(amp));
335
- if (numeric) {
336
- handlers.handleCharRef(numeric[1] ?? numeric[2]);
359
+ ENTITY_NUMERIC_RE.lastIndex = amp;
360
+ const numeric = ENTITY_NUMERIC_RE.exec(input);
361
+ if (numeric && numeric.index === amp && amp + numeric[0].length <= end) {
362
+ handlers.handleCharRef(numeric[1]);
337
363
  pos = amp + numeric[0].length;
338
364
  continue;
339
365
  }
340
- const legacy = /^&([A-Za-z][A-Za-z0-9.-]*)(?=[^A-Za-z0-9]|$)/.exec(text.slice(amp));
341
- if (legacy) {
366
+ ENTITY_LEGACY_RE.lastIndex = amp;
367
+ const legacy = ENTITY_LEGACY_RE.exec(input);
368
+ if (legacy && legacy.index === amp && amp + legacy[0].length <= end) {
342
369
  handlers.handleEntityRef(legacy[1]);
343
370
  pos = amp + legacy[0].length;
344
371
  continue;
@@ -360,28 +387,32 @@ export function feedHtml(input: string, handlers: HtmlHandlers): void {
360
387
  i++;
361
388
  continue;
362
389
  }
363
- emitText(input.slice(textStart, i));
390
+ emitTextRange(textStart, i);
364
391
  if (tag.kind === "start") {
365
- const rawCloser = RAW_TEXT_CLOSERS[tag.name!];
366
- if (rawCloser) {
367
- const after = input.slice(tag.end);
368
- const closeMatch = rawCloser.exec(after);
369
- if (closeMatch) {
370
- handlers.handleStartTag(tag.name!, tag.attrs!);
371
- emitText(after.slice(0, closeMatch.index));
372
- handlers.handleEndTag(tag.name!);
373
- textStart = tag.end + closeMatch.index + closeMatch[0].length;
392
+ const rawName = tag.name!;
393
+ if (RAW_TEXT_TAGS.includes(rawName)) {
394
+ lowerInput ??= input.toLowerCase();
395
+ const close = findRawTextClose(input, lowerInput, tag.end, rawName);
396
+ if (close) {
397
+ handlers.handleStartTag(rawName, tag.attrs!);
398
+ emitTextRange(tag.end, close.start);
399
+ handlers.handleEndTag(rawName);
400
+ textStart = close.end;
374
401
  i = textStart;
375
402
  continue;
376
403
  }
377
404
  }
378
- handlers.handleStartTag(tag.name!, tag.attrs!);
405
+ handlers.handleStartTag(rawName, tag.attrs!);
379
406
  } else if (tag.kind === "startend") handlers.handleStartEndTag(tag.name!, tag.attrs!);
380
407
  else if (tag.kind === "end") handlers.handleEndTag(tag.name!);
381
408
  textStart = tag.end;
382
409
  i = tag.end;
383
410
  }
384
- emitText(input.slice(textStart));
411
+ emitTextRange(textStart, input.length);
412
+ }
413
+
414
+ function popMarksAbove(marks: number[], index: number): void {
415
+ while (marks.length && marks[marks.length - 1] >= index) marks.pop();
385
416
  }
386
417
 
387
418
  class MarkdownRenderer {
@@ -524,8 +555,8 @@ class MarkdownRenderer {
524
555
  }
525
556
 
526
557
  private finishLink(): void {
527
- const text = this.linkTextParts.join("").replace(/\s+/g, " ").trim();
528
- const headingText = this.linkHeadingParts.join("").replace(/\s+/g, " ").trim();
558
+ const text = collapseWhitespace(this.linkTextParts.join(""));
559
+ const headingText = collapseWhitespace(this.linkHeadingParts.join(""));
529
560
  const href = this.linkHref ?? "";
530
561
  this.inLink = false;
531
562
  this.linkTextParts = [];
@@ -571,12 +602,8 @@ class MarkdownRenderer {
571
602
  }
572
603
  if (closeAt === null) break;
573
604
  this.truncateOpenTags(closeAt);
574
- while (this.hiddenMarks.length && this.hiddenMarks[this.hiddenMarks.length - 1] >= closeAt) {
575
- this.hiddenMarks.pop();
576
- }
577
- while (this.headingMarks.length && this.headingMarks[this.headingMarks.length - 1] >= closeAt) {
578
- this.headingMarks.pop();
579
- }
605
+ popMarksAbove(this.hiddenMarks, closeAt);
606
+ popMarksAbove(this.headingMarks, closeAt);
580
607
  this.closeHeaderFrames(closeAt);
581
608
  }
582
609
  }
@@ -692,12 +719,8 @@ class MarkdownRenderer {
692
719
  for (let i = this.openTags.length - 1; i >= 0; i--) {
693
720
  if (this.openTags[i] === tag) {
694
721
  this.truncateOpenTags(i);
695
- while (this.hiddenMarks.length && this.hiddenMarks[this.hiddenMarks.length - 1] >= i) {
696
- this.hiddenMarks.pop();
697
- }
698
- while (this.headingMarks.length && this.headingMarks[this.headingMarks.length - 1] >= i) {
699
- this.headingMarks.pop();
700
- }
722
+ popMarksAbove(this.hiddenMarks, i);
723
+ popMarksAbove(this.headingMarks, i);
701
724
  this.closeHeaderFrames(i, tag === "header");
702
725
  break;
703
726
  }
@@ -715,7 +738,11 @@ class MarkdownRenderer {
715
738
  return !suppressed;
716
739
  }
717
740
 
718
- handleStartEndTag(_name: string, _attrs: AttrDict): void {}
741
+ handleStartEndTag(name: string, attrs: AttrDict): void {
742
+ if (!VOID_TAGS.has(name)) return;
743
+ this.handleStartTag(name, attrs);
744
+ this.handleEndTag(name);
745
+ }
719
746
  handleStartTag(tag: string, attrs: AttrDict): void {
720
747
  if (this.skipDepth) {
721
748
  if (SKIP_TAGS.has(tag)) this.skipDepth++;