pi-unsloth-webtools 0.2.5 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +57 -11
- package/engines.ts +118 -45
- package/html-to-md.ts +93 -66
- package/index.ts +106 -60
- package/package.json +1 -1
- package/pdf.ts +97 -11
- package/web-access.ts +73 -50
- package/web-fetch.ts +401 -95
- package/web-search.ts +5 -4
package/README.md
CHANGED
|
@@ -26,33 +26,60 @@ Mirrors Unsloth Studio's `web_search` tool:
|
|
|
26
26
|
|
|
27
27
|
- Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
|
|
28
28
|
(duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
|
|
29
|
-
the same provider deduplication, href-dedupe aggregator with frequency ordering
|
|
30
|
-
|
|
29
|
+
the same provider deduplication, href-dedupe aggregator with frequency ordering (hrefs are
|
|
30
|
+
canonicalized first — `utm_*`/tracking parameters and fragments are dropped and the URL is
|
|
31
|
+
re-serialized, collapsing host-case, default-port, and trailing-slash variants — so the
|
|
32
|
+
same page found via different tracking links collapses), and the same `SimpleFilterRanker`
|
|
33
|
+
re-ranking. Formats results identically: `Title:` / `URL:` /
|
|
31
34
|
`Snippet:` blocks separated by `---`, ending with the hint to pass `{"url": "<URL>"}` to
|
|
32
35
|
read a full page.
|
|
33
|
-
- Accepts an optional `url` parameter; when given, fetches that page's text instead of
|
|
36
|
+
- Accepts an optional `url` parameter; when given, fetches that page's text instead of
|
|
37
|
+
searching (optionally truncated with `maxChars`).
|
|
34
38
|
- Rate-limit, timeout, and empty-result messages mirror Studio's `_search_failure_message`.
|
|
39
|
+
- Transient engine failures (network errors or null responses) are retried once with a short
|
|
40
|
+
backoff inside the same timeout budget (a retry that cannot fit in the remaining budget is
|
|
41
|
+
skipped); timeouts and cancellations are never retried.
|
|
42
|
+
- Sweeps stop as soon as enough results are gathered: engines still in flight are aborted
|
|
43
|
+
instead of being allowed to run to their timeout.
|
|
35
44
|
|
|
36
45
|
### web_fetch
|
|
37
46
|
|
|
38
47
|
Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
39
48
|
|
|
40
49
|
- URL scheme normalization (bare hosts like `google.com` become `https://google.com`).
|
|
41
|
-
- URL validation: http/https only, no credentials or encoded hostnames, hostname/port checks
|
|
50
|
+
- URL validation: http/https only, no credentials or encoded hostnames, hostname/port checks (any
|
|
51
|
+
port 1–65535 is permitted; SSRF protection is enforced at the resolved-IP layer, not by port
|
|
52
|
+
allowlists).
|
|
53
|
+
Canonical public IPv4 literals are accepted like IPv6 literals; private literals are still
|
|
54
|
+
blocked at the resolved-IP layer.
|
|
42
55
|
- DNS resolution with SSRF protection: every resolved address is validated against
|
|
43
56
|
private/loopback/link-local/CGNAT/documentation/multicast/reserved ranges, then the validated IP
|
|
44
57
|
is pinned for the connection (custom `lookup` + SNI `servername`), so DNS cannot rebind between
|
|
45
|
-
validation and fetch
|
|
58
|
+
validation and fetch; resolution shares the caller's abort signal and the overall deadline,
|
|
59
|
+
so a stuck resolver cannot outlive the fetch.
|
|
60
|
+
When a host publishes both IPv4 and IPv6 addresses, IPv4 is preferred (broken IPv6 routes
|
|
61
|
+
cannot stall a fetch), and a connection failure falls back to the next validated address for
|
|
62
|
+
the same host before giving up.
|
|
46
63
|
- GitHub repo root pages are rewritten to the unauthenticated README API
|
|
47
|
-
(`Accept: application/vnd.github.raw+json`), falling back to the
|
|
48
|
-
|
|
64
|
+
(`Accept: application/vnd.github.raw+json`), falling back to the raw README URL
|
|
65
|
+
(`raw.githubusercontent.com`, no API rate limit) and then to the HTML page on failure.
|
|
49
66
|
returning error-page content.
|
|
50
67
|
- Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
|
|
51
68
|
- 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
|
|
52
|
-
(`signal` cancels mid-flight).
|
|
69
|
+
(`signal` cancels mid-flight). Fetches cut off by a download cap are marked with a trailing
|
|
70
|
+
truncation notice, so a partial page is not mistaken for a complete one.
|
|
71
|
+
- Responses sent with a `Content-Encoding` of gzip, deflate, or brotli (servers that ignore the
|
|
72
|
+
`Accept-Encoding: identity` request) are decompressed while streaming, so the download caps
|
|
73
|
+
bound the decoded page text (a gzip'd PDF still gets the 10 MiB PDF budget via magic sniffing on
|
|
74
|
+
the decoded head) and a stream cut by the cap or a mid-stream decode failure returns the readable
|
|
75
|
+
decoded prefix with the truncation notice instead of a binary-content error. A 64 MiB
|
|
76
|
+
decoded-output cap bounds a compressed bomb.
|
|
77
|
+
- Long fetches and searches report a short progress note to the session before they start, so
|
|
78
|
+
slow tool calls are not silent.
|
|
53
79
|
- PDF text extraction via the official MuPDF.js engine (the same C library pymupdf wraps):
|
|
54
80
|
object streams, all filters, ToUnicode fonts, encryption detection, and a
|
|
55
|
-
pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
|
|
81
|
+
pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables),
|
|
82
|
+
running header/footer and page-number stripping,
|
|
56
83
|
with Studio's corrupted/incomplete fallback to plain text.
|
|
57
84
|
- Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
|
|
58
85
|
decoding (BOM sniffing for UTF-8/16/32 first, then the declared charset, `<meta charset>` sniffing for
|
|
@@ -62,10 +89,16 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
62
89
|
stripping (`hidden`, `aria-hidden`, inline styles); `<article>`/`<main>` main-content scoping
|
|
63
90
|
with link-density header stripping; boilerplate-line removal.
|
|
64
91
|
- No page-size budget: fetched pages and PDFs are returned in full (Studio's window-aware
|
|
65
|
-
cap is deliberately dropped; the optional `maxChars` parameter still truncates when given
|
|
92
|
+
cap is deliberately dropped; the optional `maxChars` parameter still truncates when given,
|
|
93
|
+
on `web_fetch` and on `web_search`'s url mode).
|
|
66
94
|
The 512 KiB / 10 MiB download caps still bound the raw fetch.
|
|
67
95
|
- HTML entity decoding replicates CPython's `html.unescape` (full 2,231-entry HTML5 table,
|
|
68
96
|
longest-prefix rule, Windows-1252 numeric mappings), matching Studio byte-for-byte.
|
|
97
|
+
- Fetched HTML pages are prefixed with the decoded document `<title>`, so the model can
|
|
98
|
+
see which page it is reading. `Author:` (`meta name=author` / `article:author` / `dc.creator`),
|
|
99
|
+
`Date:` (`article:published_time` / `dc.date` / `date`) and `Site:` (`og:site_name` /
|
|
100
|
+
`application-name`) lines are added when declared, so the model can judge recency and
|
|
101
|
+
provenance.
|
|
69
102
|
|
|
70
103
|
## Known differences from Studio
|
|
71
104
|
|
|
@@ -73,6 +106,11 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
73
106
|
whole line instead of per-span; superscript, subscript, underline, strikeout, and
|
|
74
107
|
highlight markers are not emitted. Tables use a conservative text-grid detector:
|
|
75
108
|
aligned text tables are detected, drawn-rule-only tables are not.
|
|
109
|
+
- Running headers and footers: lines repeated at the same page-edge position on at
|
|
110
|
+
least half the pages (two pages minimum) are dropped from the markdown layer, as is
|
|
111
|
+
any numeric-only line at a fixed edge position where page numbers appear on at least
|
|
112
|
+
half the pages (so a one-off number sharing that position is dropped too, while fused
|
|
113
|
+
labels like `Page 3 of 12` survive). Studio and pymupdf4llm return them verbatim.
|
|
76
114
|
- Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
|
|
77
115
|
impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
|
|
78
116
|
aggressively (a blocked engine simply contributes no results). User agents are a
|
|
@@ -85,6 +123,13 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
85
123
|
worst-case wall time.
|
|
86
124
|
- Proxies: Studio routes through environment proxies; this port always connects
|
|
87
125
|
directly with DNS pinning (deliberately out of scope).
|
|
126
|
+
- Dedup and titles: the aggregator keys on canonicalized hrefs (`utm_*`/tracking parameters
|
|
127
|
+
and fragments stripped, then the URL re-serialized); fetched HTML pages are prefixed with
|
|
128
|
+
the document `<title>`. Studio keys on raw hrefs and returns the converted body alone.
|
|
129
|
+
- Upstream drift: current ddgs ships ten backends (adding bing, startpage, grokipedia),
|
|
130
|
+
requires a `vqd` token for DuckDuckGo, and exposes an `extract()` mode. This port
|
|
131
|
+
deliberately pins the Studio snapshot — seven engines, bing disabled upstream, no vqd,
|
|
132
|
+
no pagination — so engine behavior matches Studio rather than ddgs head.
|
|
88
133
|
|
|
89
134
|
## Development
|
|
90
135
|
|
|
@@ -119,7 +164,8 @@ The suite ports Unsloth Studio's own tests for these tools:
|
|
|
119
164
|
ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
|
|
120
165
|
- `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
|
|
121
166
|
legacy refs, longest-prefix rule, Windows-1252 numeric mappings, invalid codepoints
|
|
122
|
-
- `test/smoke.test.ts`: live network checks against real hosts
|
|
167
|
+
- `test/smoke.test.ts`: live network checks against real hosts, including a per-engine
|
|
168
|
+
result-health sweep (at least three engines must return well-formed results)
|
|
123
169
|
|
|
124
170
|
The seams (`seams.resolve` / `seams.request` / `rawFetch`) replace the network stack
|
|
125
171
|
with fakes, mirroring how the Studio suite monkeypatches `_validate_and_resolve_host`
|
package/engines.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { randomBytes } from "node:crypto";
|
|
2
|
-
import { decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
|
|
2
|
+
import { collapseWhitespace, decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
|
|
3
3
|
import type { AttrDict } from "./html-to-md.ts";
|
|
4
4
|
import { randomUserAgent } from "./user-agents.ts";
|
|
5
5
|
export class EmptySweepError extends Error {
|
|
@@ -35,7 +35,7 @@ export function normalizeText(raw: string): string {
|
|
|
35
35
|
text = decodeHtmlEntities(text);
|
|
36
36
|
text = text.normalize("NFC");
|
|
37
37
|
text = text.replace(/[\p{Cc}\p{Cf}\p{Co}\p{Cs}\p{Cn}]/gu, "");
|
|
38
|
-
return text
|
|
38
|
+
return collapseWhitespace(text);
|
|
39
39
|
}
|
|
40
40
|
|
|
41
41
|
export function normalizeUrl(url: string): string {
|
|
@@ -47,6 +47,41 @@ export function normalizeUrl(url: string): string {
|
|
|
47
47
|
}
|
|
48
48
|
}
|
|
49
49
|
|
|
50
|
+
const TRACKING_PARAM_NAMES = new Set([
|
|
51
|
+
"_hsenc",
|
|
52
|
+
"_hsmi",
|
|
53
|
+
"dclid",
|
|
54
|
+
"fbclid",
|
|
55
|
+
"gbraid",
|
|
56
|
+
"gclid",
|
|
57
|
+
"gclsrc",
|
|
58
|
+
"igshid",
|
|
59
|
+
"mc_cid",
|
|
60
|
+
"mc_eid",
|
|
61
|
+
"msclkid",
|
|
62
|
+
"srsltid",
|
|
63
|
+
"twclid",
|
|
64
|
+
"wbraid",
|
|
65
|
+
"yclid",
|
|
66
|
+
]);
|
|
67
|
+
|
|
68
|
+
export function canonicalizeHref(href: string): string {
|
|
69
|
+
if (!href) return "";
|
|
70
|
+
try {
|
|
71
|
+
const url = new URL(href);
|
|
72
|
+
for (const key of [...url.searchParams.keys()]) {
|
|
73
|
+
const lower = key.toLowerCase();
|
|
74
|
+
if (lower.startsWith("utm_") || TRACKING_PARAM_NAMES.has(lower)) {
|
|
75
|
+
url.searchParams.delete(key);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
url.hash = "";
|
|
79
|
+
return url.toString();
|
|
80
|
+
} catch {
|
|
81
|
+
return href;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
50
85
|
export interface DomNode {
|
|
51
86
|
tag: string;
|
|
52
87
|
attrs: Record<string, string>;
|
|
@@ -382,7 +417,7 @@ export function extractResults(
|
|
|
382
417
|
["body", elementsXpath.body],
|
|
383
418
|
] as const;
|
|
384
419
|
for (const [key, value] of entries) {
|
|
385
|
-
const data = xpathText(value, item).join("")
|
|
420
|
+
const data = collapseWhitespace(xpathText(value, item).join(""));
|
|
386
421
|
if (!data) continue;
|
|
387
422
|
result[key] = key === "href" ? normalizeUrl(data) : normalizeText(data);
|
|
388
423
|
}
|
|
@@ -444,15 +479,22 @@ export interface Engine {
|
|
|
444
479
|
): Promise<SearchResult[] | null>;
|
|
445
480
|
}
|
|
446
481
|
|
|
482
|
+
interface HttpRequestOptions {
|
|
483
|
+
headers?: Record<string, string>;
|
|
484
|
+
cookies?: Record<string, string>;
|
|
485
|
+
timeoutMs: number;
|
|
486
|
+
signal?: AbortSignal;
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
interface HttpOptions extends HttpRequestOptions {
|
|
490
|
+
method?: string;
|
|
491
|
+
body?: string;
|
|
492
|
+
}
|
|
493
|
+
|
|
447
494
|
async function httpGet(
|
|
448
495
|
url: string,
|
|
449
496
|
params: Record<string, string>,
|
|
450
|
-
options:
|
|
451
|
-
headers?: Record<string, string>;
|
|
452
|
-
cookies?: Record<string, string>;
|
|
453
|
-
timeoutMs: number;
|
|
454
|
-
signal?: AbortSignal;
|
|
455
|
-
},
|
|
497
|
+
options: HttpRequestOptions,
|
|
456
498
|
): Promise<string | null> {
|
|
457
499
|
const target = new URL(url);
|
|
458
500
|
for (const [key, value] of Object.entries(params)) target.searchParams.set(key, value);
|
|
@@ -462,17 +504,15 @@ async function httpGet(
|
|
|
462
504
|
async function httpPost(
|
|
463
505
|
url: string,
|
|
464
506
|
data: Record<string, string>,
|
|
465
|
-
options:
|
|
466
|
-
headers?: Record<string, string>;
|
|
467
|
-
cookies?: Record<string, string>;
|
|
468
|
-
timeoutMs: number;
|
|
469
|
-
signal?: AbortSignal;
|
|
470
|
-
},
|
|
507
|
+
options: HttpRequestOptions,
|
|
471
508
|
): Promise<string | null> {
|
|
472
509
|
return httpFetch(url, { ...options, method: "POST", body: new URLSearchParams(data).toString() });
|
|
473
510
|
}
|
|
474
511
|
|
|
475
512
|
const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
|
|
513
|
+
const ENGINE_RETRY_BACKOFF_MS = 250;
|
|
514
|
+
|
|
515
|
+
const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
|
|
476
516
|
|
|
477
517
|
async function readBodyCapped(response: Response): Promise<string | null> {
|
|
478
518
|
const declared = Number(response.headers.get("content-length") ?? "0");
|
|
@@ -494,16 +534,15 @@ async function readBodyCapped(response: Response): Promise<string | null> {
|
|
|
494
534
|
return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
|
|
495
535
|
}
|
|
496
536
|
|
|
537
|
+
function mapFetchError(err: unknown): never {
|
|
538
|
+
if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
|
|
539
|
+
if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
|
|
540
|
+
throw err;
|
|
541
|
+
}
|
|
542
|
+
|
|
497
543
|
async function httpFetch(
|
|
498
544
|
url: string,
|
|
499
|
-
options:
|
|
500
|
-
method?: string;
|
|
501
|
-
body?: string;
|
|
502
|
-
headers?: Record<string, string>;
|
|
503
|
-
cookies?: Record<string, string>;
|
|
504
|
-
timeoutMs: number;
|
|
505
|
-
signal?: AbortSignal;
|
|
506
|
-
},
|
|
545
|
+
options: HttpOptions,
|
|
507
546
|
): Promise<string | null> {
|
|
508
547
|
const headers: Record<string, string> = {
|
|
509
548
|
"User-Agent": options.headers?.["User-Agent"] ?? randomUserAgent(),
|
|
@@ -528,17 +567,20 @@ async function httpFetch(
|
|
|
528
567
|
signal: AbortSignal.any(signals),
|
|
529
568
|
});
|
|
530
569
|
} catch (err) {
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
570
|
+
throw mapFetchError(err);
|
|
571
|
+
}
|
|
572
|
+
if (response.status !== 200) {
|
|
573
|
+
try {
|
|
574
|
+
await response.body?.cancel();
|
|
575
|
+
} catch {
|
|
576
|
+
return null;
|
|
577
|
+
}
|
|
578
|
+
return null;
|
|
534
579
|
}
|
|
535
|
-
if (response.status !== 200) return null;
|
|
536
580
|
try {
|
|
537
581
|
return await readBodyCapped(response);
|
|
538
582
|
} catch (err) {
|
|
539
|
-
|
|
540
|
-
if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
|
|
541
|
-
throw err;
|
|
583
|
+
throw mapFetchError(err);
|
|
542
584
|
}
|
|
543
585
|
}
|
|
544
586
|
|
|
@@ -649,7 +691,7 @@ const MOJEEK: Engine = {
|
|
|
649
691
|
const YAHOO: Engine = {
|
|
650
692
|
name: "yahoo",
|
|
651
693
|
provider: "bing",
|
|
652
|
-
async search(query,
|
|
694
|
+
async search(query, _ctx, timeoutMs, signal) {
|
|
653
695
|
const ylt = tokenUrlSafe(18);
|
|
654
696
|
const ylu = tokenUrlSafe(35);
|
|
655
697
|
const html = await httpGet(
|
|
@@ -675,7 +717,7 @@ const YAHOO: Engine = {
|
|
|
675
717
|
const YANDEX: Engine = {
|
|
676
718
|
name: "yandex",
|
|
677
719
|
provider: "yandex",
|
|
678
|
-
async search(query,
|
|
720
|
+
async search(query, _ctx, timeoutMs, signal) {
|
|
679
721
|
const searchid = 1000000 + Math.floor(Math.random() * 9000000);
|
|
680
722
|
const html = await httpGet(
|
|
681
723
|
"https://yandex.com/search/site/",
|
|
@@ -734,7 +776,7 @@ const WIKIPEDIA: Engine = {
|
|
|
734
776
|
},
|
|
735
777
|
};
|
|
736
778
|
|
|
737
|
-
const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
|
|
779
|
+
export const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
|
|
738
780
|
|
|
739
781
|
export class ResultsAggregator {
|
|
740
782
|
private cache = new Map<string, SearchResult>();
|
|
@@ -745,10 +787,10 @@ export class ResultsAggregator {
|
|
|
745
787
|
}
|
|
746
788
|
|
|
747
789
|
append(item: SearchResult): void {
|
|
748
|
-
const key = item.href;
|
|
790
|
+
const key = canonicalizeHref(item.href);
|
|
749
791
|
const existing = this.cache.get(key);
|
|
750
792
|
if (!existing || item.body.length > existing.body.length) {
|
|
751
|
-
this.cache.set(key, item);
|
|
793
|
+
this.cache.set(key, { ...item, href: key });
|
|
752
794
|
}
|
|
753
795
|
this.counter.set(key, (this.counter.get(key) ?? 0) + 1);
|
|
754
796
|
}
|
|
@@ -821,6 +863,7 @@ export async function autoTextSearch(
|
|
|
821
863
|
const seenProviders = new Set<string>();
|
|
822
864
|
const aggregator = new ResultsAggregator();
|
|
823
865
|
const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
|
|
866
|
+
const controller = new AbortController();
|
|
824
867
|
let timedOut = false;
|
|
825
868
|
let cancelled = false;
|
|
826
869
|
const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
|
|
@@ -828,20 +871,50 @@ export async function autoTextSearch(
|
|
|
828
871
|
let i = 0;
|
|
829
872
|
let pending: Promise<void>[] = [];
|
|
830
873
|
const run = async (engine: Engine) => {
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
874
|
+
let results: SearchResult[] | null = null;
|
|
875
|
+
for (let attempt = 0; attempt < 2 && results === null; attempt++) {
|
|
876
|
+
if (controller.signal.aborted) return;
|
|
877
|
+
const budgetLeft = deadline - Date.now();
|
|
878
|
+
if (budgetLeft <= 0) return;
|
|
879
|
+
if (attempt > 0 && budgetLeft < ENGINE_RETRY_BACKOFF_MS) return;
|
|
880
|
+
const remaining = Math.max(1, budgetLeft);
|
|
881
|
+
const engineSignal = signal
|
|
882
|
+
? AbortSignal.any([signal, controller.signal])
|
|
883
|
+
: controller.signal;
|
|
884
|
+
try {
|
|
885
|
+
results = await engine.search(query, ctx, remaining, engineSignal);
|
|
886
|
+
} catch (e) {
|
|
887
|
+
if (e instanceof SearchCancelled) {
|
|
888
|
+
if (controller.signal.aborted && !signal?.aborted) return;
|
|
889
|
+
cancelled = true;
|
|
890
|
+
return;
|
|
891
|
+
}
|
|
892
|
+
if (e instanceof SearchTimeoutError) {
|
|
893
|
+
timedOut = true;
|
|
894
|
+
return;
|
|
895
|
+
}
|
|
896
|
+
}
|
|
897
|
+
if (results === null && attempt === 0) {
|
|
898
|
+
const backoff = Math.min(ENGINE_RETRY_BACKOFF_MS, Math.max(0, deadline - Date.now()));
|
|
899
|
+
if (backoff > 0) await sleep(backoff);
|
|
900
|
+
if (signal?.aborted) {
|
|
901
|
+
cancelled = true;
|
|
902
|
+
return;
|
|
903
|
+
}
|
|
904
|
+
if (controller.signal.aborted) return;
|
|
837
905
|
}
|
|
838
|
-
}
|
|
839
|
-
|
|
840
|
-
|
|
906
|
+
}
|
|
907
|
+
if (results && results.length) {
|
|
908
|
+
aggregator.extend(results);
|
|
909
|
+
seenProviders.add(engine.provider);
|
|
910
|
+
if (aggregator.size >= maxResults) controller.abort();
|
|
841
911
|
}
|
|
842
912
|
};
|
|
843
913
|
while (i < engines.length) {
|
|
844
|
-
if (aggregator.size >= maxResults)
|
|
914
|
+
if (aggregator.size >= maxResults || cancelled) {
|
|
915
|
+
controller.abort();
|
|
916
|
+
break;
|
|
917
|
+
}
|
|
845
918
|
const engine = engines[i++];
|
|
846
919
|
if (seenProviders.has(engine.provider)) continue;
|
|
847
920
|
pending.push(run(engine));
|
package/html-to-md.ts
CHANGED
|
@@ -194,7 +194,7 @@ class HeaderFrame {
|
|
|
194
194
|
const CHARREF_RE = /&(#[0-9]+;?|#[xX][0-9a-fA-F]+;?|[^\t\n\f <&#;]{1,32};?)/g;
|
|
195
195
|
|
|
196
196
|
export function decodeHtmlEntities(text: string): string {
|
|
197
|
-
return text.replace(CHARREF_RE, (
|
|
197
|
+
return text.replace(CHARREF_RE, (_whole, s: string) => {
|
|
198
198
|
if (s[0] === "#") {
|
|
199
199
|
const hex = s[1] === "x" || s[1] === "X";
|
|
200
200
|
const num = parseInt(s.slice(hex ? 2 : 1).replace(/;+$/, ""), hex ? 16 : 10);
|
|
@@ -214,6 +214,10 @@ export function decodeHtmlEntities(text: string): string {
|
|
|
214
214
|
});
|
|
215
215
|
}
|
|
216
216
|
|
|
217
|
+
export function collapseWhitespace(text: string): string {
|
|
218
|
+
return text.replace(/\s+/g, " ").trim();
|
|
219
|
+
}
|
|
220
|
+
|
|
217
221
|
|
|
218
222
|
interface HtmlHandlers {
|
|
219
223
|
handleStartTag(name: string, attrs: AttrDict): void;
|
|
@@ -224,8 +228,11 @@ interface HtmlHandlers {
|
|
|
224
228
|
handleCharRef(name: string): void;
|
|
225
229
|
}
|
|
226
230
|
|
|
227
|
-
const START_TAG_NAME_RE =
|
|
228
|
-
const ATTR_NAME_RE =
|
|
231
|
+
const START_TAG_NAME_RE = /[a-zA-Z][^\s/>]*/y;
|
|
232
|
+
const ATTR_NAME_RE = /[^\s=/>]+/y;
|
|
233
|
+
const ENTITY_NAMED_RE = /&([A-Za-z][A-Za-z0-9.-]*);/y;
|
|
234
|
+
const ENTITY_NUMERIC_RE = /&#([xX][0-9a-fA-F]+|[0-9]+);?/y;
|
|
235
|
+
const ENTITY_LEGACY_RE = /&([A-Za-z][A-Za-z0-9.-]*)(?=[^A-Za-z0-9]|$)/y;
|
|
229
236
|
|
|
230
237
|
const RAW_TEXT_TAGS = [
|
|
231
238
|
"script",
|
|
@@ -238,9 +245,23 @@ const RAW_TEXT_TAGS = [
|
|
|
238
245
|
"noframes",
|
|
239
246
|
];
|
|
240
247
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
248
|
+
function findRawTextClose(
|
|
249
|
+
input: string,
|
|
250
|
+
lowerInput: string,
|
|
251
|
+
from: number,
|
|
252
|
+
name: string,
|
|
253
|
+
): { start: number; end: number } | null {
|
|
254
|
+
const needle = `</${name}`;
|
|
255
|
+
let pos = from;
|
|
256
|
+
while (true) {
|
|
257
|
+
const hit = lowerInput.indexOf(needle, pos);
|
|
258
|
+
if (hit === -1) return null;
|
|
259
|
+
let j = hit + needle.length;
|
|
260
|
+
while (j < input.length && /\s/.test(input[j])) j++;
|
|
261
|
+
if (j < input.length && input[j] === ">") return { start: hit, end: j + 1 };
|
|
262
|
+
pos = hit + 1;
|
|
263
|
+
}
|
|
264
|
+
}
|
|
244
265
|
|
|
245
266
|
function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, boolean] {
|
|
246
267
|
const attrs: AttrDict = {};
|
|
@@ -253,10 +274,11 @@ function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, bo
|
|
|
253
274
|
pos++;
|
|
254
275
|
continue;
|
|
255
276
|
}
|
|
256
|
-
|
|
277
|
+
ATTR_NAME_RE.lastIndex = pos;
|
|
278
|
+
const nameMatch = ATTR_NAME_RE.exec(input);
|
|
257
279
|
if (!nameMatch) return [attrs, -1, false];
|
|
258
280
|
const name = nameMatch[0].toLowerCase();
|
|
259
|
-
pos
|
|
281
|
+
pos = nameMatch.index + nameMatch[0].length;
|
|
260
282
|
while (pos < input.length && /\s/.test(input[pos])) pos++;
|
|
261
283
|
let value: string | null = null;
|
|
262
284
|
if (pos < input.length && input[pos] === "=") {
|
|
@@ -281,64 +303,69 @@ function parseAttrsUntilClose(input: string, pos: number): [AttrDict, number, bo
|
|
|
281
303
|
}
|
|
282
304
|
|
|
283
305
|
function scanTag(
|
|
284
|
-
|
|
306
|
+
input: string,
|
|
285
307
|
i: number,
|
|
286
308
|
): { end: number; kind: "comment" | "decl" | "end" | "start" | "startend"; name?: string; attrs?: AttrDict } | null {
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
const close = html.indexOf("-->", i + 4);
|
|
309
|
+
if (input.startsWith("!--", i + 1)) {
|
|
310
|
+
const close = input.indexOf("-->", i + 4);
|
|
290
311
|
if (close === -1) return null;
|
|
291
312
|
return { end: close + 3, kind: "decl" };
|
|
292
313
|
}
|
|
293
|
-
|
|
314
|
+
const next = input[i + 1];
|
|
315
|
+
if (next === "!" || next === "?") {
|
|
294
316
|
let j = i + 2;
|
|
295
|
-
while (j <
|
|
296
|
-
if (j >=
|
|
317
|
+
while (j < input.length && input[j] !== ">") j++;
|
|
318
|
+
if (j >= input.length) return null;
|
|
297
319
|
return { end: j + 1, kind: "decl" };
|
|
298
320
|
}
|
|
299
|
-
if (
|
|
321
|
+
if (next === "/") {
|
|
300
322
|
let j = i + 2;
|
|
301
|
-
while (j <
|
|
302
|
-
const name =
|
|
323
|
+
while (j < input.length && !/[\s>]/.test(input[j])) j++;
|
|
324
|
+
const name = input.slice(i + 2, j).toLowerCase();
|
|
303
325
|
if (!name) return null;
|
|
304
|
-
while (j <
|
|
305
|
-
if (j >=
|
|
326
|
+
while (j < input.length && input[j] !== ">") j++;
|
|
327
|
+
if (j >= input.length) return null;
|
|
306
328
|
return { end: j + 1, kind: "end", name };
|
|
307
329
|
}
|
|
308
|
-
|
|
330
|
+
START_TAG_NAME_RE.lastIndex = i + 1;
|
|
331
|
+
const nameMatch = START_TAG_NAME_RE.exec(input);
|
|
309
332
|
if (!nameMatch) return null;
|
|
310
333
|
const name = nameMatch[0].toLowerCase();
|
|
311
|
-
const [attrs,
|
|
312
|
-
if (
|
|
313
|
-
return { end:
|
|
334
|
+
const [attrs, nextPos, selfClosing] = parseAttrsUntilClose(input, i + 1 + nameMatch[0].length);
|
|
335
|
+
if (nextPos === -1) return null;
|
|
336
|
+
return { end: nextPos, kind: selfClosing ? "startend" : "start", name, attrs };
|
|
314
337
|
}
|
|
315
338
|
|
|
316
339
|
|
|
317
340
|
export function feedHtml(input: string, handlers: HtmlHandlers): void {
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
341
|
+
let lowerInput: string | null = null;
|
|
342
|
+
const emitTextRange = (start: number, end: number) => {
|
|
343
|
+
if (end <= start) return;
|
|
344
|
+
let pos = start;
|
|
345
|
+
while (pos < end) {
|
|
346
|
+
const amp = input.indexOf("&", pos);
|
|
347
|
+
if (amp === -1 || amp >= end) {
|
|
348
|
+
handlers.handleData(input.slice(pos, end));
|
|
325
349
|
return;
|
|
326
350
|
}
|
|
327
|
-
if (amp > pos) handlers.handleData(
|
|
328
|
-
|
|
329
|
-
|
|
351
|
+
if (amp > pos) handlers.handleData(input.slice(pos, amp));
|
|
352
|
+
ENTITY_NAMED_RE.lastIndex = amp;
|
|
353
|
+
const named = ENTITY_NAMED_RE.exec(input);
|
|
354
|
+
if (named && named.index === amp && amp + named[0].length <= end) {
|
|
330
355
|
handlers.handleEntityRef(named[1]);
|
|
331
356
|
pos = amp + named[0].length;
|
|
332
357
|
continue;
|
|
333
358
|
}
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
359
|
+
ENTITY_NUMERIC_RE.lastIndex = amp;
|
|
360
|
+
const numeric = ENTITY_NUMERIC_RE.exec(input);
|
|
361
|
+
if (numeric && numeric.index === amp && amp + numeric[0].length <= end) {
|
|
362
|
+
handlers.handleCharRef(numeric[1]);
|
|
337
363
|
pos = amp + numeric[0].length;
|
|
338
364
|
continue;
|
|
339
365
|
}
|
|
340
|
-
|
|
341
|
-
|
|
366
|
+
ENTITY_LEGACY_RE.lastIndex = amp;
|
|
367
|
+
const legacy = ENTITY_LEGACY_RE.exec(input);
|
|
368
|
+
if (legacy && legacy.index === amp && amp + legacy[0].length <= end) {
|
|
342
369
|
handlers.handleEntityRef(legacy[1]);
|
|
343
370
|
pos = amp + legacy[0].length;
|
|
344
371
|
continue;
|
|
@@ -360,28 +387,32 @@ export function feedHtml(input: string, handlers: HtmlHandlers): void {
|
|
|
360
387
|
i++;
|
|
361
388
|
continue;
|
|
362
389
|
}
|
|
363
|
-
|
|
390
|
+
emitTextRange(textStart, i);
|
|
364
391
|
if (tag.kind === "start") {
|
|
365
|
-
const
|
|
366
|
-
if (
|
|
367
|
-
|
|
368
|
-
const
|
|
369
|
-
if (
|
|
370
|
-
handlers.handleStartTag(
|
|
371
|
-
|
|
372
|
-
handlers.handleEndTag(
|
|
373
|
-
textStart =
|
|
392
|
+
const rawName = tag.name!;
|
|
393
|
+
if (RAW_TEXT_TAGS.includes(rawName)) {
|
|
394
|
+
lowerInput ??= input.toLowerCase();
|
|
395
|
+
const close = findRawTextClose(input, lowerInput, tag.end, rawName);
|
|
396
|
+
if (close) {
|
|
397
|
+
handlers.handleStartTag(rawName, tag.attrs!);
|
|
398
|
+
emitTextRange(tag.end, close.start);
|
|
399
|
+
handlers.handleEndTag(rawName);
|
|
400
|
+
textStart = close.end;
|
|
374
401
|
i = textStart;
|
|
375
402
|
continue;
|
|
376
403
|
}
|
|
377
404
|
}
|
|
378
|
-
handlers.handleStartTag(
|
|
405
|
+
handlers.handleStartTag(rawName, tag.attrs!);
|
|
379
406
|
} else if (tag.kind === "startend") handlers.handleStartEndTag(tag.name!, tag.attrs!);
|
|
380
407
|
else if (tag.kind === "end") handlers.handleEndTag(tag.name!);
|
|
381
408
|
textStart = tag.end;
|
|
382
409
|
i = tag.end;
|
|
383
410
|
}
|
|
384
|
-
|
|
411
|
+
emitTextRange(textStart, input.length);
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
function popMarksAbove(marks: number[], index: number): void {
|
|
415
|
+
while (marks.length && marks[marks.length - 1] >= index) marks.pop();
|
|
385
416
|
}
|
|
386
417
|
|
|
387
418
|
class MarkdownRenderer {
|
|
@@ -524,8 +555,8 @@ class MarkdownRenderer {
|
|
|
524
555
|
}
|
|
525
556
|
|
|
526
557
|
private finishLink(): void {
|
|
527
|
-
const text = this.linkTextParts.join("")
|
|
528
|
-
const headingText = this.linkHeadingParts.join("")
|
|
558
|
+
const text = collapseWhitespace(this.linkTextParts.join(""));
|
|
559
|
+
const headingText = collapseWhitespace(this.linkHeadingParts.join(""));
|
|
529
560
|
const href = this.linkHref ?? "";
|
|
530
561
|
this.inLink = false;
|
|
531
562
|
this.linkTextParts = [];
|
|
@@ -571,12 +602,8 @@ class MarkdownRenderer {
|
|
|
571
602
|
}
|
|
572
603
|
if (closeAt === null) break;
|
|
573
604
|
this.truncateOpenTags(closeAt);
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
}
|
|
577
|
-
while (this.headingMarks.length && this.headingMarks[this.headingMarks.length - 1] >= closeAt) {
|
|
578
|
-
this.headingMarks.pop();
|
|
579
|
-
}
|
|
605
|
+
popMarksAbove(this.hiddenMarks, closeAt);
|
|
606
|
+
popMarksAbove(this.headingMarks, closeAt);
|
|
580
607
|
this.closeHeaderFrames(closeAt);
|
|
581
608
|
}
|
|
582
609
|
}
|
|
@@ -692,12 +719,8 @@ class MarkdownRenderer {
|
|
|
692
719
|
for (let i = this.openTags.length - 1; i >= 0; i--) {
|
|
693
720
|
if (this.openTags[i] === tag) {
|
|
694
721
|
this.truncateOpenTags(i);
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
}
|
|
698
|
-
while (this.headingMarks.length && this.headingMarks[this.headingMarks.length - 1] >= i) {
|
|
699
|
-
this.headingMarks.pop();
|
|
700
|
-
}
|
|
722
|
+
popMarksAbove(this.hiddenMarks, i);
|
|
723
|
+
popMarksAbove(this.headingMarks, i);
|
|
701
724
|
this.closeHeaderFrames(i, tag === "header");
|
|
702
725
|
break;
|
|
703
726
|
}
|
|
@@ -715,7 +738,11 @@ class MarkdownRenderer {
|
|
|
715
738
|
return !suppressed;
|
|
716
739
|
}
|
|
717
740
|
|
|
718
|
-
handleStartEndTag(
|
|
741
|
+
handleStartEndTag(name: string, attrs: AttrDict): void {
|
|
742
|
+
if (!VOID_TAGS.has(name)) return;
|
|
743
|
+
this.handleStartTag(name, attrs);
|
|
744
|
+
this.handleEndTag(name);
|
|
745
|
+
}
|
|
719
746
|
handleStartTag(tag: string, attrs: AttrDict): void {
|
|
720
747
|
if (this.skipDepth) {
|
|
721
748
|
if (SKIP_TAGS.has(tag)) this.skipDepth++;
|