pi-unsloth-webtools 0.2.5 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -11
- package/engines.ts +107 -45
- package/html-to-md.ts +22 -18
- package/index.ts +21 -11
- package/package.json +1 -1
- package/pdf.ts +92 -11
- package/web-access.ts +68 -50
- package/web-fetch.ts +382 -93
- package/web-search.ts +5 -4
package/web-fetch.ts
CHANGED
|
@@ -1,22 +1,30 @@
|
|
|
1
1
|
import { lookup as dnsLookup } from "node:dns/promises";
|
|
2
2
|
import http from "node:http";
|
|
3
3
|
import https from "node:https";
|
|
4
|
+
import { createBrotliDecompress, createGunzip, createInflate, createInflateRaw } from "node:zlib";
|
|
4
5
|
import type { IncomingMessage } from "node:http";
|
|
6
|
+
import type { Transform } from "node:stream";
|
|
5
7
|
import {
|
|
6
8
|
checkUrlAccess,
|
|
9
|
+
githubRepoRawReadmeUrl,
|
|
7
10
|
githubRepoReadmeApiUrl,
|
|
8
11
|
isPublicIp,
|
|
9
12
|
normalizeUrlScheme,
|
|
10
13
|
type WebsitePolicy,
|
|
11
14
|
} from "./web-access.ts";
|
|
12
|
-
import { htmlToMarkdown } from "./html-to-md.ts";
|
|
15
|
+
import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown } from "./html-to-md.ts";
|
|
16
|
+
import type { AttrDict } from "./html-to-md.ts";
|
|
13
17
|
import { INVALID_CHARREFS } from "./entities.ts";
|
|
14
|
-
import { extractPdfText
|
|
18
|
+
import { extractPdfText } from "./pdf.ts";
|
|
15
19
|
import { randomUserAgent } from "./user-agents.ts";
|
|
16
20
|
|
|
17
21
|
const MAX_FETCH_BYTES = 512 * 1024;
|
|
22
|
+
const MAX_DECOMPRESSED_BYTES = 64 * 1024 * 1024;
|
|
23
|
+
const META_VALUE_MAX_CHARS = 300;
|
|
18
24
|
const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
|
|
19
25
|
const MAX_REQUESTS = 5;
|
|
26
|
+
const MAX_SIGNAL_TIMEOUT_MS = 2 ** 31 - 1;
|
|
27
|
+
export const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
|
|
20
28
|
|
|
21
29
|
const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
|
|
22
30
|
const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
|
|
@@ -109,6 +117,28 @@ export class FetchTimeoutError extends Error {
|
|
|
109
117
|
super("timed out");
|
|
110
118
|
}
|
|
111
119
|
}
|
|
120
|
+
const FETCH_CANCELLED_MESSAGE = "Failed to fetch URL: cancelled.";
|
|
121
|
+
const FETCH_TIMEOUT_MESSAGE = "Failed to fetch URL: timed out.";
|
|
122
|
+
const TRUNCATED_BODY_NOTICE = "\n\n... (page truncated at the download limit)";
|
|
123
|
+
const TRUNCATED_BODY_SUFFIX = "... (page truncated at the download limit)";
|
|
124
|
+
|
|
125
|
+
function fetchErrorMessage(err: unknown): string {
|
|
126
|
+
if (err instanceof FetchCancelledError) return FETCH_CANCELLED_MESSAGE;
|
|
127
|
+
if (err instanceof FetchTimeoutError) return FETCH_TIMEOUT_MESSAGE;
|
|
128
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
129
|
+
if (message === "cancelled") return FETCH_CANCELLED_MESSAGE;
|
|
130
|
+
if (message === "timed out") return FETCH_TIMEOUT_MESSAGE;
|
|
131
|
+
return `Failed to fetch URL: ${message}`;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function statusErrorResult(status: number): RawFetchResult {
|
|
135
|
+
const reason = http.STATUS_CODES[status] ?? "";
|
|
136
|
+
return {
|
|
137
|
+
error: `Failed to fetch URL: HTTP ${status}${reason ? ` ${reason}` : ""}`,
|
|
138
|
+
body: "",
|
|
139
|
+
contentType: "",
|
|
140
|
+
};
|
|
141
|
+
}
|
|
112
142
|
|
|
113
143
|
export interface FetchPageOptions {
|
|
114
144
|
timeoutMs?: number;
|
|
@@ -127,6 +157,7 @@ export interface HopResponse {
|
|
|
127
157
|
status: number;
|
|
128
158
|
headers: Record<string, string | string[] | undefined>;
|
|
129
159
|
body: Buffer;
|
|
160
|
+
truncated?: boolean;
|
|
130
161
|
}
|
|
131
162
|
|
|
132
163
|
export interface ResolvedHost {
|
|
@@ -172,20 +203,32 @@ export interface RawFetchResult {
|
|
|
172
203
|
contentType: string;
|
|
173
204
|
}
|
|
174
205
|
|
|
206
|
+
function htmlProbe(body: string, re: RegExp): boolean {
|
|
207
|
+
let probe = body;
|
|
208
|
+
while (true) {
|
|
209
|
+
probe = probe.replace(/^[ \t\n\r\f\v]+/, "");
|
|
210
|
+
const stripped = probe.replace(/^(?:<!--[\s\S]*?-->|<\?[\s\S]*?\?>)/, "");
|
|
211
|
+
if (stripped === probe) break;
|
|
212
|
+
probe = stripped;
|
|
213
|
+
}
|
|
214
|
+
return re.test(probe.slice(0, 256).toLowerCase());
|
|
215
|
+
}
|
|
216
|
+
|
|
175
217
|
export function looksLikeHtml(body: string): boolean {
|
|
176
|
-
|
|
177
|
-
return HTML_LEADING_RE.test(probe);
|
|
218
|
+
return htmlProbe(body, HTML_LEADING_RE);
|
|
178
219
|
}
|
|
179
220
|
|
|
180
221
|
export function looksLikeHtmlDocument(body: string): boolean {
|
|
181
|
-
|
|
182
|
-
|
|
222
|
+
return htmlProbe(body, HTML_DOCUMENT_RE);
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
function parseContentType(value: string | null | undefined): string {
|
|
226
|
+
return /^[\w.+-]+\/[\w.+-]+/.exec(value ?? "")?.[0] ?? "";
|
|
183
227
|
}
|
|
184
228
|
|
|
185
229
|
export function isTextCandidateContentType(contentType: string | null): boolean {
|
|
186
|
-
const
|
|
187
|
-
if (!
|
|
188
|
-
const ct = match[0].toLowerCase();
|
|
230
|
+
const ct = parseContentType(contentType).toLowerCase();
|
|
231
|
+
if (!ct) return true;
|
|
189
232
|
if (ct.startsWith("text/")) return true;
|
|
190
233
|
if (ct.startsWith("application/")) {
|
|
191
234
|
const subtype = ct.slice("application/".length);
|
|
@@ -321,7 +364,12 @@ function sniffMetaCharsetForHtml(bytes: Buffer, contentType: string): string | n
|
|
|
321
364
|
|
|
322
365
|
function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
|
|
323
366
|
let out = "";
|
|
324
|
-
|
|
367
|
+
let i = 0;
|
|
368
|
+
if (bytes.length >= 4) {
|
|
369
|
+
const first = littleEndian ? bytes.readUInt32LE(0) : bytes.readUInt32BE(0);
|
|
370
|
+
if (first === 0xfeff) i = 4;
|
|
371
|
+
}
|
|
372
|
+
for (; i + 3 < bytes.length; i += 4) {
|
|
325
373
|
const v = littleEndian ? bytes.readUInt32LE(i) : bytes.readUInt32BE(i);
|
|
326
374
|
if (v === 0 || v > 0x10ffff || (v >= 0xd800 && v <= 0xdfff)) {
|
|
327
375
|
out += "\ufffd";
|
|
@@ -334,7 +382,8 @@ function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
|
|
|
334
382
|
|
|
335
383
|
function decodeUtf16Be(bytes: Buffer): string {
|
|
336
384
|
let out = "";
|
|
337
|
-
|
|
385
|
+
let i = bytes.length >= 2 && bytes.readUInt16BE(0) === 0xfeff ? 2 : 0;
|
|
386
|
+
for (; i + 1 < bytes.length; i += 2) {
|
|
338
387
|
out += String.fromCharCode(bytes.readUInt16BE(i));
|
|
339
388
|
}
|
|
340
389
|
return out;
|
|
@@ -406,10 +455,32 @@ function decodeTis620(bytes: Buffer): string {
|
|
|
406
455
|
}
|
|
407
456
|
|
|
408
457
|
|
|
458
|
+
function withAbort<T>(promise: Promise<T>, signal?: AbortSignal): Promise<T> {
|
|
459
|
+
if (!signal) return promise;
|
|
460
|
+
return new Promise<T>((resolve, reject) => {
|
|
461
|
+
const onAbort = () => reject(new DOMException("aborted", "AbortError"));
|
|
462
|
+
if (signal.aborted) {
|
|
463
|
+
onAbort();
|
|
464
|
+
return;
|
|
465
|
+
}
|
|
466
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
467
|
+
promise.then(
|
|
468
|
+
(value) => {
|
|
469
|
+
signal.removeEventListener("abort", onAbort);
|
|
470
|
+
resolve(value);
|
|
471
|
+
},
|
|
472
|
+
(err) => {
|
|
473
|
+
signal.removeEventListener("abort", onAbort);
|
|
474
|
+
reject(err);
|
|
475
|
+
},
|
|
476
|
+
);
|
|
477
|
+
});
|
|
478
|
+
}
|
|
479
|
+
|
|
409
480
|
async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promise<ResolvedHost> {
|
|
410
481
|
let addresses: { address: string; family: number }[];
|
|
411
482
|
try {
|
|
412
|
-
addresses = await dnsLookup(hostname, { all: true, verbatim: true });
|
|
483
|
+
addresses = await withAbort(dnsLookup(hostname, { all: true, verbatim: true }), signal);
|
|
413
484
|
} catch (err) {
|
|
414
485
|
return { ok: false, reason: `Failed to resolve host: ${err}`, ip: "", family: 0 };
|
|
415
486
|
}
|
|
@@ -426,17 +497,32 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
|
|
|
426
497
|
}
|
|
427
498
|
|
|
428
499
|
|
|
429
|
-
function
|
|
500
|
+
function budgetExceededResult(
|
|
430
501
|
deadline: number | null,
|
|
431
502
|
signal: AbortSignal | undefined,
|
|
432
503
|
now: () => number = Date.now,
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
if (
|
|
504
|
+
contentType = "",
|
|
505
|
+
): RawFetchResult | null {
|
|
506
|
+
if (signal?.aborted) return { error: FETCH_CANCELLED_MESSAGE, body: "", contentType };
|
|
507
|
+
if (deadline !== null && now() >= deadline) return { error: FETCH_TIMEOUT_MESSAGE, body: "", contentType };
|
|
436
508
|
return null;
|
|
437
509
|
}
|
|
438
510
|
|
|
439
511
|
|
|
512
|
+
function contentEncodingCodec(value: string | string[] | undefined): string | null {
|
|
513
|
+
const declared = Array.isArray(value) ? value[0] : value;
|
|
514
|
+
const codec = (declared ?? "").split(",", 1)[0].trim().toLowerCase();
|
|
515
|
+
if (codec === "gzip" || codec === "x-gzip" || codec === "deflate" || codec === "br") return codec;
|
|
516
|
+
return null;
|
|
517
|
+
}
|
|
518
|
+
|
|
519
|
+
function createDecodeStream(codec: string): Transform {
|
|
520
|
+
const options = { maxOutputLength: MAX_DECOMPRESSED_BYTES };
|
|
521
|
+
if (codec === "br") return createBrotliDecompress(options);
|
|
522
|
+
if (codec === "deflate") return createInflate(options);
|
|
523
|
+
return createGunzip(options);
|
|
524
|
+
}
|
|
525
|
+
|
|
440
526
|
export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
441
527
|
return new Promise((resolve, reject) => {
|
|
442
528
|
const url = opts.url;
|
|
@@ -460,12 +546,18 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
460
546
|
action();
|
|
461
547
|
};
|
|
462
548
|
const request = transport.request(options, (res: IncomingMessage) => {
|
|
549
|
+
const codec = contentEncodingCodec(res.headers["content-encoding"]);
|
|
463
550
|
const chunks: Buffer[] = [];
|
|
464
551
|
let total = 0;
|
|
465
552
|
let head = Buffer.alloc(0);
|
|
553
|
+
let truncated = false;
|
|
554
|
+
let decoder: Transform | null = null;
|
|
466
555
|
const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
|
|
467
556
|
let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
|
|
468
557
|
let extendedForPdf = false;
|
|
558
|
+
const declaredLengthHeader = res.headers["content-length"];
|
|
559
|
+
const declaredLength =
|
|
560
|
+
declaredLengthHeader === undefined ? NaN : Number(declaredLengthHeader);
|
|
469
561
|
const finish = (err: string | null, body: Buffer) => {
|
|
470
562
|
settle(() => {
|
|
471
563
|
if (err) {
|
|
@@ -477,44 +569,112 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
477
569
|
status: res.statusCode ?? 0,
|
|
478
570
|
headers: res.headers as Record<string, string | string[] | undefined>,
|
|
479
571
|
body,
|
|
572
|
+
truncated,
|
|
480
573
|
});
|
|
481
574
|
}
|
|
482
575
|
});
|
|
483
576
|
};
|
|
484
|
-
|
|
577
|
+
const timeoutAndDestroy = () => {
|
|
578
|
+
decoder?.destroy();
|
|
579
|
+
res.destroy();
|
|
580
|
+
finish("timed out", Buffer.concat(chunks));
|
|
581
|
+
};
|
|
582
|
+
const acceptData = (chunk: Buffer) => {
|
|
485
583
|
if (settled) return;
|
|
486
584
|
const now = opts.nowMs ?? Date.now;
|
|
487
585
|
if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
|
|
488
|
-
|
|
489
|
-
finish("timed out", Buffer.concat(chunks));
|
|
586
|
+
timeoutAndDestroy();
|
|
490
587
|
return;
|
|
491
588
|
}
|
|
492
589
|
if (head.length < 1024) {
|
|
493
590
|
const need = 1024 - head.length;
|
|
494
591
|
head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
|
|
495
592
|
}
|
|
496
|
-
if (!declaredPdf && !extendedForPdf && total + chunk.length >
|
|
593
|
+
if (!declaredPdf && !extendedForPdf && total + chunk.length > limit) {
|
|
497
594
|
if (hasPdfMagic(head)) {
|
|
498
595
|
limit = opts.maxPdfBytes;
|
|
499
596
|
extendedForPdf = true;
|
|
500
597
|
}
|
|
501
598
|
}
|
|
502
599
|
const space = limit - total;
|
|
503
|
-
if (space <= 0) {
|
|
504
|
-
res.destroy();
|
|
505
|
-
finish(null, Buffer.concat(chunks));
|
|
506
|
-
return;
|
|
507
|
-
}
|
|
508
600
|
const take = chunk.subarray(0, Math.min(chunk.length, space));
|
|
509
601
|
chunks.push(take);
|
|
510
602
|
total += take.length;
|
|
511
603
|
if (total >= limit) {
|
|
604
|
+
truncated =
|
|
605
|
+
codec !== null ||
|
|
606
|
+
!(Number.isFinite(declaredLength) && declaredLength === total);
|
|
607
|
+
decoder?.destroy();
|
|
512
608
|
res.destroy();
|
|
513
609
|
finish(null, Buffer.concat(chunks));
|
|
514
610
|
}
|
|
611
|
+
};
|
|
612
|
+
if (codec === null) {
|
|
613
|
+
res.on("data", (chunk: Buffer) => {
|
|
614
|
+
acceptData(chunk);
|
|
615
|
+
});
|
|
616
|
+
res.on("end", () => finish(null, Buffer.concat(chunks)));
|
|
617
|
+
} else {
|
|
618
|
+
let rawBuffer: Buffer[] = [];
|
|
619
|
+
let rawBufferBytes = 0;
|
|
620
|
+
let outputStarted = false;
|
|
621
|
+
let rawFallbackTried = false;
|
|
622
|
+
let resEnded = false;
|
|
623
|
+
const wire = (d: Transform) => {
|
|
624
|
+
d.on("data", (chunk: Buffer) => {
|
|
625
|
+
outputStarted = true;
|
|
626
|
+
acceptData(chunk);
|
|
627
|
+
});
|
|
628
|
+
d.on("end", () => finish(null, Buffer.concat(chunks)));
|
|
629
|
+
d.on("drain", () => res.resume());
|
|
630
|
+
d.on("error", (err: NodeJS.ErrnoException) => {
|
|
631
|
+
if (settled) return;
|
|
632
|
+
if (
|
|
633
|
+
codec === "deflate" &&
|
|
634
|
+
!rawFallbackTried &&
|
|
635
|
+
!outputStarted &&
|
|
636
|
+
err.code === "Z_DATA_ERROR"
|
|
637
|
+
) {
|
|
638
|
+
rawFallbackTried = true;
|
|
639
|
+
d.destroy();
|
|
640
|
+
decoder = createInflateRaw({ maxOutputLength: MAX_DECOMPRESSED_BYTES });
|
|
641
|
+
wire(decoder);
|
|
642
|
+
for (const buffered of rawBuffer) decoder.write(buffered);
|
|
643
|
+
if (resEnded) decoder.end();
|
|
644
|
+
return;
|
|
645
|
+
}
|
|
646
|
+
if (outputStarted) {
|
|
647
|
+
truncated = true;
|
|
648
|
+
finish(null, Buffer.concat(chunks));
|
|
649
|
+
return;
|
|
650
|
+
}
|
|
651
|
+
finish(null, Buffer.concat(rawBuffer));
|
|
652
|
+
});
|
|
653
|
+
};
|
|
654
|
+
decoder = createDecodeStream(codec);
|
|
655
|
+
wire(decoder);
|
|
656
|
+
res.on("data", (chunk: Buffer) => {
|
|
657
|
+
if (settled) return;
|
|
658
|
+
const now = opts.nowMs ?? Date.now;
|
|
659
|
+
if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
|
|
660
|
+
timeoutAndDestroy();
|
|
661
|
+
return;
|
|
662
|
+
}
|
|
663
|
+
if (!outputStarted && !rawFallbackTried && rawBufferBytes < limit) {
|
|
664
|
+
rawBuffer.push(chunk);
|
|
665
|
+
rawBufferBytes += chunk.length;
|
|
666
|
+
}
|
|
667
|
+
if (!decoder!.write(chunk)) res.pause();
|
|
668
|
+
});
|
|
669
|
+
res.on("end", () => {
|
|
670
|
+
resEnded = true;
|
|
671
|
+
if (!settled) decoder!.end();
|
|
672
|
+
});
|
|
673
|
+
}
|
|
674
|
+
res.on("error", (err) => {
|
|
675
|
+
decoder?.destroy();
|
|
676
|
+
finish(err.message, Buffer.concat(chunks));
|
|
515
677
|
});
|
|
516
|
-
res.on("end", () => finish(null, Buffer.concat(chunks)));
|
|
517
|
-
res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
|
|
518
678
|
});
|
|
519
679
|
const onAbort = () => request.destroy(new FetchCancelledError());
|
|
520
680
|
opts.signal?.addEventListener("abort", onAbort, { once: true });
|
|
@@ -540,14 +700,23 @@ export async function fetchUrlRaw(
|
|
|
540
700
|
const seams = options.seams ?? {};
|
|
541
701
|
const resolveHost = seams.resolve ?? resolveAndValidate;
|
|
542
702
|
const performRequest = seams.request ?? requestHop;
|
|
703
|
+
const resolveWithBudget = async (hostname: string): Promise<ResolvedHost> => {
|
|
704
|
+
const deadlineSignal = AbortSignal.timeout(Math.min(MAX_SIGNAL_TIMEOUT_MS, Math.max(1, deadline - now())));
|
|
705
|
+
const resolveSignal = signal ? AbortSignal.any([signal, deadlineSignal]) : deadlineSignal;
|
|
706
|
+
const resolved = await resolveHost(hostname, resolveSignal);
|
|
707
|
+
if (resolved.ok) return resolved;
|
|
708
|
+
if (signal?.aborted) return { ...resolved, reason: FETCH_CANCELLED_MESSAGE };
|
|
709
|
+
if (resolveSignal.aborted) return { ...resolved, reason: FETCH_TIMEOUT_MESSAGE };
|
|
710
|
+
return resolved;
|
|
711
|
+
};
|
|
543
712
|
|
|
544
713
|
url = normalizeUrlScheme(url);
|
|
545
714
|
const [allowed, reason, hostname] = checkUrlAccess(url, policy);
|
|
546
715
|
if (!allowed) return { error: reason, body: "", contentType: "" };
|
|
547
716
|
|
|
548
|
-
|
|
549
|
-
if (
|
|
550
|
-
let resolved = await
|
|
717
|
+
const budgetResult = budgetExceededResult(deadline, signal, now);
|
|
718
|
+
if (budgetResult !== null) return budgetResult;
|
|
719
|
+
let resolved = await resolveWithBudget(hostname);
|
|
551
720
|
if (!resolved.ok) return { error: resolved.reason, body: "", contentType: "" };
|
|
552
721
|
|
|
553
722
|
let currentUrl = url;
|
|
@@ -556,8 +725,8 @@ export async function fetchUrlRaw(
|
|
|
556
725
|
const userAgent = randomUserAgent();
|
|
557
726
|
|
|
558
727
|
for (let hop = 0; hop < MAX_REQUESTS; hop++) {
|
|
559
|
-
|
|
560
|
-
if (
|
|
728
|
+
const budgetResult = budgetExceededResult(deadline, signal, now);
|
|
729
|
+
if (budgetResult !== null) return budgetResult;
|
|
561
730
|
const parsed = new URL(currentUrl);
|
|
562
731
|
const hostHeader = parsed.hostname + (parsed.port ? `:${parsed.port}` : "");
|
|
563
732
|
const headers: Record<string, string> = {
|
|
@@ -583,26 +752,12 @@ export async function fetchUrlRaw(
|
|
|
583
752
|
signal,
|
|
584
753
|
});
|
|
585
754
|
} catch (err) {
|
|
586
|
-
|
|
587
|
-
return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
|
|
588
|
-
if (err instanceof FetchTimeoutError)
|
|
589
|
-
return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
|
|
590
|
-
const message = err instanceof Error ? err.message : String(err);
|
|
591
|
-
if (message === "cancelled")
|
|
592
|
-
return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
|
|
593
|
-
if (message === "timed out")
|
|
594
|
-
return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
|
|
595
|
-
return { error: `Failed to fetch URL: ${message}`, body: "", contentType: "" };
|
|
755
|
+
return { error: fetchErrorMessage(err), body: "", contentType: "" };
|
|
596
756
|
}
|
|
597
757
|
|
|
598
758
|
if (response.status >= 300 && response.status < 400) {
|
|
599
759
|
if (![301, 302, 303, 307, 308].includes(response.status)) {
|
|
600
|
-
|
|
601
|
-
return {
|
|
602
|
-
error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
|
|
603
|
-
body: "",
|
|
604
|
-
contentType: "",
|
|
605
|
-
};
|
|
760
|
+
return statusErrorResult(response.status);
|
|
606
761
|
}
|
|
607
762
|
const rawLocation = response.headers.location;
|
|
608
763
|
const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
|
|
@@ -623,29 +778,22 @@ export async function fetchUrlRaw(
|
|
|
623
778
|
policy,
|
|
624
779
|
);
|
|
625
780
|
if (!redirectAllowed) return { error: redirectReason, body: "", contentType: "" };
|
|
626
|
-
const redirected = await
|
|
781
|
+
const redirected = await resolveWithBudget(redirectHost);
|
|
627
782
|
if (!redirected.ok) return { error: redirected.reason, body: "", contentType: "" };
|
|
628
783
|
pinnedIp = redirected.ip;
|
|
629
784
|
pinnedFamily = redirected.family;
|
|
630
785
|
continue;
|
|
631
786
|
}
|
|
632
787
|
|
|
633
|
-
|
|
634
|
-
if (
|
|
788
|
+
const postBudgetResult = budgetExceededResult(deadline, signal, now);
|
|
789
|
+
if (postBudgetResult !== null) return postBudgetResult;
|
|
635
790
|
|
|
636
791
|
if (response.status >= 400) {
|
|
637
|
-
|
|
638
|
-
return {
|
|
639
|
-
error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
|
|
640
|
-
body: "",
|
|
641
|
-
contentType: "",
|
|
642
|
-
};
|
|
792
|
+
return statusErrorResult(response.status);
|
|
643
793
|
}
|
|
644
794
|
|
|
645
795
|
const contentTypeHeader = response.headers["content-type"];
|
|
646
|
-
const contentType = contentTypeHeader
|
|
647
|
-
? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
|
|
648
|
-
: "";
|
|
796
|
+
const contentType = parseContentType(contentTypeHeader ? String(contentTypeHeader).toLowerCase() : null);
|
|
649
797
|
const declaredCharset = contentTypeHeader
|
|
650
798
|
? (/charset=([^;\s]+)/i.exec(String(contentTypeHeader))?.[1] ?? null)
|
|
651
799
|
: null;
|
|
@@ -664,16 +812,23 @@ export async function fetchUrlRaw(
|
|
|
664
812
|
try {
|
|
665
813
|
pdfText = await extractPdfText(response.body);
|
|
666
814
|
} catch {
|
|
667
|
-
return {
|
|
815
|
+
return {
|
|
816
|
+
error: response.truncated
|
|
817
|
+
? "(PDF content could not be read as text; the download was truncated at the download limit)"
|
|
818
|
+
: "(PDF content could not be read as text)",
|
|
819
|
+
body: "",
|
|
820
|
+
contentType,
|
|
821
|
+
};
|
|
668
822
|
}
|
|
669
|
-
|
|
670
|
-
if (
|
|
823
|
+
const budgetResult = budgetExceededResult(deadline, signal, now, contentType);
|
|
824
|
+
if (budgetResult !== null) return budgetResult;
|
|
671
825
|
if (!pdfText) pdfText = "(PDF contains no extractable text)";
|
|
826
|
+
if (response.truncated) pdfText += TRUNCATED_BODY_NOTICE;
|
|
672
827
|
return { error: null, body: pdfText, contentType: "application/pdf" };
|
|
673
828
|
}
|
|
674
829
|
|
|
675
830
|
if (!isTextCandidateContentType(contentType)) {
|
|
676
|
-
const safeType =
|
|
831
|
+
const safeType = parseContentType(contentType) || "unknown type";
|
|
677
832
|
return {
|
|
678
833
|
error: `(non-text content: ${safeType}, ${response.body.length} bytes; not readable as text)`,
|
|
679
834
|
body: "",
|
|
@@ -691,10 +846,11 @@ export async function fetchUrlRaw(
|
|
|
691
846
|
|
|
692
847
|
const declaredCodec = declaredCharset ? normalizeCharset(declaredCharset) : null;
|
|
693
848
|
const bomCodec = bomCodecFor(response.body);
|
|
694
|
-
const rawHtml =
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
849
|
+
const rawHtml =
|
|
850
|
+
decodeWithCodec(
|
|
851
|
+
response.body,
|
|
852
|
+
bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
|
|
853
|
+
) + (response.truncated ? TRUNCATED_BODY_NOTICE : "");
|
|
698
854
|
|
|
699
855
|
if (looksBinary(rawHtml)) {
|
|
700
856
|
let alt: string | null = null;
|
|
@@ -706,6 +862,7 @@ export async function fetchUrlRaw(
|
|
|
706
862
|
if (!looksBinary(candidate)) alt = candidate;
|
|
707
863
|
}
|
|
708
864
|
if (alt !== null) {
|
|
865
|
+
if (response.truncated) alt += TRUNCATED_BODY_NOTICE;
|
|
709
866
|
return { error: null, body: alt, contentType };
|
|
710
867
|
}
|
|
711
868
|
return {
|
|
@@ -732,16 +889,147 @@ function bomCodecFor(bytes: Buffer): string | null {
|
|
|
732
889
|
export function truncatePageText(text: string, maxChars?: number): string {
|
|
733
890
|
if (!text) return "(page returned no readable text)";
|
|
734
891
|
if (typeof maxChars === "number" && maxChars > 0 && text.length > maxChars) {
|
|
735
|
-
|
|
892
|
+
const hadCapNotice = text.endsWith(TRUNCATED_BODY_SUFFIX);
|
|
893
|
+
const core = (hadCapNotice ? text.slice(0, -TRUNCATED_BODY_SUFFIX.length) : text).trimEnd();
|
|
894
|
+
return (
|
|
895
|
+
cutAtCharBoundary(core, maxChars) +
|
|
896
|
+
`\n\n... (truncated, ${text.length} chars total)` +
|
|
897
|
+
(hadCapNotice ? TRUNCATED_BODY_NOTICE : "")
|
|
898
|
+
);
|
|
736
899
|
}
|
|
737
900
|
return text;
|
|
738
901
|
}
|
|
739
902
|
|
|
903
|
+
const NON_DOCUMENT_TITLE_TAGS = new Set(["math", "noscript", "svg", "template"]);
|
|
904
|
+
|
|
905
|
+
function extractPageTitle(html: string): string {
|
|
906
|
+
let inTitle = false;
|
|
907
|
+
let done = false;
|
|
908
|
+
let skipDepth = 0;
|
|
909
|
+
const parts: string[] = [];
|
|
910
|
+
feedHtml(html, {
|
|
911
|
+
handleStartTag(name: string) {
|
|
912
|
+
if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
|
|
913
|
+
skipDepth++;
|
|
914
|
+
return;
|
|
915
|
+
}
|
|
916
|
+
if (skipDepth) return;
|
|
917
|
+
if (name === "title" && !done) {
|
|
918
|
+
inTitle = true;
|
|
919
|
+
} else if (inTitle) {
|
|
920
|
+
inTitle = false;
|
|
921
|
+
}
|
|
922
|
+
},
|
|
923
|
+
handleStartEndTag() {},
|
|
924
|
+
handleEndTag(name: string) {
|
|
925
|
+
if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
|
|
926
|
+
skipDepth = Math.max(0, skipDepth - 1);
|
|
927
|
+
return;
|
|
928
|
+
}
|
|
929
|
+
if (skipDepth) return;
|
|
930
|
+
if (name === "title") {
|
|
931
|
+
inTitle = false;
|
|
932
|
+
done = true;
|
|
933
|
+
}
|
|
934
|
+
},
|
|
935
|
+
handleData(text: string) {
|
|
936
|
+
if (inTitle) parts.push(text);
|
|
937
|
+
},
|
|
938
|
+
handleEntityRef(name: string) {
|
|
939
|
+
if (inTitle) parts.push(decodeHtmlEntities(`&${name};`));
|
|
940
|
+
},
|
|
941
|
+
handleCharRef(name: string) {
|
|
942
|
+
if (inTitle) parts.push(decodeHtmlEntities(`&#${name};`));
|
|
943
|
+
},
|
|
944
|
+
});
|
|
945
|
+
return collapseWhitespace(parts.join(""));
|
|
946
|
+
}
|
|
947
|
+
|
|
948
|
+
interface PageMeta {
|
|
949
|
+
title: string;
|
|
950
|
+
author: string;
|
|
951
|
+
date: string;
|
|
952
|
+
site: string;
|
|
953
|
+
}
|
|
954
|
+
|
|
955
|
+
const META_KEYS: Record<string, keyof PageMeta> = {
|
|
956
|
+
author: "author",
|
|
957
|
+
"article:author": "author",
|
|
958
|
+
"dc.creator": "author",
|
|
959
|
+
"article:published_time": "date",
|
|
960
|
+
date: "date",
|
|
961
|
+
"dc.date": "date",
|
|
962
|
+
datepublished: "date",
|
|
963
|
+
"og:site_name": "site",
|
|
964
|
+
"application-name": "site",
|
|
965
|
+
};
|
|
966
|
+
|
|
967
|
+
function cutAtCharBoundary(text: string, maxChars: number): string {
|
|
968
|
+
const sliced = text.slice(0, maxChars);
|
|
969
|
+
const last = sliced.charCodeAt(sliced.length - 1);
|
|
970
|
+
return last >= 0xd800 && last <= 0xdbff ? sliced.slice(0, -1) : sliced;
|
|
971
|
+
}
|
|
972
|
+
|
|
973
|
+
function capMetaValue(value: string): string {
|
|
974
|
+
if (value.length <= META_VALUE_MAX_CHARS) return value;
|
|
975
|
+
return cutAtCharBoundary(value, META_VALUE_MAX_CHARS);
|
|
976
|
+
}
|
|
977
|
+
|
|
978
|
+
function extractPageMeta(html: string): PageMeta {
|
|
979
|
+
const meta: PageMeta = { title: extractPageTitle(html), author: "", date: "", site: "" };
|
|
980
|
+
const seen = new Set<string>();
|
|
981
|
+
const record = (name: string, attrs: AttrDict) => {
|
|
982
|
+
if (name !== "meta") return;
|
|
983
|
+
const key = (attrs["property"] ?? attrs["name"] ?? "").toLowerCase();
|
|
984
|
+
const content = collapseWhitespace(attrs["content"] ?? "");
|
|
985
|
+
if (!key || !content || seen.has(key)) return;
|
|
986
|
+
seen.add(key);
|
|
987
|
+
const field = META_KEYS[key];
|
|
988
|
+
if (field && !meta[field]) meta[field] = capMetaValue(content);
|
|
989
|
+
};
|
|
990
|
+
feedHtml(html, {
|
|
991
|
+
handleStartTag(name: string, attrs: AttrDict) {
|
|
992
|
+
record(name, attrs);
|
|
993
|
+
},
|
|
994
|
+
handleStartEndTag(name: string, attrs: AttrDict) {
|
|
995
|
+
record(name, attrs);
|
|
996
|
+
},
|
|
997
|
+
handleEndTag() {},
|
|
998
|
+
handleData() {},
|
|
999
|
+
handleEntityRef() {},
|
|
1000
|
+
handleCharRef() {},
|
|
1001
|
+
});
|
|
1002
|
+
return meta;
|
|
1003
|
+
}
|
|
1004
|
+
|
|
1005
|
+
function pagePrefixedMarkdown(html: string): string {
|
|
1006
|
+
const meta = extractPageMeta(html);
|
|
1007
|
+
const lines: string[] = [];
|
|
1008
|
+
if (meta.title) lines.push(`Title: ${meta.title}`);
|
|
1009
|
+
if (meta.author) lines.push(`Author: ${meta.author}`);
|
|
1010
|
+
if (meta.date) lines.push(`Date: ${meta.date}`);
|
|
1011
|
+
if (meta.site) lines.push(`Site: ${meta.site}`);
|
|
1012
|
+
const markdown = htmlToMarkdown(html, true);
|
|
1013
|
+
const converted = lines.length ? `${lines.join("\n")}\n\n${markdown}` : markdown;
|
|
1014
|
+
if (html.endsWith(TRUNCATED_BODY_SUFFIX) && !converted.endsWith(TRUNCATED_BODY_SUFFIX)) {
|
|
1015
|
+
return converted + TRUNCATED_BODY_NOTICE;
|
|
1016
|
+
}
|
|
1017
|
+
return converted;
|
|
1018
|
+
}
|
|
1019
|
+
|
|
1020
|
+
function formatReadmeBody(body: string): string {
|
|
1021
|
+
if (looksLikeHtmlDocument(body)) {
|
|
1022
|
+
const converted = pagePrefixedMarkdown(body);
|
|
1023
|
+
if (converted.trim()) return converted;
|
|
1024
|
+
}
|
|
1025
|
+
return body;
|
|
1026
|
+
}
|
|
1027
|
+
|
|
740
1028
|
export async function fetchPageText(
|
|
741
1029
|
url: string,
|
|
742
1030
|
options: FetchPageOptions = {},
|
|
743
1031
|
): Promise<string> {
|
|
744
|
-
const timeoutMs = options.timeoutMs ??
|
|
1032
|
+
const timeoutMs = options.timeoutMs ?? DEFAULT_FETCH_TIMEOUT_MS;
|
|
745
1033
|
const now = options.nowMs ?? Date.now;
|
|
746
1034
|
const deadlineMs = options.deadlineMs ?? now() + timeoutMs;
|
|
747
1035
|
const signal = options.signal;
|
|
@@ -752,48 +1040,49 @@ export async function fetchPageText(
|
|
|
752
1040
|
url = normalizeUrlScheme(url);
|
|
753
1041
|
const [allowed, reason] = checkUrlAccess(url, policy);
|
|
754
1042
|
if (!allowed) return reason;
|
|
1043
|
+
const rawFetchOptions = {
|
|
1044
|
+
deadlineMs,
|
|
1045
|
+
signal,
|
|
1046
|
+
websitePolicy: policy,
|
|
1047
|
+
maxBytes: options.maxBytes,
|
|
1048
|
+
maxPdfBytes: options.maxPdfBytes,
|
|
1049
|
+
seams: options.seams,
|
|
1050
|
+
};
|
|
755
1051
|
|
|
756
1052
|
const readmeApiUrl = githubRepoReadmeApiUrl(url);
|
|
757
1053
|
if (readmeApiUrl) {
|
|
758
1054
|
const readmeResult = await rawFetch(readmeApiUrl, {
|
|
759
|
-
|
|
760
|
-
signal,
|
|
761
|
-
websitePolicy: policy,
|
|
762
|
-
maxBytes: options.maxBytes,
|
|
763
|
-
maxPdfBytes: options.maxPdfBytes,
|
|
764
|
-
seams: options.seams,
|
|
1055
|
+
...rawFetchOptions,
|
|
765
1056
|
extraHeaders: {
|
|
766
1057
|
Accept: "application/vnd.github.raw+json",
|
|
767
1058
|
"X-GitHub-Api-Version": "2022-11-28",
|
|
768
1059
|
},
|
|
769
1060
|
});
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
1061
|
+
const apiBody = readmeResult.error === null ? formatReadmeBody(readmeResult.body) : "";
|
|
1062
|
+
if (apiBody.trim()) {
|
|
1063
|
+
return truncatePageText(
|
|
1064
|
+
`README of ${url} (fetched via the GitHub README API):\n\n` + apiBody,
|
|
1065
|
+
maxChars,
|
|
1066
|
+
);
|
|
1067
|
+
}
|
|
1068
|
+
const rawReadmeUrl = githubRepoRawReadmeUrl(url);
|
|
1069
|
+
if (rawReadmeUrl) {
|
|
1070
|
+
const rawResult = await rawFetch(rawReadmeUrl, rawFetchOptions);
|
|
1071
|
+
const rawBody = rawResult.error === null ? formatReadmeBody(rawResult.body) : "";
|
|
1072
|
+
if (rawBody.trim()) {
|
|
777
1073
|
return truncatePageText(
|
|
778
|
-
`README of ${url} (fetched via the GitHub README
|
|
1074
|
+
`README of ${url} (fetched via the GitHub raw README URL):\n\n` + rawBody,
|
|
779
1075
|
maxChars,
|
|
780
1076
|
);
|
|
781
1077
|
}
|
|
782
1078
|
}
|
|
783
1079
|
}
|
|
784
1080
|
|
|
785
|
-
const result = await rawFetch(url,
|
|
786
|
-
deadlineMs,
|
|
787
|
-
signal,
|
|
788
|
-
websitePolicy: policy,
|
|
789
|
-
maxBytes: options.maxBytes,
|
|
790
|
-
maxPdfBytes: options.maxPdfBytes,
|
|
791
|
-
seams: options.seams,
|
|
792
|
-
});
|
|
1081
|
+
const result = await rawFetch(url, rawFetchOptions);
|
|
793
1082
|
if (result.error !== null) return result.error;
|
|
794
1083
|
|
|
795
1084
|
const isHtml = result.contentType.includes("html") || looksLikeHtml(result.body);
|
|
796
1085
|
if (!isHtml) return truncatePageText(result.body.trim(), maxChars);
|
|
797
1086
|
|
|
798
|
-
return truncatePageText(
|
|
1087
|
+
return truncatePageText(pagePrefixedMarkdown(result.body), maxChars);
|
|
799
1088
|
}
|