pi-unsloth-webtools 0.2.5 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/web-fetch.ts CHANGED
@@ -1,22 +1,31 @@
1
1
  import { lookup as dnsLookup } from "node:dns/promises";
2
+ import type { LookupAllOptions } from "node:dns";
2
3
  import http from "node:http";
3
4
  import https from "node:https";
5
+ import { createBrotliDecompress, createGunzip, createInflate, createInflateRaw } from "node:zlib";
4
6
  import type { IncomingMessage } from "node:http";
7
+ import type { Transform } from "node:stream";
5
8
  import {
6
9
  checkUrlAccess,
10
+ githubRepoRawReadmeUrl,
7
11
  githubRepoReadmeApiUrl,
8
12
  isPublicIp,
9
13
  normalizeUrlScheme,
10
14
  type WebsitePolicy,
11
15
  } from "./web-access.ts";
12
- import { htmlToMarkdown } from "./html-to-md.ts";
16
+ import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown } from "./html-to-md.ts";
17
+ import type { AttrDict } from "./html-to-md.ts";
13
18
  import { INVALID_CHARREFS } from "./entities.ts";
14
- import { extractPdfText, PdfParseError } from "./pdf.ts";
19
+ import { extractPdfText } from "./pdf.ts";
15
20
  import { randomUserAgent } from "./user-agents.ts";
16
21
 
17
22
  const MAX_FETCH_BYTES = 512 * 1024;
23
+ const MAX_DECOMPRESSED_BYTES = 64 * 1024 * 1024;
24
+ const META_VALUE_MAX_CHARS = 300;
18
25
  const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
19
26
  const MAX_REQUESTS = 5;
27
+ const MAX_SIGNAL_TIMEOUT_MS = 2 ** 31 - 1;
28
+ export const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
20
29
 
21
30
  const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
22
31
  const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
@@ -109,6 +118,28 @@ export class FetchTimeoutError extends Error {
109
118
  super("timed out");
110
119
  }
111
120
  }
121
+ const FETCH_CANCELLED_MESSAGE = "Failed to fetch URL: cancelled.";
122
+ const FETCH_TIMEOUT_MESSAGE = "Failed to fetch URL: timed out.";
123
+ const TRUNCATED_BODY_NOTICE = "\n\n... (page truncated at the download limit)";
124
+ const TRUNCATED_BODY_SUFFIX = "... (page truncated at the download limit)";
125
+
126
+ function fetchErrorMessage(err: unknown): string {
127
+ if (err instanceof FetchCancelledError) return FETCH_CANCELLED_MESSAGE;
128
+ if (err instanceof FetchTimeoutError) return FETCH_TIMEOUT_MESSAGE;
129
+ const message = err instanceof Error ? err.message : String(err);
130
+ if (message === "cancelled") return FETCH_CANCELLED_MESSAGE;
131
+ if (message === "timed out") return FETCH_TIMEOUT_MESSAGE;
132
+ return `Failed to fetch URL: ${message}`;
133
+ }
134
+
135
+ function statusErrorResult(status: number): RawFetchResult {
136
+ const reason = http.STATUS_CODES[status] ?? "";
137
+ return {
138
+ error: `Failed to fetch URL: HTTP ${status}${reason ? ` ${reason}` : ""}`,
139
+ body: "",
140
+ contentType: "",
141
+ };
142
+ }
112
143
 
113
144
  export interface FetchPageOptions {
114
145
  timeoutMs?: number;
@@ -127,6 +158,7 @@ export interface HopResponse {
127
158
  status: number;
128
159
  headers: Record<string, string | string[] | undefined>;
129
160
  body: Buffer;
161
+ truncated?: boolean;
130
162
  }
131
163
 
132
164
  export interface ResolvedHost {
@@ -134,6 +166,7 @@ export interface ResolvedHost {
134
166
  reason: string;
135
167
  ip: string;
136
168
  family: number;
169
+ alternates?: { ip: string; family: number }[];
137
170
  }
138
171
 
139
172
  export interface FetchSeams {
@@ -172,20 +205,43 @@ export interface RawFetchResult {
172
205
  contentType: string;
173
206
  }
174
207
 
208
+ function htmlProbe(body: string, re: RegExp): boolean {
209
+ let i = 0;
210
+ const n = body.length;
211
+ while (i < n) {
212
+ while (i < n && /[ \t\n\r\f\v]/.test(body[i])) i++;
213
+ if (body.startsWith("<!--", i)) {
214
+ const close = body.indexOf("-->", i + 4);
215
+ if (close === -1) break;
216
+ i = close + 3;
217
+ continue;
218
+ }
219
+ if (body.startsWith("<?", i)) {
220
+ const close = body.indexOf("?>", i + 2);
221
+ if (close === -1) break;
222
+ i = close + 2;
223
+ continue;
224
+ }
225
+ break;
226
+ }
227
+ return re.test(body.slice(i, i + 256).toLowerCase());
228
+ }
229
+
175
230
  export function looksLikeHtml(body: string): boolean {
176
- const probe = body.replace(/^[ \t\n\r\f\v]+/, "").slice(0, 256).toLowerCase();
177
- return HTML_LEADING_RE.test(probe);
231
+ return htmlProbe(body, HTML_LEADING_RE);
178
232
  }
179
233
 
180
234
  export function looksLikeHtmlDocument(body: string): boolean {
181
- const probe = body.replace(/^[ \t\n\r\f\v]+/, "").slice(0, 256).toLowerCase();
182
- return HTML_DOCUMENT_RE.test(probe);
235
+ return htmlProbe(body, HTML_DOCUMENT_RE);
236
+ }
237
+
238
+ function parseContentType(value: string | null | undefined): string {
239
+ return /^[\w.+-]+\/[\w.+-]+/.exec(value ?? "")?.[0] ?? "";
183
240
  }
184
241
 
185
242
  export function isTextCandidateContentType(contentType: string | null): boolean {
186
- const match = /^[\w.+-]+\/[\w.+-]+/.exec(contentType ?? "");
187
- if (!match) return true;
188
- const ct = match[0].toLowerCase();
243
+ const ct = parseContentType(contentType).toLowerCase();
244
+ if (!ct) return true;
189
245
  if (ct.startsWith("text/")) return true;
190
246
  if (ct.startsWith("application/")) {
191
247
  const subtype = ct.slice("application/".length);
@@ -321,7 +377,12 @@ function sniffMetaCharsetForHtml(bytes: Buffer, contentType: string): string | n
321
377
 
322
378
  function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
323
379
  let out = "";
324
- for (let i = 0; i + 3 < bytes.length; i += 4) {
380
+ let i = 0;
381
+ if (bytes.length >= 4) {
382
+ const first = littleEndian ? bytes.readUInt32LE(0) : bytes.readUInt32BE(0);
383
+ if (first === 0xfeff) i = 4;
384
+ }
385
+ for (; i + 3 < bytes.length; i += 4) {
325
386
  const v = littleEndian ? bytes.readUInt32LE(i) : bytes.readUInt32BE(i);
326
387
  if (v === 0 || v > 0x10ffff || (v >= 0xd800 && v <= 0xdfff)) {
327
388
  out += "\ufffd";
@@ -334,7 +395,8 @@ function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
334
395
 
335
396
  function decodeUtf16Be(bytes: Buffer): string {
336
397
  let out = "";
337
- for (let i = 0; i + 1 < bytes.length; i += 2) {
398
+ let i = bytes.length >= 2 && bytes.readUInt16BE(0) === 0xfeff ? 2 : 0;
399
+ for (; i + 1 < bytes.length; i += 2) {
338
400
  out += String.fromCharCode(bytes.readUInt16BE(i));
339
401
  }
340
402
  return out;
@@ -409,7 +471,12 @@ function decodeTis620(bytes: Buffer): string {
409
471
  async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promise<ResolvedHost> {
410
472
  let addresses: { address: string; family: number }[];
411
473
  try {
412
- addresses = await dnsLookup(hostname, { all: true, verbatim: true });
474
+ const lookupOptions: LookupAllOptions & { signal?: AbortSignal } = {
475
+ all: true,
476
+ verbatim: true,
477
+ signal,
478
+ };
479
+ addresses = await dnsLookup(hostname, lookupOptions);
413
480
  } catch (err) {
414
481
  return { ok: false, reason: `Failed to resolve host: ${err}`, ip: "", family: 0 };
415
482
  }
@@ -421,22 +488,44 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
421
488
  return { ok: false, reason: `Blocked: refusing to fetch the non-public address ${entry.address}.`, ip: "", family: 0 };
422
489
  }
423
490
  }
491
+ addresses.sort((a, b) => (a.family === 4 ? 0 : 1) - (b.family === 4 ? 0 : 1));
424
492
  const first = addresses[0];
425
- return { ok: true, reason: "", ip: first.address, family: first.family };
493
+ return {
494
+ ok: true,
495
+ reason: "",
496
+ ip: first.address,
497
+ family: first.family,
498
+ alternates: addresses.slice(1).map((entry) => ({ ip: entry.address, family: entry.family })),
499
+ };
426
500
  }
427
501
 
428
502
 
429
- function fetchBudgetExceeded(
503
+ function budgetExceededResult(
430
504
  deadline: number | null,
431
505
  signal: AbortSignal | undefined,
432
506
  now: () => number = Date.now,
433
- ): string | null {
434
- if (signal?.aborted) return "Failed to fetch URL: cancelled.";
435
- if (deadline !== null && now() >= deadline) return "Failed to fetch URL: timed out.";
507
+ contentType = "",
508
+ ): RawFetchResult | null {
509
+ if (signal?.aborted) return { error: FETCH_CANCELLED_MESSAGE, body: "", contentType };
510
+ if (deadline !== null && now() >= deadline) return { error: FETCH_TIMEOUT_MESSAGE, body: "", contentType };
436
511
  return null;
437
512
  }
438
513
 
439
514
 
515
+ function contentEncodingCodec(value: string | string[] | undefined): string | null {
516
+ const declared = Array.isArray(value) ? value[0] : value;
517
+ const codec = (declared ?? "").split(",", 1)[0].trim().toLowerCase();
518
+ if (codec === "gzip" || codec === "x-gzip" || codec === "deflate" || codec === "br") return codec;
519
+ return null;
520
+ }
521
+
522
+ function createDecodeStream(codec: string): Transform {
523
+ const options = { maxOutputLength: MAX_DECOMPRESSED_BYTES };
524
+ if (codec === "br") return createBrotliDecompress(options);
525
+ if (codec === "deflate") return createInflate(options);
526
+ return createGunzip(options);
527
+ }
528
+
440
529
  export function requestHop(opts: HopOptions): Promise<HopResponse> {
441
530
  return new Promise((resolve, reject) => {
442
531
  const url = opts.url;
@@ -460,12 +549,18 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
460
549
  action();
461
550
  };
462
551
  const request = transport.request(options, (res: IncomingMessage) => {
552
+ const codec = contentEncodingCodec(res.headers["content-encoding"]);
463
553
  const chunks: Buffer[] = [];
464
554
  let total = 0;
465
555
  let head = Buffer.alloc(0);
556
+ let truncated = false;
557
+ let decoder: Transform | null = null;
466
558
  const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
467
559
  let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
468
560
  let extendedForPdf = false;
561
+ const declaredLengthHeader = res.headers["content-length"];
562
+ const declaredLength =
563
+ declaredLengthHeader === undefined ? NaN : Number(declaredLengthHeader);
469
564
  const finish = (err: string | null, body: Buffer) => {
470
565
  settle(() => {
471
566
  if (err) {
@@ -477,44 +572,113 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
477
572
  status: res.statusCode ?? 0,
478
573
  headers: res.headers as Record<string, string | string[] | undefined>,
479
574
  body,
575
+ truncated,
480
576
  });
481
577
  }
482
578
  });
483
579
  };
484
- res.on("data", (chunk: Buffer) => {
580
+ const timeoutAndDestroy = () => {
581
+ decoder?.destroy();
582
+ res.destroy();
583
+ finish("timed out", Buffer.concat(chunks));
584
+ };
585
+ const acceptData = (chunk: Buffer) => {
485
586
  if (settled) return;
486
587
  const now = opts.nowMs ?? Date.now;
487
588
  if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
488
- res.destroy();
489
- finish("timed out", Buffer.concat(chunks));
589
+ timeoutAndDestroy();
490
590
  return;
491
591
  }
492
592
  if (head.length < 1024) {
493
593
  const need = 1024 - head.length;
494
594
  head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
495
595
  }
496
- if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
596
+ if (!declaredPdf && !extendedForPdf && total + chunk.length > limit) {
497
597
  if (hasPdfMagic(head)) {
498
598
  limit = opts.maxPdfBytes;
499
599
  extendedForPdf = true;
500
600
  }
501
601
  }
502
602
  const space = limit - total;
503
- if (space <= 0) {
504
- res.destroy();
505
- finish(null, Buffer.concat(chunks));
506
- return;
507
- }
508
603
  const take = chunk.subarray(0, Math.min(chunk.length, space));
509
604
  chunks.push(take);
510
605
  total += take.length;
511
606
  if (total >= limit) {
607
+ truncated =
608
+ codec !== null ||
609
+ take.length < chunk.length ||
610
+ (Number.isFinite(declaredLength) && declaredLength > total);
611
+ decoder?.destroy();
512
612
  res.destroy();
513
613
  finish(null, Buffer.concat(chunks));
514
614
  }
615
+ };
616
+ if (codec === null) {
617
+ res.on("data", (chunk: Buffer) => {
618
+ acceptData(chunk);
619
+ });
620
+ res.on("end", () => finish(null, Buffer.concat(chunks)));
621
+ } else {
622
+ let rawBuffer: Buffer[] = [];
623
+ let rawBufferBytes = 0;
624
+ let outputStarted = false;
625
+ let rawFallbackTried = false;
626
+ let resEnded = false;
627
+ const wire = (d: Transform) => {
628
+ d.on("data", (chunk: Buffer) => {
629
+ outputStarted = true;
630
+ acceptData(chunk);
631
+ });
632
+ d.on("end", () => finish(null, Buffer.concat(chunks)));
633
+ d.on("drain", () => res.resume());
634
+ d.on("error", (err: NodeJS.ErrnoException) => {
635
+ if (settled) return;
636
+ if (
637
+ codec === "deflate" &&
638
+ !rawFallbackTried &&
639
+ !outputStarted &&
640
+ err.code === "Z_DATA_ERROR"
641
+ ) {
642
+ rawFallbackTried = true;
643
+ d.destroy();
644
+ decoder = createInflateRaw({ maxOutputLength: MAX_DECOMPRESSED_BYTES });
645
+ wire(decoder);
646
+ for (const buffered of rawBuffer) decoder.write(buffered);
647
+ if (resEnded) decoder.end();
648
+ return;
649
+ }
650
+ if (outputStarted) {
651
+ truncated = true;
652
+ finish(null, Buffer.concat(chunks));
653
+ return;
654
+ }
655
+ finish(null, Buffer.concat(rawBuffer));
656
+ });
657
+ };
658
+ decoder = createDecodeStream(codec);
659
+ wire(decoder);
660
+ res.on("data", (chunk: Buffer) => {
661
+ if (settled) return;
662
+ const now = opts.nowMs ?? Date.now;
663
+ if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
664
+ timeoutAndDestroy();
665
+ return;
666
+ }
667
+ if (!outputStarted && !rawFallbackTried && rawBufferBytes < limit) {
668
+ rawBuffer.push(chunk);
669
+ rawBufferBytes += chunk.length;
670
+ }
671
+ if (!decoder!.write(chunk)) res.pause();
672
+ });
673
+ res.on("end", () => {
674
+ resEnded = true;
675
+ if (!settled) decoder!.end();
676
+ });
677
+ }
678
+ res.on("error", (err) => {
679
+ decoder?.destroy();
680
+ finish(err.message, Buffer.concat(chunks));
515
681
  });
516
- res.on("end", () => finish(null, Buffer.concat(chunks)));
517
- res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
518
682
  });
519
683
  const onAbort = () => request.destroy(new FetchCancelledError());
520
684
  opts.signal?.addEventListener("abort", onAbort, { once: true });
@@ -540,24 +704,34 @@ export async function fetchUrlRaw(
540
704
  const seams = options.seams ?? {};
541
705
  const resolveHost = seams.resolve ?? resolveAndValidate;
542
706
  const performRequest = seams.request ?? requestHop;
707
+ const resolveWithBudget = async (hostname: string): Promise<ResolvedHost> => {
708
+ const deadlineSignal = AbortSignal.timeout(Math.min(MAX_SIGNAL_TIMEOUT_MS, Math.max(1, deadline - now())));
709
+ const resolveSignal = signal ? AbortSignal.any([signal, deadlineSignal]) : deadlineSignal;
710
+ const resolved = await resolveHost(hostname, resolveSignal);
711
+ if (resolved.ok) return resolved;
712
+ if (signal?.aborted) return { ...resolved, reason: FETCH_CANCELLED_MESSAGE };
713
+ if (resolveSignal.aborted) return { ...resolved, reason: FETCH_TIMEOUT_MESSAGE };
714
+ return resolved;
715
+ };
543
716
 
544
717
  url = normalizeUrlScheme(url);
545
718
  const [allowed, reason, hostname] = checkUrlAccess(url, policy);
546
719
  if (!allowed) return { error: reason, body: "", contentType: "" };
547
720
 
548
- let budgetError = fetchBudgetExceeded(deadline, signal, now);
549
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
550
- let resolved = await resolveHost(hostname, signal);
721
+ const budgetResult = budgetExceededResult(deadline, signal, now);
722
+ if (budgetResult !== null) return budgetResult;
723
+ let resolved = await resolveWithBudget(hostname);
551
724
  if (!resolved.ok) return { error: resolved.reason, body: "", contentType: "" };
552
725
 
553
726
  let currentUrl = url;
554
727
  let pinnedIp = resolved.ip;
555
728
  let pinnedFamily = resolved.family;
729
+ let alternates: { ip: string; family: number }[] = resolved.alternates ?? [];
730
+ let alternateIndex = 0;
556
731
  const userAgent = randomUserAgent();
557
-
558
732
  for (let hop = 0; hop < MAX_REQUESTS; hop++) {
559
- budgetError = fetchBudgetExceeded(deadline, signal, now);
560
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
733
+ const budgetResult = budgetExceededResult(deadline, signal, now);
734
+ if (budgetResult !== null) return budgetResult;
561
735
  const parsed = new URL(currentUrl);
562
736
  const hostHeader = parsed.hostname + (parsed.port ? `:${parsed.port}` : "");
563
737
  const headers: Record<string, string> = {
@@ -583,26 +757,22 @@ export async function fetchUrlRaw(
583
757
  signal,
584
758
  });
585
759
  } catch (err) {
586
- if (err instanceof FetchCancelledError)
587
- return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
588
- if (err instanceof FetchTimeoutError)
589
- return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
590
- const message = err instanceof Error ? err.message : String(err);
591
- if (message === "cancelled")
592
- return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
593
- if (message === "timed out")
594
- return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
595
- return { error: `Failed to fetch URL: ${message}`, body: "", contentType: "" };
760
+ if (
761
+ !(err instanceof FetchCancelledError) &&
762
+ !(err instanceof FetchTimeoutError) &&
763
+ alternateIndex < alternates.length
764
+ ) {
765
+ const next = alternates[alternateIndex++];
766
+ pinnedIp = next.ip;
767
+ pinnedFamily = next.family;
768
+ continue;
769
+ }
770
+ return { error: fetchErrorMessage(err), body: "", contentType: "" };
596
771
  }
597
772
 
598
773
  if (response.status >= 300 && response.status < 400) {
599
774
  if (![301, 302, 303, 307, 308].includes(response.status)) {
600
- const reason = http.STATUS_CODES[response.status] ?? "";
601
- return {
602
- error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
603
- body: "",
604
- contentType: "",
605
- };
775
+ return statusErrorResult(response.status);
606
776
  }
607
777
  const rawLocation = response.headers.location;
608
778
  const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
@@ -623,29 +793,24 @@ export async function fetchUrlRaw(
623
793
  policy,
624
794
  );
625
795
  if (!redirectAllowed) return { error: redirectReason, body: "", contentType: "" };
626
- const redirected = await resolveHost(redirectHost, signal);
796
+ const redirected = await resolveWithBudget(redirectHost);
627
797
  if (!redirected.ok) return { error: redirected.reason, body: "", contentType: "" };
628
798
  pinnedIp = redirected.ip;
629
799
  pinnedFamily = redirected.family;
800
+ alternates = redirected.alternates ?? [];
801
+ alternateIndex = 0;
630
802
  continue;
631
803
  }
632
804
 
633
- budgetError = fetchBudgetExceeded(deadline, signal, now);
634
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
805
+ const postBudgetResult = budgetExceededResult(deadline, signal, now);
806
+ if (postBudgetResult !== null) return postBudgetResult;
635
807
 
636
808
  if (response.status >= 400) {
637
- const reason = http.STATUS_CODES[response.status] ?? "";
638
- return {
639
- error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
640
- body: "",
641
- contentType: "",
642
- };
809
+ return statusErrorResult(response.status);
643
810
  }
644
811
 
645
812
  const contentTypeHeader = response.headers["content-type"];
646
- const contentType = contentTypeHeader
647
- ? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
648
- : "";
813
+ const contentType = parseContentType(contentTypeHeader ? String(contentTypeHeader).toLowerCase() : null);
649
814
  const declaredCharset = contentTypeHeader
650
815
  ? (/charset=([^;\s]+)/i.exec(String(contentTypeHeader))?.[1] ?? null)
651
816
  : null;
@@ -664,16 +829,23 @@ export async function fetchUrlRaw(
664
829
  try {
665
830
  pdfText = await extractPdfText(response.body);
666
831
  } catch {
667
- return { error: "(PDF content could not be read as text)", body: "", contentType };
832
+ return {
833
+ error: response.truncated
834
+ ? "(PDF content could not be read as text; the download was truncated at the download limit)"
835
+ : "(PDF content could not be read as text)",
836
+ body: "",
837
+ contentType,
838
+ };
668
839
  }
669
- budgetError = fetchBudgetExceeded(deadline, signal, now);
670
- if (budgetError !== null) return { error: budgetError, body: "", contentType };
840
+ const budgetResult = budgetExceededResult(deadline, signal, now, contentType);
841
+ if (budgetResult !== null) return budgetResult;
671
842
  if (!pdfText) pdfText = "(PDF contains no extractable text)";
843
+ if (response.truncated) pdfText += TRUNCATED_BODY_NOTICE;
672
844
  return { error: null, body: pdfText, contentType: "application/pdf" };
673
845
  }
674
846
 
675
847
  if (!isTextCandidateContentType(contentType)) {
676
- const safeType = /^[\w.+-]+\/[\w.+-]+/.exec(contentType ?? "")?.[0] ?? "unknown type";
848
+ const safeType = parseContentType(contentType) || "unknown type";
677
849
  return {
678
850
  error: `(non-text content: ${safeType}, ${response.body.length} bytes; not readable as text)`,
679
851
  body: "",
@@ -691,10 +863,11 @@ export async function fetchUrlRaw(
691
863
 
692
864
  const declaredCodec = declaredCharset ? normalizeCharset(declaredCharset) : null;
693
865
  const bomCodec = bomCodecFor(response.body);
694
- const rawHtml = decodeWithCodec(
695
- response.body,
696
- bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
697
- );
866
+ const rawHtml =
867
+ decodeWithCodec(
868
+ response.body,
869
+ bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
870
+ ) + (response.truncated ? TRUNCATED_BODY_NOTICE : "");
698
871
 
699
872
  if (looksBinary(rawHtml)) {
700
873
  let alt: string | null = null;
@@ -706,6 +879,7 @@ export async function fetchUrlRaw(
706
879
  if (!looksBinary(candidate)) alt = candidate;
707
880
  }
708
881
  if (alt !== null) {
882
+ if (response.truncated) alt += TRUNCATED_BODY_NOTICE;
709
883
  return { error: null, body: alt, contentType };
710
884
  }
711
885
  return {
@@ -732,16 +906,147 @@ function bomCodecFor(bytes: Buffer): string | null {
732
906
  export function truncatePageText(text: string, maxChars?: number): string {
733
907
  if (!text) return "(page returned no readable text)";
734
908
  if (typeof maxChars === "number" && maxChars > 0 && text.length > maxChars) {
735
- return text.slice(0, maxChars) + `\n\n... (truncated, ${text.length} chars total)`;
909
+ const hadCapNotice = text.endsWith(TRUNCATED_BODY_SUFFIX);
910
+ const core = (hadCapNotice ? text.slice(0, -TRUNCATED_BODY_SUFFIX.length) : text).trimEnd();
911
+ return (
912
+ cutAtCharBoundary(core, maxChars) +
913
+ `\n\n... (truncated, ${text.length} chars total)` +
914
+ (hadCapNotice ? TRUNCATED_BODY_NOTICE : "")
915
+ );
736
916
  }
737
917
  return text;
738
918
  }
739
919
 
920
+ const NON_DOCUMENT_TITLE_TAGS = new Set(["math", "noscript", "svg", "template"]);
921
+
922
+ function extractPageTitle(html: string): string {
923
+ let inTitle = false;
924
+ let done = false;
925
+ let skipDepth = 0;
926
+ const parts: string[] = [];
927
+ feedHtml(html, {
928
+ handleStartTag(name: string) {
929
+ if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
930
+ skipDepth++;
931
+ return;
932
+ }
933
+ if (skipDepth) return;
934
+ if (name === "title" && !done) {
935
+ inTitle = true;
936
+ } else if (inTitle) {
937
+ inTitle = false;
938
+ }
939
+ },
940
+ handleStartEndTag() {},
941
+ handleEndTag(name: string) {
942
+ if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
943
+ skipDepth = Math.max(0, skipDepth - 1);
944
+ return;
945
+ }
946
+ if (skipDepth) return;
947
+ if (name === "title") {
948
+ inTitle = false;
949
+ done = true;
950
+ }
951
+ },
952
+ handleData(text: string) {
953
+ if (inTitle) parts.push(text);
954
+ },
955
+ handleEntityRef(name: string) {
956
+ if (inTitle) parts.push(decodeHtmlEntities(`&${name};`));
957
+ },
958
+ handleCharRef(name: string) {
959
+ if (inTitle) parts.push(decodeHtmlEntities(`&#${name};`));
960
+ },
961
+ });
962
+ return collapseWhitespace(parts.join(""));
963
+ }
964
+
965
+ interface PageMeta {
966
+ title: string;
967
+ author: string;
968
+ date: string;
969
+ site: string;
970
+ }
971
+
972
+ const META_KEYS: Record<string, keyof PageMeta> = {
973
+ author: "author",
974
+ "article:author": "author",
975
+ "dc.creator": "author",
976
+ "article:published_time": "date",
977
+ date: "date",
978
+ "dc.date": "date",
979
+ datepublished: "date",
980
+ "og:site_name": "site",
981
+ "application-name": "site",
982
+ };
983
+
984
+ function cutAtCharBoundary(text: string, maxChars: number): string {
985
+ const sliced = text.slice(0, maxChars);
986
+ const last = sliced.charCodeAt(sliced.length - 1);
987
+ return last >= 0xd800 && last <= 0xdbff ? sliced.slice(0, -1) : sliced;
988
+ }
989
+
990
+ function capMetaValue(value: string): string {
991
+ if (value.length <= META_VALUE_MAX_CHARS) return value;
992
+ return cutAtCharBoundary(value, META_VALUE_MAX_CHARS);
993
+ }
994
+
995
+ function extractPageMeta(html: string): PageMeta {
996
+ const meta: PageMeta = { title: extractPageTitle(html), author: "", date: "", site: "" };
997
+ const seen = new Set<string>();
998
+ const record = (name: string, attrs: AttrDict) => {
999
+ if (name !== "meta") return;
1000
+ const key = (attrs["property"] ?? attrs["name"] ?? "").toLowerCase();
1001
+ const content = collapseWhitespace(attrs["content"] ?? "");
1002
+ if (!key || !content || seen.has(key)) return;
1003
+ seen.add(key);
1004
+ const field = META_KEYS[key];
1005
+ if (field && !meta[field]) meta[field] = capMetaValue(content);
1006
+ };
1007
+ feedHtml(html, {
1008
+ handleStartTag(name: string, attrs: AttrDict) {
1009
+ record(name, attrs);
1010
+ },
1011
+ handleStartEndTag(name: string, attrs: AttrDict) {
1012
+ record(name, attrs);
1013
+ },
1014
+ handleEndTag() {},
1015
+ handleData() {},
1016
+ handleEntityRef() {},
1017
+ handleCharRef() {},
1018
+ });
1019
+ return meta;
1020
+ }
1021
+
1022
+ function pagePrefixedMarkdown(html: string): string {
1023
+ const meta = extractPageMeta(html);
1024
+ const lines: string[] = [];
1025
+ if (meta.title) lines.push(`Title: ${meta.title}`);
1026
+ if (meta.author) lines.push(`Author: ${meta.author}`);
1027
+ if (meta.date) lines.push(`Date: ${meta.date}`);
1028
+ if (meta.site) lines.push(`Site: ${meta.site}`);
1029
+ const markdown = htmlToMarkdown(html, true);
1030
+ const converted = lines.length ? `${lines.join("\n")}\n\n${markdown}` : markdown;
1031
+ if (html.endsWith(TRUNCATED_BODY_SUFFIX) && !converted.endsWith(TRUNCATED_BODY_SUFFIX)) {
1032
+ return converted + TRUNCATED_BODY_NOTICE;
1033
+ }
1034
+ return converted;
1035
+ }
1036
+
1037
+ function formatReadmeBody(body: string): string {
1038
+ if (looksLikeHtmlDocument(body)) {
1039
+ const converted = pagePrefixedMarkdown(body);
1040
+ if (converted.trim()) return converted;
1041
+ }
1042
+ return body;
1043
+ }
1044
+
740
1045
  export async function fetchPageText(
741
1046
  url: string,
742
1047
  options: FetchPageOptions = {},
743
1048
  ): Promise<string> {
744
- const timeoutMs = options.timeoutMs ?? 60_000;
1049
+ const timeoutMs = options.timeoutMs ?? DEFAULT_FETCH_TIMEOUT_MS;
745
1050
  const now = options.nowMs ?? Date.now;
746
1051
  const deadlineMs = options.deadlineMs ?? now() + timeoutMs;
747
1052
  const signal = options.signal;
@@ -752,48 +1057,49 @@ export async function fetchPageText(
752
1057
  url = normalizeUrlScheme(url);
753
1058
  const [allowed, reason] = checkUrlAccess(url, policy);
754
1059
  if (!allowed) return reason;
1060
+ const rawFetchOptions = {
1061
+ deadlineMs,
1062
+ signal,
1063
+ websitePolicy: policy,
1064
+ maxBytes: options.maxBytes,
1065
+ maxPdfBytes: options.maxPdfBytes,
1066
+ seams: options.seams,
1067
+ };
755
1068
 
756
1069
  const readmeApiUrl = githubRepoReadmeApiUrl(url);
757
1070
  if (readmeApiUrl) {
758
1071
  const readmeResult = await rawFetch(readmeApiUrl, {
759
- deadlineMs,
760
- signal,
761
- websitePolicy: policy,
762
- maxBytes: options.maxBytes,
763
- maxPdfBytes: options.maxPdfBytes,
764
- seams: options.seams,
1072
+ ...rawFetchOptions,
765
1073
  extraHeaders: {
766
1074
  Accept: "application/vnd.github.raw+json",
767
1075
  "X-GitHub-Api-Version": "2022-11-28",
768
1076
  },
769
1077
  });
770
- if (readmeResult.error === null && readmeResult.body.trim()) {
771
- let readmeBody = readmeResult.body;
772
- if (looksLikeHtmlDocument(readmeBody)) {
773
- const converted = htmlToMarkdown(readmeBody, true);
774
- if (converted.trim()) readmeBody = converted;
775
- }
776
- if (readmeBody.trim()) {
1078
+ const apiBody = readmeResult.error === null ? formatReadmeBody(readmeResult.body) : "";
1079
+ if (apiBody.trim()) {
1080
+ return truncatePageText(
1081
+ `README of ${url} (fetched via the GitHub README API):\n\n` + apiBody,
1082
+ maxChars,
1083
+ );
1084
+ }
1085
+ const rawReadmeUrl = githubRepoRawReadmeUrl(url);
1086
+ if (rawReadmeUrl) {
1087
+ const rawResult = await rawFetch(rawReadmeUrl, rawFetchOptions);
1088
+ const rawBody = rawResult.error === null ? formatReadmeBody(rawResult.body) : "";
1089
+ if (rawBody.trim()) {
777
1090
  return truncatePageText(
778
- `README of ${url} (fetched via the GitHub README API):\n\n` + readmeBody,
1091
+ `README of ${url} (fetched via the GitHub raw README URL):\n\n` + rawBody,
779
1092
  maxChars,
780
1093
  );
781
1094
  }
782
1095
  }
783
1096
  }
784
1097
 
785
- const result = await rawFetch(url, {
786
- deadlineMs,
787
- signal,
788
- websitePolicy: policy,
789
- maxBytes: options.maxBytes,
790
- maxPdfBytes: options.maxPdfBytes,
791
- seams: options.seams,
792
- });
1098
+ const result = await rawFetch(url, rawFetchOptions);
793
1099
  if (result.error !== null) return result.error;
794
1100
 
795
1101
  const isHtml = result.contentType.includes("html") || looksLikeHtml(result.body);
796
1102
  if (!isHtml) return truncatePageText(result.body.trim(), maxChars);
797
1103
 
798
- return truncatePageText(htmlToMarkdown(result.body, true), maxChars);
1104
+ return truncatePageText(pagePrefixedMarkdown(result.body), maxChars);
799
1105
  }