pi-unsloth-webtools 0.2.5 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/web-fetch.ts CHANGED
@@ -1,22 +1,30 @@
1
1
  import { lookup as dnsLookup } from "node:dns/promises";
2
2
  import http from "node:http";
3
3
  import https from "node:https";
4
+ import { createBrotliDecompress, createGunzip, createInflate, createInflateRaw } from "node:zlib";
4
5
  import type { IncomingMessage } from "node:http";
6
+ import type { Transform } from "node:stream";
5
7
  import {
6
8
  checkUrlAccess,
9
+ githubRepoRawReadmeUrl,
7
10
  githubRepoReadmeApiUrl,
8
11
  isPublicIp,
9
12
  normalizeUrlScheme,
10
13
  type WebsitePolicy,
11
14
  } from "./web-access.ts";
12
- import { htmlToMarkdown } from "./html-to-md.ts";
15
+ import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown } from "./html-to-md.ts";
16
+ import type { AttrDict } from "./html-to-md.ts";
13
17
  import { INVALID_CHARREFS } from "./entities.ts";
14
- import { extractPdfText, PdfParseError } from "./pdf.ts";
18
+ import { extractPdfText } from "./pdf.ts";
15
19
  import { randomUserAgent } from "./user-agents.ts";
16
20
 
17
21
  const MAX_FETCH_BYTES = 512 * 1024;
22
+ const MAX_DECOMPRESSED_BYTES = 64 * 1024 * 1024;
23
+ const META_VALUE_MAX_CHARS = 300;
18
24
  const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
19
25
  const MAX_REQUESTS = 5;
26
+ const MAX_SIGNAL_TIMEOUT_MS = 2 ** 31 - 1;
27
+ export const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
20
28
 
21
29
  const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
22
30
  const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
@@ -109,6 +117,28 @@ export class FetchTimeoutError extends Error {
109
117
  super("timed out");
110
118
  }
111
119
  }
120
+ const FETCH_CANCELLED_MESSAGE = "Failed to fetch URL: cancelled.";
121
+ const FETCH_TIMEOUT_MESSAGE = "Failed to fetch URL: timed out.";
122
+ const TRUNCATED_BODY_NOTICE = "\n\n... (page truncated at the download limit)";
123
+ const TRUNCATED_BODY_SUFFIX = "... (page truncated at the download limit)";
124
+
125
+ function fetchErrorMessage(err: unknown): string {
126
+ if (err instanceof FetchCancelledError) return FETCH_CANCELLED_MESSAGE;
127
+ if (err instanceof FetchTimeoutError) return FETCH_TIMEOUT_MESSAGE;
128
+ const message = err instanceof Error ? err.message : String(err);
129
+ if (message === "cancelled") return FETCH_CANCELLED_MESSAGE;
130
+ if (message === "timed out") return FETCH_TIMEOUT_MESSAGE;
131
+ return `Failed to fetch URL: ${message}`;
132
+ }
133
+
134
+ function statusErrorResult(status: number): RawFetchResult {
135
+ const reason = http.STATUS_CODES[status] ?? "";
136
+ return {
137
+ error: `Failed to fetch URL: HTTP ${status}${reason ? ` ${reason}` : ""}`,
138
+ body: "",
139
+ contentType: "",
140
+ };
141
+ }
112
142
 
113
143
  export interface FetchPageOptions {
114
144
  timeoutMs?: number;
@@ -127,6 +157,7 @@ export interface HopResponse {
127
157
  status: number;
128
158
  headers: Record<string, string | string[] | undefined>;
129
159
  body: Buffer;
160
+ truncated?: boolean;
130
161
  }
131
162
 
132
163
  export interface ResolvedHost {
@@ -172,20 +203,32 @@ export interface RawFetchResult {
172
203
  contentType: string;
173
204
  }
174
205
 
206
+ function htmlProbe(body: string, re: RegExp): boolean {
207
+ let probe = body;
208
+ while (true) {
209
+ probe = probe.replace(/^[ \t\n\r\f\v]+/, "");
210
+ const stripped = probe.replace(/^(?:<!--[\s\S]*?-->|<\?[\s\S]*?\?>)/, "");
211
+ if (stripped === probe) break;
212
+ probe = stripped;
213
+ }
214
+ return re.test(probe.slice(0, 256).toLowerCase());
215
+ }
216
+
175
217
  export function looksLikeHtml(body: string): boolean {
176
- const probe = body.replace(/^[ \t\n\r\f\v]+/, "").slice(0, 256).toLowerCase();
177
- return HTML_LEADING_RE.test(probe);
218
+ return htmlProbe(body, HTML_LEADING_RE);
178
219
  }
179
220
 
180
221
  export function looksLikeHtmlDocument(body: string): boolean {
181
- const probe = body.replace(/^[ \t\n\r\f\v]+/, "").slice(0, 256).toLowerCase();
182
- return HTML_DOCUMENT_RE.test(probe);
222
+ return htmlProbe(body, HTML_DOCUMENT_RE);
223
+ }
224
+
225
+ function parseContentType(value: string | null | undefined): string {
226
+ return /^[\w.+-]+\/[\w.+-]+/.exec(value ?? "")?.[0] ?? "";
183
227
  }
184
228
 
185
229
  export function isTextCandidateContentType(contentType: string | null): boolean {
186
- const match = /^[\w.+-]+\/[\w.+-]+/.exec(contentType ?? "");
187
- if (!match) return true;
188
- const ct = match[0].toLowerCase();
230
+ const ct = parseContentType(contentType).toLowerCase();
231
+ if (!ct) return true;
189
232
  if (ct.startsWith("text/")) return true;
190
233
  if (ct.startsWith("application/")) {
191
234
  const subtype = ct.slice("application/".length);
@@ -321,7 +364,12 @@ function sniffMetaCharsetForHtml(bytes: Buffer, contentType: string): string | n
321
364
 
322
365
  function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
323
366
  let out = "";
324
- for (let i = 0; i + 3 < bytes.length; i += 4) {
367
+ let i = 0;
368
+ if (bytes.length >= 4) {
369
+ const first = littleEndian ? bytes.readUInt32LE(0) : bytes.readUInt32BE(0);
370
+ if (first === 0xfeff) i = 4;
371
+ }
372
+ for (; i + 3 < bytes.length; i += 4) {
325
373
  const v = littleEndian ? bytes.readUInt32LE(i) : bytes.readUInt32BE(i);
326
374
  if (v === 0 || v > 0x10ffff || (v >= 0xd800 && v <= 0xdfff)) {
327
375
  out += "\ufffd";
@@ -334,7 +382,8 @@ function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
334
382
 
335
383
  function decodeUtf16Be(bytes: Buffer): string {
336
384
  let out = "";
337
- for (let i = 0; i + 1 < bytes.length; i += 2) {
385
+ let i = bytes.length >= 2 && bytes.readUInt16BE(0) === 0xfeff ? 2 : 0;
386
+ for (; i + 1 < bytes.length; i += 2) {
338
387
  out += String.fromCharCode(bytes.readUInt16BE(i));
339
388
  }
340
389
  return out;
@@ -406,10 +455,32 @@ function decodeTis620(bytes: Buffer): string {
406
455
  }
407
456
 
408
457
 
458
+ function withAbort<T>(promise: Promise<T>, signal?: AbortSignal): Promise<T> {
459
+ if (!signal) return promise;
460
+ return new Promise<T>((resolve, reject) => {
461
+ const onAbort = () => reject(new DOMException("aborted", "AbortError"));
462
+ if (signal.aborted) {
463
+ onAbort();
464
+ return;
465
+ }
466
+ signal.addEventListener("abort", onAbort, { once: true });
467
+ promise.then(
468
+ (value) => {
469
+ signal.removeEventListener("abort", onAbort);
470
+ resolve(value);
471
+ },
472
+ (err) => {
473
+ signal.removeEventListener("abort", onAbort);
474
+ reject(err);
475
+ },
476
+ );
477
+ });
478
+ }
479
+
409
480
  async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promise<ResolvedHost> {
410
481
  let addresses: { address: string; family: number }[];
411
482
  try {
412
- addresses = await dnsLookup(hostname, { all: true, verbatim: true });
483
+ addresses = await withAbort(dnsLookup(hostname, { all: true, verbatim: true }), signal);
413
484
  } catch (err) {
414
485
  return { ok: false, reason: `Failed to resolve host: ${err}`, ip: "", family: 0 };
415
486
  }
@@ -426,17 +497,32 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
426
497
  }
427
498
 
428
499
 
429
- function fetchBudgetExceeded(
500
+ function budgetExceededResult(
430
501
  deadline: number | null,
431
502
  signal: AbortSignal | undefined,
432
503
  now: () => number = Date.now,
433
- ): string | null {
434
- if (signal?.aborted) return "Failed to fetch URL: cancelled.";
435
- if (deadline !== null && now() >= deadline) return "Failed to fetch URL: timed out.";
504
+ contentType = "",
505
+ ): RawFetchResult | null {
506
+ if (signal?.aborted) return { error: FETCH_CANCELLED_MESSAGE, body: "", contentType };
507
+ if (deadline !== null && now() >= deadline) return { error: FETCH_TIMEOUT_MESSAGE, body: "", contentType };
436
508
  return null;
437
509
  }
438
510
 
439
511
 
512
+ function contentEncodingCodec(value: string | string[] | undefined): string | null {
513
+ const declared = Array.isArray(value) ? value[0] : value;
514
+ const codec = (declared ?? "").split(",", 1)[0].trim().toLowerCase();
515
+ if (codec === "gzip" || codec === "x-gzip" || codec === "deflate" || codec === "br") return codec;
516
+ return null;
517
+ }
518
+
519
+ function createDecodeStream(codec: string): Transform {
520
+ const options = { maxOutputLength: MAX_DECOMPRESSED_BYTES };
521
+ if (codec === "br") return createBrotliDecompress(options);
522
+ if (codec === "deflate") return createInflate(options);
523
+ return createGunzip(options);
524
+ }
525
+
440
526
  export function requestHop(opts: HopOptions): Promise<HopResponse> {
441
527
  return new Promise((resolve, reject) => {
442
528
  const url = opts.url;
@@ -460,12 +546,18 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
460
546
  action();
461
547
  };
462
548
  const request = transport.request(options, (res: IncomingMessage) => {
549
+ const codec = contentEncodingCodec(res.headers["content-encoding"]);
463
550
  const chunks: Buffer[] = [];
464
551
  let total = 0;
465
552
  let head = Buffer.alloc(0);
553
+ let truncated = false;
554
+ let decoder: Transform | null = null;
466
555
  const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
467
556
  let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
468
557
  let extendedForPdf = false;
558
+ const declaredLengthHeader = res.headers["content-length"];
559
+ const declaredLength =
560
+ declaredLengthHeader === undefined ? NaN : Number(declaredLengthHeader);
469
561
  const finish = (err: string | null, body: Buffer) => {
470
562
  settle(() => {
471
563
  if (err) {
@@ -477,44 +569,112 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
477
569
  status: res.statusCode ?? 0,
478
570
  headers: res.headers as Record<string, string | string[] | undefined>,
479
571
  body,
572
+ truncated,
480
573
  });
481
574
  }
482
575
  });
483
576
  };
484
- res.on("data", (chunk: Buffer) => {
577
+ const timeoutAndDestroy = () => {
578
+ decoder?.destroy();
579
+ res.destroy();
580
+ finish("timed out", Buffer.concat(chunks));
581
+ };
582
+ const acceptData = (chunk: Buffer) => {
485
583
  if (settled) return;
486
584
  const now = opts.nowMs ?? Date.now;
487
585
  if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
488
- res.destroy();
489
- finish("timed out", Buffer.concat(chunks));
586
+ timeoutAndDestroy();
490
587
  return;
491
588
  }
492
589
  if (head.length < 1024) {
493
590
  const need = 1024 - head.length;
494
591
  head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
495
592
  }
496
- if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
593
+ if (!declaredPdf && !extendedForPdf && total + chunk.length > limit) {
497
594
  if (hasPdfMagic(head)) {
498
595
  limit = opts.maxPdfBytes;
499
596
  extendedForPdf = true;
500
597
  }
501
598
  }
502
599
  const space = limit - total;
503
- if (space <= 0) {
504
- res.destroy();
505
- finish(null, Buffer.concat(chunks));
506
- return;
507
- }
508
600
  const take = chunk.subarray(0, Math.min(chunk.length, space));
509
601
  chunks.push(take);
510
602
  total += take.length;
511
603
  if (total >= limit) {
604
+ truncated =
605
+ codec !== null ||
606
+ !(Number.isFinite(declaredLength) && declaredLength === total);
607
+ decoder?.destroy();
512
608
  res.destroy();
513
609
  finish(null, Buffer.concat(chunks));
514
610
  }
611
+ };
612
+ if (codec === null) {
613
+ res.on("data", (chunk: Buffer) => {
614
+ acceptData(chunk);
615
+ });
616
+ res.on("end", () => finish(null, Buffer.concat(chunks)));
617
+ } else {
618
+ let rawBuffer: Buffer[] = [];
619
+ let rawBufferBytes = 0;
620
+ let outputStarted = false;
621
+ let rawFallbackTried = false;
622
+ let resEnded = false;
623
+ const wire = (d: Transform) => {
624
+ d.on("data", (chunk: Buffer) => {
625
+ outputStarted = true;
626
+ acceptData(chunk);
627
+ });
628
+ d.on("end", () => finish(null, Buffer.concat(chunks)));
629
+ d.on("drain", () => res.resume());
630
+ d.on("error", (err: NodeJS.ErrnoException) => {
631
+ if (settled) return;
632
+ if (
633
+ codec === "deflate" &&
634
+ !rawFallbackTried &&
635
+ !outputStarted &&
636
+ err.code === "Z_DATA_ERROR"
637
+ ) {
638
+ rawFallbackTried = true;
639
+ d.destroy();
640
+ decoder = createInflateRaw({ maxOutputLength: MAX_DECOMPRESSED_BYTES });
641
+ wire(decoder);
642
+ for (const buffered of rawBuffer) decoder.write(buffered);
643
+ if (resEnded) decoder.end();
644
+ return;
645
+ }
646
+ if (outputStarted) {
647
+ truncated = true;
648
+ finish(null, Buffer.concat(chunks));
649
+ return;
650
+ }
651
+ finish(null, Buffer.concat(rawBuffer));
652
+ });
653
+ };
654
+ decoder = createDecodeStream(codec);
655
+ wire(decoder);
656
+ res.on("data", (chunk: Buffer) => {
657
+ if (settled) return;
658
+ const now = opts.nowMs ?? Date.now;
659
+ if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
660
+ timeoutAndDestroy();
661
+ return;
662
+ }
663
+ if (!outputStarted && !rawFallbackTried && rawBufferBytes < limit) {
664
+ rawBuffer.push(chunk);
665
+ rawBufferBytes += chunk.length;
666
+ }
667
+ if (!decoder!.write(chunk)) res.pause();
668
+ });
669
+ res.on("end", () => {
670
+ resEnded = true;
671
+ if (!settled) decoder!.end();
672
+ });
673
+ }
674
+ res.on("error", (err) => {
675
+ decoder?.destroy();
676
+ finish(err.message, Buffer.concat(chunks));
515
677
  });
516
- res.on("end", () => finish(null, Buffer.concat(chunks)));
517
- res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
518
678
  });
519
679
  const onAbort = () => request.destroy(new FetchCancelledError());
520
680
  opts.signal?.addEventListener("abort", onAbort, { once: true });
@@ -540,14 +700,23 @@ export async function fetchUrlRaw(
540
700
  const seams = options.seams ?? {};
541
701
  const resolveHost = seams.resolve ?? resolveAndValidate;
542
702
  const performRequest = seams.request ?? requestHop;
703
+ const resolveWithBudget = async (hostname: string): Promise<ResolvedHost> => {
704
+ const deadlineSignal = AbortSignal.timeout(Math.min(MAX_SIGNAL_TIMEOUT_MS, Math.max(1, deadline - now())));
705
+ const resolveSignal = signal ? AbortSignal.any([signal, deadlineSignal]) : deadlineSignal;
706
+ const resolved = await resolveHost(hostname, resolveSignal);
707
+ if (resolved.ok) return resolved;
708
+ if (signal?.aborted) return { ...resolved, reason: FETCH_CANCELLED_MESSAGE };
709
+ if (resolveSignal.aborted) return { ...resolved, reason: FETCH_TIMEOUT_MESSAGE };
710
+ return resolved;
711
+ };
543
712
 
544
713
  url = normalizeUrlScheme(url);
545
714
  const [allowed, reason, hostname] = checkUrlAccess(url, policy);
546
715
  if (!allowed) return { error: reason, body: "", contentType: "" };
547
716
 
548
- let budgetError = fetchBudgetExceeded(deadline, signal, now);
549
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
550
- let resolved = await resolveHost(hostname, signal);
717
+ const budgetResult = budgetExceededResult(deadline, signal, now);
718
+ if (budgetResult !== null) return budgetResult;
719
+ let resolved = await resolveWithBudget(hostname);
551
720
  if (!resolved.ok) return { error: resolved.reason, body: "", contentType: "" };
552
721
 
553
722
  let currentUrl = url;
@@ -556,8 +725,8 @@ export async function fetchUrlRaw(
556
725
  const userAgent = randomUserAgent();
557
726
 
558
727
  for (let hop = 0; hop < MAX_REQUESTS; hop++) {
559
- budgetError = fetchBudgetExceeded(deadline, signal, now);
560
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
728
+ const budgetResult = budgetExceededResult(deadline, signal, now);
729
+ if (budgetResult !== null) return budgetResult;
561
730
  const parsed = new URL(currentUrl);
562
731
  const hostHeader = parsed.hostname + (parsed.port ? `:${parsed.port}` : "");
563
732
  const headers: Record<string, string> = {
@@ -583,26 +752,12 @@ export async function fetchUrlRaw(
583
752
  signal,
584
753
  });
585
754
  } catch (err) {
586
- if (err instanceof FetchCancelledError)
587
- return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
588
- if (err instanceof FetchTimeoutError)
589
- return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
590
- const message = err instanceof Error ? err.message : String(err);
591
- if (message === "cancelled")
592
- return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
593
- if (message === "timed out")
594
- return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
595
- return { error: `Failed to fetch URL: ${message}`, body: "", contentType: "" };
755
+ return { error: fetchErrorMessage(err), body: "", contentType: "" };
596
756
  }
597
757
 
598
758
  if (response.status >= 300 && response.status < 400) {
599
759
  if (![301, 302, 303, 307, 308].includes(response.status)) {
600
- const reason = http.STATUS_CODES[response.status] ?? "";
601
- return {
602
- error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
603
- body: "",
604
- contentType: "",
605
- };
760
+ return statusErrorResult(response.status);
606
761
  }
607
762
  const rawLocation = response.headers.location;
608
763
  const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
@@ -623,29 +778,22 @@ export async function fetchUrlRaw(
623
778
  policy,
624
779
  );
625
780
  if (!redirectAllowed) return { error: redirectReason, body: "", contentType: "" };
626
- const redirected = await resolveHost(redirectHost, signal);
781
+ const redirected = await resolveWithBudget(redirectHost);
627
782
  if (!redirected.ok) return { error: redirected.reason, body: "", contentType: "" };
628
783
  pinnedIp = redirected.ip;
629
784
  pinnedFamily = redirected.family;
630
785
  continue;
631
786
  }
632
787
 
633
- budgetError = fetchBudgetExceeded(deadline, signal, now);
634
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
788
+ const postBudgetResult = budgetExceededResult(deadline, signal, now);
789
+ if (postBudgetResult !== null) return postBudgetResult;
635
790
 
636
791
  if (response.status >= 400) {
637
- const reason = http.STATUS_CODES[response.status] ?? "";
638
- return {
639
- error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
640
- body: "",
641
- contentType: "",
642
- };
792
+ return statusErrorResult(response.status);
643
793
  }
644
794
 
645
795
  const contentTypeHeader = response.headers["content-type"];
646
- const contentType = contentTypeHeader
647
- ? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
648
- : "";
796
+ const contentType = parseContentType(contentTypeHeader ? String(contentTypeHeader).toLowerCase() : null);
649
797
  const declaredCharset = contentTypeHeader
650
798
  ? (/charset=([^;\s]+)/i.exec(String(contentTypeHeader))?.[1] ?? null)
651
799
  : null;
@@ -664,16 +812,23 @@ export async function fetchUrlRaw(
664
812
  try {
665
813
  pdfText = await extractPdfText(response.body);
666
814
  } catch {
667
- return { error: "(PDF content could not be read as text)", body: "", contentType };
815
+ return {
816
+ error: response.truncated
817
+ ? "(PDF content could not be read as text; the download was truncated at the download limit)"
818
+ : "(PDF content could not be read as text)",
819
+ body: "",
820
+ contentType,
821
+ };
668
822
  }
669
- budgetError = fetchBudgetExceeded(deadline, signal, now);
670
- if (budgetError !== null) return { error: budgetError, body: "", contentType };
823
+ const budgetResult = budgetExceededResult(deadline, signal, now, contentType);
824
+ if (budgetResult !== null) return budgetResult;
671
825
  if (!pdfText) pdfText = "(PDF contains no extractable text)";
826
+ if (response.truncated) pdfText += TRUNCATED_BODY_NOTICE;
672
827
  return { error: null, body: pdfText, contentType: "application/pdf" };
673
828
  }
674
829
 
675
830
  if (!isTextCandidateContentType(contentType)) {
676
- const safeType = /^[\w.+-]+\/[\w.+-]+/.exec(contentType ?? "")?.[0] ?? "unknown type";
831
+ const safeType = parseContentType(contentType) || "unknown type";
677
832
  return {
678
833
  error: `(non-text content: ${safeType}, ${response.body.length} bytes; not readable as text)`,
679
834
  body: "",
@@ -691,10 +846,11 @@ export async function fetchUrlRaw(
691
846
 
692
847
  const declaredCodec = declaredCharset ? normalizeCharset(declaredCharset) : null;
693
848
  const bomCodec = bomCodecFor(response.body);
694
- const rawHtml = decodeWithCodec(
695
- response.body,
696
- bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
697
- );
849
+ const rawHtml =
850
+ decodeWithCodec(
851
+ response.body,
852
+ bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
853
+ ) + (response.truncated ? TRUNCATED_BODY_NOTICE : "");
698
854
 
699
855
  if (looksBinary(rawHtml)) {
700
856
  let alt: string | null = null;
@@ -706,6 +862,7 @@ export async function fetchUrlRaw(
706
862
  if (!looksBinary(candidate)) alt = candidate;
707
863
  }
708
864
  if (alt !== null) {
865
+ if (response.truncated) alt += TRUNCATED_BODY_NOTICE;
709
866
  return { error: null, body: alt, contentType };
710
867
  }
711
868
  return {
@@ -732,16 +889,147 @@ function bomCodecFor(bytes: Buffer): string | null {
732
889
  export function truncatePageText(text: string, maxChars?: number): string {
733
890
  if (!text) return "(page returned no readable text)";
734
891
  if (typeof maxChars === "number" && maxChars > 0 && text.length > maxChars) {
735
- return text.slice(0, maxChars) + `\n\n... (truncated, ${text.length} chars total)`;
892
+ const hadCapNotice = text.endsWith(TRUNCATED_BODY_SUFFIX);
893
+ const core = (hadCapNotice ? text.slice(0, -TRUNCATED_BODY_SUFFIX.length) : text).trimEnd();
894
+ return (
895
+ cutAtCharBoundary(core, maxChars) +
896
+ `\n\n... (truncated, ${text.length} chars total)` +
897
+ (hadCapNotice ? TRUNCATED_BODY_NOTICE : "")
898
+ );
736
899
  }
737
900
  return text;
738
901
  }
739
902
 
903
+ const NON_DOCUMENT_TITLE_TAGS = new Set(["math", "noscript", "svg", "template"]);
904
+
905
+ function extractPageTitle(html: string): string {
906
+ let inTitle = false;
907
+ let done = false;
908
+ let skipDepth = 0;
909
+ const parts: string[] = [];
910
+ feedHtml(html, {
911
+ handleStartTag(name: string) {
912
+ if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
913
+ skipDepth++;
914
+ return;
915
+ }
916
+ if (skipDepth) return;
917
+ if (name === "title" && !done) {
918
+ inTitle = true;
919
+ } else if (inTitle) {
920
+ inTitle = false;
921
+ }
922
+ },
923
+ handleStartEndTag() {},
924
+ handleEndTag(name: string) {
925
+ if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
926
+ skipDepth = Math.max(0, skipDepth - 1);
927
+ return;
928
+ }
929
+ if (skipDepth) return;
930
+ if (name === "title") {
931
+ inTitle = false;
932
+ done = true;
933
+ }
934
+ },
935
+ handleData(text: string) {
936
+ if (inTitle) parts.push(text);
937
+ },
938
+ handleEntityRef(name: string) {
939
+ if (inTitle) parts.push(decodeHtmlEntities(`&${name};`));
940
+ },
941
+ handleCharRef(name: string) {
942
+ if (inTitle) parts.push(decodeHtmlEntities(`&#${name};`));
943
+ },
944
+ });
945
+ return collapseWhitespace(parts.join(""));
946
+ }
947
+
948
+ interface PageMeta {
949
+ title: string;
950
+ author: string;
951
+ date: string;
952
+ site: string;
953
+ }
954
+
955
+ const META_KEYS: Record<string, keyof PageMeta> = {
956
+ author: "author",
957
+ "article:author": "author",
958
+ "dc.creator": "author",
959
+ "article:published_time": "date",
960
+ date: "date",
961
+ "dc.date": "date",
962
+ datepublished: "date",
963
+ "og:site_name": "site",
964
+ "application-name": "site",
965
+ };
966
+
967
+ function cutAtCharBoundary(text: string, maxChars: number): string {
968
+ const sliced = text.slice(0, maxChars);
969
+ const last = sliced.charCodeAt(sliced.length - 1);
970
+ return last >= 0xd800 && last <= 0xdbff ? sliced.slice(0, -1) : sliced;
971
+ }
972
+
973
+ function capMetaValue(value: string): string {
974
+ if (value.length <= META_VALUE_MAX_CHARS) return value;
975
+ return cutAtCharBoundary(value, META_VALUE_MAX_CHARS);
976
+ }
977
+
978
+ function extractPageMeta(html: string): PageMeta {
979
+ const meta: PageMeta = { title: extractPageTitle(html), author: "", date: "", site: "" };
980
+ const seen = new Set<string>();
981
+ const record = (name: string, attrs: AttrDict) => {
982
+ if (name !== "meta") return;
983
+ const key = (attrs["property"] ?? attrs["name"] ?? "").toLowerCase();
984
+ const content = collapseWhitespace(attrs["content"] ?? "");
985
+ if (!key || !content || seen.has(key)) return;
986
+ seen.add(key);
987
+ const field = META_KEYS[key];
988
+ if (field && !meta[field]) meta[field] = capMetaValue(content);
989
+ };
990
+ feedHtml(html, {
991
+ handleStartTag(name: string, attrs: AttrDict) {
992
+ record(name, attrs);
993
+ },
994
+ handleStartEndTag(name: string, attrs: AttrDict) {
995
+ record(name, attrs);
996
+ },
997
+ handleEndTag() {},
998
+ handleData() {},
999
+ handleEntityRef() {},
1000
+ handleCharRef() {},
1001
+ });
1002
+ return meta;
1003
+ }
1004
+
1005
+ function pagePrefixedMarkdown(html: string): string {
1006
+ const meta = extractPageMeta(html);
1007
+ const lines: string[] = [];
1008
+ if (meta.title) lines.push(`Title: ${meta.title}`);
1009
+ if (meta.author) lines.push(`Author: ${meta.author}`);
1010
+ if (meta.date) lines.push(`Date: ${meta.date}`);
1011
+ if (meta.site) lines.push(`Site: ${meta.site}`);
1012
+ const markdown = htmlToMarkdown(html, true);
1013
+ const converted = lines.length ? `${lines.join("\n")}\n\n${markdown}` : markdown;
1014
+ if (html.endsWith(TRUNCATED_BODY_SUFFIX) && !converted.endsWith(TRUNCATED_BODY_SUFFIX)) {
1015
+ return converted + TRUNCATED_BODY_NOTICE;
1016
+ }
1017
+ return converted;
1018
+ }
1019
+
1020
+ function formatReadmeBody(body: string): string {
1021
+ if (looksLikeHtmlDocument(body)) {
1022
+ const converted = pagePrefixedMarkdown(body);
1023
+ if (converted.trim()) return converted;
1024
+ }
1025
+ return body;
1026
+ }
1027
+
740
1028
  export async function fetchPageText(
741
1029
  url: string,
742
1030
  options: FetchPageOptions = {},
743
1031
  ): Promise<string> {
744
- const timeoutMs = options.timeoutMs ?? 60_000;
1032
+ const timeoutMs = options.timeoutMs ?? DEFAULT_FETCH_TIMEOUT_MS;
745
1033
  const now = options.nowMs ?? Date.now;
746
1034
  const deadlineMs = options.deadlineMs ?? now() + timeoutMs;
747
1035
  const signal = options.signal;
@@ -752,48 +1040,49 @@ export async function fetchPageText(
752
1040
  url = normalizeUrlScheme(url);
753
1041
  const [allowed, reason] = checkUrlAccess(url, policy);
754
1042
  if (!allowed) return reason;
1043
+ const rawFetchOptions = {
1044
+ deadlineMs,
1045
+ signal,
1046
+ websitePolicy: policy,
1047
+ maxBytes: options.maxBytes,
1048
+ maxPdfBytes: options.maxPdfBytes,
1049
+ seams: options.seams,
1050
+ };
755
1051
 
756
1052
  const readmeApiUrl = githubRepoReadmeApiUrl(url);
757
1053
  if (readmeApiUrl) {
758
1054
  const readmeResult = await rawFetch(readmeApiUrl, {
759
- deadlineMs,
760
- signal,
761
- websitePolicy: policy,
762
- maxBytes: options.maxBytes,
763
- maxPdfBytes: options.maxPdfBytes,
764
- seams: options.seams,
1055
+ ...rawFetchOptions,
765
1056
  extraHeaders: {
766
1057
  Accept: "application/vnd.github.raw+json",
767
1058
  "X-GitHub-Api-Version": "2022-11-28",
768
1059
  },
769
1060
  });
770
- if (readmeResult.error === null && readmeResult.body.trim()) {
771
- let readmeBody = readmeResult.body;
772
- if (looksLikeHtmlDocument(readmeBody)) {
773
- const converted = htmlToMarkdown(readmeBody, true);
774
- if (converted.trim()) readmeBody = converted;
775
- }
776
- if (readmeBody.trim()) {
1061
+ const apiBody = readmeResult.error === null ? formatReadmeBody(readmeResult.body) : "";
1062
+ if (apiBody.trim()) {
1063
+ return truncatePageText(
1064
+ `README of ${url} (fetched via the GitHub README API):\n\n` + apiBody,
1065
+ maxChars,
1066
+ );
1067
+ }
1068
+ const rawReadmeUrl = githubRepoRawReadmeUrl(url);
1069
+ if (rawReadmeUrl) {
1070
+ const rawResult = await rawFetch(rawReadmeUrl, rawFetchOptions);
1071
+ const rawBody = rawResult.error === null ? formatReadmeBody(rawResult.body) : "";
1072
+ if (rawBody.trim()) {
777
1073
  return truncatePageText(
778
- `README of ${url} (fetched via the GitHub README API):\n\n` + readmeBody,
1074
+ `README of ${url} (fetched via the GitHub raw README URL):\n\n` + rawBody,
779
1075
  maxChars,
780
1076
  );
781
1077
  }
782
1078
  }
783
1079
  }
784
1080
 
785
- const result = await rawFetch(url, {
786
- deadlineMs,
787
- signal,
788
- websitePolicy: policy,
789
- maxBytes: options.maxBytes,
790
- maxPdfBytes: options.maxPdfBytes,
791
- seams: options.seams,
792
- });
1081
+ const result = await rawFetch(url, rawFetchOptions);
793
1082
  if (result.error !== null) return result.error;
794
1083
 
795
1084
  const isHtml = result.contentType.includes("html") || looksLikeHtml(result.body);
796
1085
  if (!isHtml) return truncatePageText(result.body.trim(), maxChars);
797
1086
 
798
- return truncatePageText(htmlToMarkdown(result.body, true), maxChars);
1087
+ return truncatePageText(pagePrefixedMarkdown(result.body), maxChars);
799
1088
  }