pi-unsloth-webtools 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/web-fetch.ts CHANGED
@@ -1,22 +1,30 @@
1
1
  import { lookup as dnsLookup } from "node:dns/promises";
2
2
  import http from "node:http";
3
3
  import https from "node:https";
4
+ import { createBrotliDecompress, createGunzip, createInflate, createInflateRaw } from "node:zlib";
4
5
  import type { IncomingMessage } from "node:http";
6
+ import type { Transform } from "node:stream";
5
7
  import {
6
8
  checkUrlAccess,
9
+ githubRepoRawReadmeUrl,
7
10
  githubRepoReadmeApiUrl,
8
11
  isPublicIp,
9
12
  normalizeUrlScheme,
10
13
  type WebsitePolicy,
11
14
  } from "./web-access.ts";
12
- import { htmlToMarkdown } from "./html-to-md.ts";
15
+ import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown } from "./html-to-md.ts";
16
+ import type { AttrDict } from "./html-to-md.ts";
13
17
  import { INVALID_CHARREFS } from "./entities.ts";
14
- import { extractPdfText, PdfParseError } from "./pdf.ts";
18
+ import { extractPdfText } from "./pdf.ts";
15
19
  import { randomUserAgent } from "./user-agents.ts";
16
20
 
17
21
  const MAX_FETCH_BYTES = 512 * 1024;
22
+ const MAX_DECOMPRESSED_BYTES = 64 * 1024 * 1024;
23
+ const META_VALUE_MAX_CHARS = 300;
18
24
  const MAX_PDF_FETCH_BYTES = 10 * 1024 * 1024;
19
25
  const MAX_REQUESTS = 5;
26
+ const MAX_SIGNAL_TIMEOUT_MS = 2 ** 31 - 1;
27
+ export const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
20
28
 
21
29
  const UTF32_LE_BOM = Buffer.from([0xff, 0xfe, 0x00, 0x00]);
22
30
  const UTF32_BE_BOM = Buffer.from([0x00, 0x00, 0xfe, 0xff]);
@@ -109,6 +117,28 @@ export class FetchTimeoutError extends Error {
109
117
  super("timed out");
110
118
  }
111
119
  }
120
+ const FETCH_CANCELLED_MESSAGE = "Failed to fetch URL: cancelled.";
121
+ const FETCH_TIMEOUT_MESSAGE = "Failed to fetch URL: timed out.";
122
+ const TRUNCATED_BODY_NOTICE = "\n\n... (page truncated at the download limit)";
123
+ const TRUNCATED_BODY_SUFFIX = "... (page truncated at the download limit)";
124
+
125
+ function fetchErrorMessage(err: unknown): string {
126
+ if (err instanceof FetchCancelledError) return FETCH_CANCELLED_MESSAGE;
127
+ if (err instanceof FetchTimeoutError) return FETCH_TIMEOUT_MESSAGE;
128
+ const message = err instanceof Error ? err.message : String(err);
129
+ if (message === "cancelled") return FETCH_CANCELLED_MESSAGE;
130
+ if (message === "timed out") return FETCH_TIMEOUT_MESSAGE;
131
+ return `Failed to fetch URL: ${message}`;
132
+ }
133
+
134
+ function statusErrorResult(status: number): RawFetchResult {
135
+ const reason = http.STATUS_CODES[status] ?? "";
136
+ return {
137
+ error: `Failed to fetch URL: HTTP ${status}${reason ? ` ${reason}` : ""}`,
138
+ body: "",
139
+ contentType: "",
140
+ };
141
+ }
112
142
 
113
143
  export interface FetchPageOptions {
114
144
  timeoutMs?: number;
@@ -127,6 +157,7 @@ export interface HopResponse {
127
157
  status: number;
128
158
  headers: Record<string, string | string[] | undefined>;
129
159
  body: Buffer;
160
+ truncated?: boolean;
130
161
  }
131
162
 
132
163
  export interface ResolvedHost {
@@ -172,20 +203,32 @@ export interface RawFetchResult {
172
203
  contentType: string;
173
204
  }
174
205
 
206
+ function htmlProbe(body: string, re: RegExp): boolean {
207
+ let probe = body;
208
+ while (true) {
209
+ probe = probe.replace(/^[ \t\n\r\f\v]+/, "");
210
+ const stripped = probe.replace(/^(?:<!--[\s\S]*?-->|<\?[\s\S]*?\?>)/, "");
211
+ if (stripped === probe) break;
212
+ probe = stripped;
213
+ }
214
+ return re.test(probe.slice(0, 256).toLowerCase());
215
+ }
216
+
175
217
  export function looksLikeHtml(body: string): boolean {
176
- const probe = body.replace(/^[ \t\n\r\f\v]+/, "").slice(0, 256).toLowerCase();
177
- return HTML_LEADING_RE.test(probe);
218
+ return htmlProbe(body, HTML_LEADING_RE);
178
219
  }
179
220
 
180
221
  export function looksLikeHtmlDocument(body: string): boolean {
181
- const probe = body.replace(/^[ \t\n\r\f\v]+/, "").slice(0, 256).toLowerCase();
182
- return HTML_DOCUMENT_RE.test(probe);
222
+ return htmlProbe(body, HTML_DOCUMENT_RE);
223
+ }
224
+
225
+ function parseContentType(value: string | null | undefined): string {
226
+ return /^[\w.+-]+\/[\w.+-]+/.exec(value ?? "")?.[0] ?? "";
183
227
  }
184
228
 
185
229
  export function isTextCandidateContentType(contentType: string | null): boolean {
186
- const match = /^[\w.+-]+\/[\w.+-]+/.exec(contentType ?? "");
187
- if (!match) return true;
188
- const ct = match[0].toLowerCase();
230
+ const ct = parseContentType(contentType).toLowerCase();
231
+ if (!ct) return true;
189
232
  if (ct.startsWith("text/")) return true;
190
233
  if (ct.startsWith("application/")) {
191
234
  const subtype = ct.slice("application/".length);
@@ -321,7 +364,12 @@ function sniffMetaCharsetForHtml(bytes: Buffer, contentType: string): string | n
321
364
 
322
365
  function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
323
366
  let out = "";
324
- for (let i = 0; i + 3 < bytes.length; i += 4) {
367
+ let i = 0;
368
+ if (bytes.length >= 4) {
369
+ const first = littleEndian ? bytes.readUInt32LE(0) : bytes.readUInt32BE(0);
370
+ if (first === 0xfeff) i = 4;
371
+ }
372
+ for (; i + 3 < bytes.length; i += 4) {
325
373
  const v = littleEndian ? bytes.readUInt32LE(i) : bytes.readUInt32BE(i);
326
374
  if (v === 0 || v > 0x10ffff || (v >= 0xd800 && v <= 0xdfff)) {
327
375
  out += "\ufffd";
@@ -334,7 +382,8 @@ function decodeUtf32(bytes: Buffer, littleEndian: boolean): string {
334
382
 
335
383
  function decodeUtf16Be(bytes: Buffer): string {
336
384
  let out = "";
337
- for (let i = 0; i + 1 < bytes.length; i += 2) {
385
+ let i = bytes.length >= 2 && bytes.readUInt16BE(0) === 0xfeff ? 2 : 0;
386
+ for (; i + 1 < bytes.length; i += 2) {
338
387
  out += String.fromCharCode(bytes.readUInt16BE(i));
339
388
  }
340
389
  return out;
@@ -406,10 +455,32 @@ function decodeTis620(bytes: Buffer): string {
406
455
  }
407
456
 
408
457
 
458
+ function withAbort<T>(promise: Promise<T>, signal?: AbortSignal): Promise<T> {
459
+ if (!signal) return promise;
460
+ return new Promise<T>((resolve, reject) => {
461
+ const onAbort = () => reject(new DOMException("aborted", "AbortError"));
462
+ if (signal.aborted) {
463
+ onAbort();
464
+ return;
465
+ }
466
+ signal.addEventListener("abort", onAbort, { once: true });
467
+ promise.then(
468
+ (value) => {
469
+ signal.removeEventListener("abort", onAbort);
470
+ resolve(value);
471
+ },
472
+ (err) => {
473
+ signal.removeEventListener("abort", onAbort);
474
+ reject(err);
475
+ },
476
+ );
477
+ });
478
+ }
479
+
409
480
  async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promise<ResolvedHost> {
410
481
  let addresses: { address: string; family: number }[];
411
482
  try {
412
- addresses = await dnsLookup(hostname, { all: true, verbatim: true });
483
+ addresses = await withAbort(dnsLookup(hostname, { all: true, verbatim: true }), signal);
413
484
  } catch (err) {
414
485
  return { ok: false, reason: `Failed to resolve host: ${err}`, ip: "", family: 0 };
415
486
  }
@@ -426,17 +497,32 @@ async function resolveAndValidate(hostname: string, signal?: AbortSignal): Promi
426
497
  }
427
498
 
428
499
 
429
- function fetchBudgetExceeded(
500
+ function budgetExceededResult(
430
501
  deadline: number | null,
431
502
  signal: AbortSignal | undefined,
432
503
  now: () => number = Date.now,
433
- ): string | null {
434
- if (signal?.aborted) return "Failed to fetch URL: cancelled.";
435
- if (deadline !== null && now() >= deadline) return "Failed to fetch URL: timed out.";
504
+ contentType = "",
505
+ ): RawFetchResult | null {
506
+ if (signal?.aborted) return { error: FETCH_CANCELLED_MESSAGE, body: "", contentType };
507
+ if (deadline !== null && now() >= deadline) return { error: FETCH_TIMEOUT_MESSAGE, body: "", contentType };
436
508
  return null;
437
509
  }
438
510
 
439
511
 
512
+ function contentEncodingCodec(value: string | string[] | undefined): string | null {
513
+ const declared = Array.isArray(value) ? value[0] : value;
514
+ const codec = (declared ?? "").split(",", 1)[0].trim().toLowerCase();
515
+ if (codec === "gzip" || codec === "x-gzip" || codec === "deflate" || codec === "br") return codec;
516
+ return null;
517
+ }
518
+
519
+ function createDecodeStream(codec: string): Transform {
520
+ const options = { maxOutputLength: MAX_DECOMPRESSED_BYTES };
521
+ if (codec === "br") return createBrotliDecompress(options);
522
+ if (codec === "deflate") return createInflate(options);
523
+ return createGunzip(options);
524
+ }
525
+
440
526
  export function requestHop(opts: HopOptions): Promise<HopResponse> {
441
527
  return new Promise((resolve, reject) => {
442
528
  const url = opts.url;
@@ -460,11 +546,18 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
460
546
  action();
461
547
  };
462
548
  const request = transport.request(options, (res: IncomingMessage) => {
549
+ const codec = contentEncodingCodec(res.headers["content-encoding"]);
463
550
  const chunks: Buffer[] = [];
464
551
  let total = 0;
552
+ let head = Buffer.alloc(0);
553
+ let truncated = false;
554
+ let decoder: Transform | null = null;
465
555
  const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
466
556
  let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
467
557
  let extendedForPdf = false;
558
+ const declaredLengthHeader = res.headers["content-length"];
559
+ const declaredLength =
560
+ declaredLengthHeader === undefined ? NaN : Number(declaredLengthHeader);
468
561
  const finish = (err: string | null, body: Buffer) => {
469
562
  settle(() => {
470
563
  if (err) {
@@ -476,40 +569,112 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
476
569
  status: res.statusCode ?? 0,
477
570
  headers: res.headers as Record<string, string | string[] | undefined>,
478
571
  body,
572
+ truncated,
479
573
  });
480
574
  }
481
575
  });
482
576
  };
483
- res.on("data", (chunk: Buffer) => {
577
+ const timeoutAndDestroy = () => {
578
+ decoder?.destroy();
579
+ res.destroy();
580
+ finish("timed out", Buffer.concat(chunks));
581
+ };
582
+ const acceptData = (chunk: Buffer) => {
484
583
  if (settled) return;
485
584
  const now = opts.nowMs ?? Date.now;
486
585
  if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
487
- res.destroy();
488
- finish("timed out", Buffer.concat(chunks));
586
+ timeoutAndDestroy();
489
587
  return;
490
588
  }
491
- if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
492
- if (hasPdfMagic(Buffer.concat(chunks))) {
589
+ if (head.length < 1024) {
590
+ const need = 1024 - head.length;
591
+ head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
592
+ }
593
+ if (!declaredPdf && !extendedForPdf && total + chunk.length > limit) {
594
+ if (hasPdfMagic(head)) {
493
595
  limit = opts.maxPdfBytes;
494
596
  extendedForPdf = true;
495
597
  }
496
598
  }
497
599
  const space = limit - total;
498
- if (space <= 0) {
499
- res.destroy();
500
- finish(null, Buffer.concat(chunks));
501
- return;
502
- }
503
600
  const take = chunk.subarray(0, Math.min(chunk.length, space));
504
601
  chunks.push(take);
505
602
  total += take.length;
506
603
  if (total >= limit) {
604
+ truncated =
605
+ codec !== null ||
606
+ !(Number.isFinite(declaredLength) && declaredLength === total);
607
+ decoder?.destroy();
507
608
  res.destroy();
508
609
  finish(null, Buffer.concat(chunks));
509
610
  }
611
+ };
612
+ if (codec === null) {
613
+ res.on("data", (chunk: Buffer) => {
614
+ acceptData(chunk);
615
+ });
616
+ res.on("end", () => finish(null, Buffer.concat(chunks)));
617
+ } else {
618
+ let rawBuffer: Buffer[] = [];
619
+ let rawBufferBytes = 0;
620
+ let outputStarted = false;
621
+ let rawFallbackTried = false;
622
+ let resEnded = false;
623
+ const wire = (d: Transform) => {
624
+ d.on("data", (chunk: Buffer) => {
625
+ outputStarted = true;
626
+ acceptData(chunk);
627
+ });
628
+ d.on("end", () => finish(null, Buffer.concat(chunks)));
629
+ d.on("drain", () => res.resume());
630
+ d.on("error", (err: NodeJS.ErrnoException) => {
631
+ if (settled) return;
632
+ if (
633
+ codec === "deflate" &&
634
+ !rawFallbackTried &&
635
+ !outputStarted &&
636
+ err.code === "Z_DATA_ERROR"
637
+ ) {
638
+ rawFallbackTried = true;
639
+ d.destroy();
640
+ decoder = createInflateRaw({ maxOutputLength: MAX_DECOMPRESSED_BYTES });
641
+ wire(decoder);
642
+ for (const buffered of rawBuffer) decoder.write(buffered);
643
+ if (resEnded) decoder.end();
644
+ return;
645
+ }
646
+ if (outputStarted) {
647
+ truncated = true;
648
+ finish(null, Buffer.concat(chunks));
649
+ return;
650
+ }
651
+ finish(null, Buffer.concat(rawBuffer));
652
+ });
653
+ };
654
+ decoder = createDecodeStream(codec);
655
+ wire(decoder);
656
+ res.on("data", (chunk: Buffer) => {
657
+ if (settled) return;
658
+ const now = opts.nowMs ?? Date.now;
659
+ if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
660
+ timeoutAndDestroy();
661
+ return;
662
+ }
663
+ if (!outputStarted && !rawFallbackTried && rawBufferBytes < limit) {
664
+ rawBuffer.push(chunk);
665
+ rawBufferBytes += chunk.length;
666
+ }
667
+ if (!decoder!.write(chunk)) res.pause();
668
+ });
669
+ res.on("end", () => {
670
+ resEnded = true;
671
+ if (!settled) decoder!.end();
672
+ });
673
+ }
674
+ res.on("error", (err) => {
675
+ decoder?.destroy();
676
+ finish(err.message, Buffer.concat(chunks));
510
677
  });
511
- res.on("end", () => finish(null, Buffer.concat(chunks)));
512
- res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
513
678
  });
514
679
  const onAbort = () => request.destroy(new FetchCancelledError());
515
680
  opts.signal?.addEventListener("abort", onAbort, { once: true });
@@ -535,14 +700,23 @@ export async function fetchUrlRaw(
535
700
  const seams = options.seams ?? {};
536
701
  const resolveHost = seams.resolve ?? resolveAndValidate;
537
702
  const performRequest = seams.request ?? requestHop;
703
+ const resolveWithBudget = async (hostname: string): Promise<ResolvedHost> => {
704
+ const deadlineSignal = AbortSignal.timeout(Math.min(MAX_SIGNAL_TIMEOUT_MS, Math.max(1, deadline - now())));
705
+ const resolveSignal = signal ? AbortSignal.any([signal, deadlineSignal]) : deadlineSignal;
706
+ const resolved = await resolveHost(hostname, resolveSignal);
707
+ if (resolved.ok) return resolved;
708
+ if (signal?.aborted) return { ...resolved, reason: FETCH_CANCELLED_MESSAGE };
709
+ if (resolveSignal.aborted) return { ...resolved, reason: FETCH_TIMEOUT_MESSAGE };
710
+ return resolved;
711
+ };
538
712
 
539
713
  url = normalizeUrlScheme(url);
540
714
  const [allowed, reason, hostname] = checkUrlAccess(url, policy);
541
715
  if (!allowed) return { error: reason, body: "", contentType: "" };
542
716
 
543
- let budgetError = fetchBudgetExceeded(deadline, signal, now);
544
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
545
- let resolved = await resolveHost(hostname, signal);
717
+ const budgetResult = budgetExceededResult(deadline, signal, now);
718
+ if (budgetResult !== null) return budgetResult;
719
+ let resolved = await resolveWithBudget(hostname);
546
720
  if (!resolved.ok) return { error: resolved.reason, body: "", contentType: "" };
547
721
 
548
722
  let currentUrl = url;
@@ -551,8 +725,8 @@ export async function fetchUrlRaw(
551
725
  const userAgent = randomUserAgent();
552
726
 
553
727
  for (let hop = 0; hop < MAX_REQUESTS; hop++) {
554
- budgetError = fetchBudgetExceeded(deadline, signal, now);
555
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
728
+ const budgetResult = budgetExceededResult(deadline, signal, now);
729
+ if (budgetResult !== null) return budgetResult;
556
730
  const parsed = new URL(currentUrl);
557
731
  const hostHeader = parsed.hostname + (parsed.port ? `:${parsed.port}` : "");
558
732
  const headers: Record<string, string> = {
@@ -578,26 +752,12 @@ export async function fetchUrlRaw(
578
752
  signal,
579
753
  });
580
754
  } catch (err) {
581
- if (err instanceof FetchCancelledError)
582
- return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
583
- if (err instanceof FetchTimeoutError)
584
- return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
585
- const message = err instanceof Error ? err.message : String(err);
586
- if (message === "cancelled")
587
- return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
588
- if (message === "timed out")
589
- return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
590
- return { error: `Failed to fetch URL: ${message}`, body: "", contentType: "" };
755
+ return { error: fetchErrorMessage(err), body: "", contentType: "" };
591
756
  }
592
757
 
593
758
  if (response.status >= 300 && response.status < 400) {
594
759
  if (![301, 302, 303, 307, 308].includes(response.status)) {
595
- const reason = http.STATUS_CODES[response.status] ?? "";
596
- return {
597
- error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
598
- body: "",
599
- contentType: "",
600
- };
760
+ return statusErrorResult(response.status);
601
761
  }
602
762
  const rawLocation = response.headers.location;
603
763
  const location = Array.isArray(rawLocation) ? rawLocation[0] : rawLocation;
@@ -618,20 +778,22 @@ export async function fetchUrlRaw(
618
778
  policy,
619
779
  );
620
780
  if (!redirectAllowed) return { error: redirectReason, body: "", contentType: "" };
621
- const redirected = await resolveHost(redirectHost, signal);
781
+ const redirected = await resolveWithBudget(redirectHost);
622
782
  if (!redirected.ok) return { error: redirected.reason, body: "", contentType: "" };
623
783
  pinnedIp = redirected.ip;
624
784
  pinnedFamily = redirected.family;
625
785
  continue;
626
786
  }
627
787
 
628
- budgetError = fetchBudgetExceeded(deadline, signal, now);
629
- if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
788
+ const postBudgetResult = budgetExceededResult(deadline, signal, now);
789
+ if (postBudgetResult !== null) return postBudgetResult;
790
+
791
+ if (response.status >= 400) {
792
+ return statusErrorResult(response.status);
793
+ }
630
794
 
631
795
  const contentTypeHeader = response.headers["content-type"];
632
- const contentType = contentTypeHeader
633
- ? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
634
- : "";
796
+ const contentType = parseContentType(contentTypeHeader ? String(contentTypeHeader).toLowerCase() : null);
635
797
  const declaredCharset = contentTypeHeader
636
798
  ? (/charset=([^;\s]+)/i.exec(String(contentTypeHeader))?.[1] ?? null)
637
799
  : null;
@@ -650,16 +812,23 @@ export async function fetchUrlRaw(
650
812
  try {
651
813
  pdfText = await extractPdfText(response.body);
652
814
  } catch {
653
- return { error: "(PDF content could not be read as text)", body: "", contentType };
815
+ return {
816
+ error: response.truncated
817
+ ? "(PDF content could not be read as text; the download was truncated at the download limit)"
818
+ : "(PDF content could not be read as text)",
819
+ body: "",
820
+ contentType,
821
+ };
654
822
  }
655
- budgetError = fetchBudgetExceeded(deadline, signal, now);
656
- if (budgetError !== null) return { error: budgetError, body: "", contentType };
823
+ const budgetResult = budgetExceededResult(deadline, signal, now, contentType);
824
+ if (budgetResult !== null) return budgetResult;
657
825
  if (!pdfText) pdfText = "(PDF contains no extractable text)";
826
+ if (response.truncated) pdfText += TRUNCATED_BODY_NOTICE;
658
827
  return { error: null, body: pdfText, contentType: "application/pdf" };
659
828
  }
660
829
 
661
830
  if (!isTextCandidateContentType(contentType)) {
662
- const safeType = /^[\w.+-]+\/[\w.+-]+/.exec(contentType ?? "")?.[0] ?? "unknown type";
831
+ const safeType = parseContentType(contentType) || "unknown type";
663
832
  return {
664
833
  error: `(non-text content: ${safeType}, ${response.body.length} bytes; not readable as text)`,
665
834
  body: "",
@@ -677,10 +846,11 @@ export async function fetchUrlRaw(
677
846
 
678
847
  const declaredCodec = declaredCharset ? normalizeCharset(declaredCharset) : null;
679
848
  const bomCodec = bomCodecFor(response.body);
680
- const rawHtml = decodeWithCodec(
681
- response.body,
682
- declaredCodec ?? bomCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
683
- );
849
+ const rawHtml =
850
+ decodeWithCodec(
851
+ response.body,
852
+ bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
853
+ ) + (response.truncated ? TRUNCATED_BODY_NOTICE : "");
684
854
 
685
855
  if (looksBinary(rawHtml)) {
686
856
  let alt: string | null = null;
@@ -692,6 +862,7 @@ export async function fetchUrlRaw(
692
862
  if (!looksBinary(candidate)) alt = candidate;
693
863
  }
694
864
  if (alt !== null) {
865
+ if (response.truncated) alt += TRUNCATED_BODY_NOTICE;
695
866
  return { error: null, body: alt, contentType };
696
867
  }
697
868
  return {
@@ -718,16 +889,147 @@ function bomCodecFor(bytes: Buffer): string | null {
718
889
  export function truncatePageText(text: string, maxChars?: number): string {
719
890
  if (!text) return "(page returned no readable text)";
720
891
  if (typeof maxChars === "number" && maxChars > 0 && text.length > maxChars) {
721
- return text.slice(0, maxChars) + `\n\n... (truncated, ${text.length} chars total)`;
892
+ const hadCapNotice = text.endsWith(TRUNCATED_BODY_SUFFIX);
893
+ const core = (hadCapNotice ? text.slice(0, -TRUNCATED_BODY_SUFFIX.length) : text).trimEnd();
894
+ return (
895
+ cutAtCharBoundary(core, maxChars) +
896
+ `\n\n... (truncated, ${text.length} chars total)` +
897
+ (hadCapNotice ? TRUNCATED_BODY_NOTICE : "")
898
+ );
722
899
  }
723
900
  return text;
724
901
  }
725
902
 
903
+ const NON_DOCUMENT_TITLE_TAGS = new Set(["math", "noscript", "svg", "template"]);
904
+
905
+ function extractPageTitle(html: string): string {
906
+ let inTitle = false;
907
+ let done = false;
908
+ let skipDepth = 0;
909
+ const parts: string[] = [];
910
+ feedHtml(html, {
911
+ handleStartTag(name: string) {
912
+ if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
913
+ skipDepth++;
914
+ return;
915
+ }
916
+ if (skipDepth) return;
917
+ if (name === "title" && !done) {
918
+ inTitle = true;
919
+ } else if (inTitle) {
920
+ inTitle = false;
921
+ }
922
+ },
923
+ handleStartEndTag() {},
924
+ handleEndTag(name: string) {
925
+ if (NON_DOCUMENT_TITLE_TAGS.has(name)) {
926
+ skipDepth = Math.max(0, skipDepth - 1);
927
+ return;
928
+ }
929
+ if (skipDepth) return;
930
+ if (name === "title") {
931
+ inTitle = false;
932
+ done = true;
933
+ }
934
+ },
935
+ handleData(text: string) {
936
+ if (inTitle) parts.push(text);
937
+ },
938
+ handleEntityRef(name: string) {
939
+ if (inTitle) parts.push(decodeHtmlEntities(`&${name};`));
940
+ },
941
+ handleCharRef(name: string) {
942
+ if (inTitle) parts.push(decodeHtmlEntities(`&#${name};`));
943
+ },
944
+ });
945
+ return collapseWhitespace(parts.join(""));
946
+ }
947
+
948
+ interface PageMeta {
949
+ title: string;
950
+ author: string;
951
+ date: string;
952
+ site: string;
953
+ }
954
+
955
+ const META_KEYS: Record<string, keyof PageMeta> = {
956
+ author: "author",
957
+ "article:author": "author",
958
+ "dc.creator": "author",
959
+ "article:published_time": "date",
960
+ date: "date",
961
+ "dc.date": "date",
962
+ datepublished: "date",
963
+ "og:site_name": "site",
964
+ "application-name": "site",
965
+ };
966
+
967
+ function cutAtCharBoundary(text: string, maxChars: number): string {
968
+ const sliced = text.slice(0, maxChars);
969
+ const last = sliced.charCodeAt(sliced.length - 1);
970
+ return last >= 0xd800 && last <= 0xdbff ? sliced.slice(0, -1) : sliced;
971
+ }
972
+
973
+ function capMetaValue(value: string): string {
974
+ if (value.length <= META_VALUE_MAX_CHARS) return value;
975
+ return cutAtCharBoundary(value, META_VALUE_MAX_CHARS);
976
+ }
977
+
978
+ function extractPageMeta(html: string): PageMeta {
979
+ const meta: PageMeta = { title: extractPageTitle(html), author: "", date: "", site: "" };
980
+ const seen = new Set<string>();
981
+ const record = (name: string, attrs: AttrDict) => {
982
+ if (name !== "meta") return;
983
+ const key = (attrs["property"] ?? attrs["name"] ?? "").toLowerCase();
984
+ const content = collapseWhitespace(attrs["content"] ?? "");
985
+ if (!key || !content || seen.has(key)) return;
986
+ seen.add(key);
987
+ const field = META_KEYS[key];
988
+ if (field && !meta[field]) meta[field] = capMetaValue(content);
989
+ };
990
+ feedHtml(html, {
991
+ handleStartTag(name: string, attrs: AttrDict) {
992
+ record(name, attrs);
993
+ },
994
+ handleStartEndTag(name: string, attrs: AttrDict) {
995
+ record(name, attrs);
996
+ },
997
+ handleEndTag() {},
998
+ handleData() {},
999
+ handleEntityRef() {},
1000
+ handleCharRef() {},
1001
+ });
1002
+ return meta;
1003
+ }
1004
+
1005
+ function pagePrefixedMarkdown(html: string): string {
1006
+ const meta = extractPageMeta(html);
1007
+ const lines: string[] = [];
1008
+ if (meta.title) lines.push(`Title: ${meta.title}`);
1009
+ if (meta.author) lines.push(`Author: ${meta.author}`);
1010
+ if (meta.date) lines.push(`Date: ${meta.date}`);
1011
+ if (meta.site) lines.push(`Site: ${meta.site}`);
1012
+ const markdown = htmlToMarkdown(html, true);
1013
+ const converted = lines.length ? `${lines.join("\n")}\n\n${markdown}` : markdown;
1014
+ if (html.endsWith(TRUNCATED_BODY_SUFFIX) && !converted.endsWith(TRUNCATED_BODY_SUFFIX)) {
1015
+ return converted + TRUNCATED_BODY_NOTICE;
1016
+ }
1017
+ return converted;
1018
+ }
1019
+
1020
+ function formatReadmeBody(body: string): string {
1021
+ if (looksLikeHtmlDocument(body)) {
1022
+ const converted = pagePrefixedMarkdown(body);
1023
+ if (converted.trim()) return converted;
1024
+ }
1025
+ return body;
1026
+ }
1027
+
726
1028
  export async function fetchPageText(
727
1029
  url: string,
728
1030
  options: FetchPageOptions = {},
729
1031
  ): Promise<string> {
730
- const timeoutMs = options.timeoutMs ?? 60_000;
1032
+ const timeoutMs = options.timeoutMs ?? DEFAULT_FETCH_TIMEOUT_MS;
731
1033
  const now = options.nowMs ?? Date.now;
732
1034
  const deadlineMs = options.deadlineMs ?? now() + timeoutMs;
733
1035
  const signal = options.signal;
@@ -738,48 +1040,49 @@ export async function fetchPageText(
738
1040
  url = normalizeUrlScheme(url);
739
1041
  const [allowed, reason] = checkUrlAccess(url, policy);
740
1042
  if (!allowed) return reason;
1043
+ const rawFetchOptions = {
1044
+ deadlineMs,
1045
+ signal,
1046
+ websitePolicy: policy,
1047
+ maxBytes: options.maxBytes,
1048
+ maxPdfBytes: options.maxPdfBytes,
1049
+ seams: options.seams,
1050
+ };
741
1051
 
742
1052
  const readmeApiUrl = githubRepoReadmeApiUrl(url);
743
1053
  if (readmeApiUrl) {
744
1054
  const readmeResult = await rawFetch(readmeApiUrl, {
745
- deadlineMs,
746
- signal,
747
- websitePolicy: policy,
748
- maxBytes: options.maxBytes,
749
- maxPdfBytes: options.maxPdfBytes,
750
- seams: options.seams,
1055
+ ...rawFetchOptions,
751
1056
  extraHeaders: {
752
1057
  Accept: "application/vnd.github.raw+json",
753
1058
  "X-GitHub-Api-Version": "2022-11-28",
754
1059
  },
755
1060
  });
756
- if (readmeResult.error === null && readmeResult.body.trim()) {
757
- let readmeBody = readmeResult.body;
758
- if (looksLikeHtmlDocument(readmeBody)) {
759
- const converted = htmlToMarkdown(readmeBody, true);
760
- if (converted.trim()) readmeBody = converted;
761
- }
762
- if (readmeBody.trim()) {
1061
+ const apiBody = readmeResult.error === null ? formatReadmeBody(readmeResult.body) : "";
1062
+ if (apiBody.trim()) {
1063
+ return truncatePageText(
1064
+ `README of ${url} (fetched via the GitHub README API):\n\n` + apiBody,
1065
+ maxChars,
1066
+ );
1067
+ }
1068
+ const rawReadmeUrl = githubRepoRawReadmeUrl(url);
1069
+ if (rawReadmeUrl) {
1070
+ const rawResult = await rawFetch(rawReadmeUrl, rawFetchOptions);
1071
+ const rawBody = rawResult.error === null ? formatReadmeBody(rawResult.body) : "";
1072
+ if (rawBody.trim()) {
763
1073
  return truncatePageText(
764
- `README of ${url} (fetched via the GitHub README API):\n\n` + readmeBody,
1074
+ `README of ${url} (fetched via the GitHub raw README URL):\n\n` + rawBody,
765
1075
  maxChars,
766
1076
  );
767
1077
  }
768
1078
  }
769
1079
  }
770
1080
 
771
- const result = await rawFetch(url, {
772
- deadlineMs,
773
- signal,
774
- websitePolicy: policy,
775
- maxBytes: options.maxBytes,
776
- maxPdfBytes: options.maxPdfBytes,
777
- seams: options.seams,
778
- });
1081
+ const result = await rawFetch(url, rawFetchOptions);
779
1082
  if (result.error !== null) return result.error;
780
1083
 
781
1084
  const isHtml = result.contentType.includes("html") || looksLikeHtml(result.body);
782
1085
  if (!isHtml) return truncatePageText(result.body.trim(), maxChars);
783
1086
 
784
- return truncatePageText(htmlToMarkdown(result.body, true), maxChars);
1087
+ return truncatePageText(pagePrefixedMarkdown(result.body), maxChars);
785
1088
  }