@a-t-h-i/bot-lobby 0.6.3 → 0.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/README.md +102 -11
  2. package/package.json +1 -4
  3. package/prompts/master.md +47 -1
  4. package/prompts/researcher.md +8 -2
  5. package/prompts/reviewer.md +42 -0
  6. package/prompts/worker.md +7 -0
  7. package/src/ask/dialog.ts +167 -0
  8. package/src/ask/image.ts +202 -0
  9. package/src/ask/png.ts +179 -0
  10. package/src/ask/relay.ts +89 -0
  11. package/src/ask/state.ts +160 -0
  12. package/src/ask/tool.ts +126 -0
  13. package/src/ask/types.ts +55 -0
  14. package/src/ask/view.ts +159 -0
  15. package/src/execution/agent-runner.ts +103 -11
  16. package/src/execution/git.ts +111 -14
  17. package/src/execution/pi-runner.ts +164 -11
  18. package/src/index.ts +6 -0
  19. package/src/lobby/ask.ts +10 -120
  20. package/src/lobby/layout.ts +27 -8
  21. package/src/lobby/markdown.ts +36 -6
  22. package/src/lobby/planner.ts +1 -1
  23. package/src/lobby/quickfix.ts +21 -0
  24. package/src/lobby/runtime.ts +14 -36
  25. package/src/lobby/tabs/home.ts +112 -48
  26. package/src/lobby/tabs/issues.ts +4 -3
  27. package/src/lobby/tabs/plan.ts +4 -4
  28. package/src/lobby/tabs/quickfix.ts +7 -1
  29. package/src/lobby/tabs/tasks.ts +14 -4
  30. package/src/lobby/theme.ts +30 -0
  31. package/src/lobby/view.ts +27 -5
  32. package/src/master/decisions.ts +1 -1
  33. package/src/master/master.ts +41 -4
  34. package/src/master/research.ts +5 -2
  35. package/src/pi/commands.ts +94 -10
  36. package/src/pi/events.ts +47 -10
  37. package/src/pi/quiet.ts +22 -4
  38. package/src/pi/start-task.ts +8 -2
  39. package/src/pi/tools.ts +26 -8
  40. package/src/pi/ui.ts +6 -1
  41. package/src/pi/zen-metrics.ts +13 -3
  42. package/src/pi/zen.ts +16 -9
  43. package/src/roles/reviewer.ts +23 -4
  44. package/src/roles/worker.ts +18 -0
  45. package/src/schemas/configuration.ts +4 -0
  46. package/src/schemas/findings.ts +14 -0
  47. package/src/schemas/task.ts +21 -0
  48. package/src/state/budget.ts +274 -0
  49. package/src/state/changes.ts +231 -0
  50. package/src/text.ts +28 -2
  51. package/src/web/extract.ts +332 -0
  52. package/src/web/fetch.ts +232 -0
  53. package/src/web/html.ts +183 -0
  54. package/src/web/read.ts +113 -0
  55. package/src/web/search.ts +202 -0
  56. package/src/web/tools.ts +279 -0
  57. package/src/workflow/workflow.ts +592 -42
@@ -0,0 +1,332 @@
1
+ /**
2
+ * A web page as a model reads it best: its title and dates, then the main
3
+ * content as Markdown (headings, paragraphs, lists, tables, code, links made
4
+ * absolute), with navigation, scripts, forms, cookie banners and the like
5
+ * left out.
6
+ */
7
+ import { findAll, findFirst, parseHtml, textOf, type HtmlElement, type HtmlNode } from "./html.ts";
8
+
9
+ export interface PageMeta {
10
+ title?: string;
11
+ description?: string;
12
+ siteName?: string;
13
+ published?: string;
14
+ modified?: string;
15
+ lang?: string;
16
+ }
17
+
18
+ export interface ReadablePage extends PageMeta {
19
+ markdown: string;
20
+ }
21
+
22
+ /** Elements never worth reading. */
23
+ const DROP = new Set(["script", "style", "noscript", "template", "svg", "canvas", "iframe", "object", "embed", "video", "audio", "map", "head", "link", "meta", "button", "input", "select", "textarea", "form", "dialog", "nav"]);
24
+ /** Page chrome, dropped unless it is all there is. */
25
+ const CHROME = new Set(["aside", "footer"]);
26
+ /** Class or id words that mark page chrome rather than content. */
27
+ const CHROME_HINT = /(?:^|[\s_-])(cookie|consent|gdpr|banner|newsletter|subscribe|signup|sidebar|share|sharing|social|advert|ads?|promo|popup|modal|breadcrumbs?|related|comments?|skip-link|toc-mobile)(?:$|[\s_-])/i;
28
+ const BLOCK = new Set(["address", "article", "aside", "blockquote", "body", "center", "dd", "details", "dialog", "div", "dl", "dt", "fieldset", "figcaption", "figure", "footer", "h1", "h2", "h3", "h4", "h5", "h6", "header", "hgroup", "hr", "html", "li", "main", "ol", "p", "pre", "section", "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", "#root"]);
29
+
30
+ function isChrome(element: HtmlElement): boolean {
31
+ if (element.attrs.hidden !== undefined || element.attrs["aria-hidden"] === "true") return true;
32
+ if (element.attrs.role === "navigation" || element.attrs.role === "banner" || element.attrs.role === "contentinfo" || element.attrs.role === "dialog") return true;
33
+ const style = element.attrs.style ?? "";
34
+ if (/display\s*:\s*none|visibility\s*:\s*hidden/i.test(style)) return true;
35
+ if (!["div", "section", "aside", "ul", "span", "p", "header", "footer"].includes(element.name)) return false;
36
+ return CHROME_HINT.test(`${element.attrs.class ?? ""} ${element.attrs.id ?? ""}`);
37
+ }
38
+
39
+ function metaContent(root: HtmlElement, ...keys: string[]): string | undefined {
40
+ const metas = findAll(root, (element) => element.name === "meta");
41
+ for (const key of keys) {
42
+ const meta = metas.find((element) => (element.attrs.property ?? element.attrs.name ?? element.attrs.itemprop ?? "").toLowerCase() === key);
43
+ const content = meta?.attrs.content?.trim();
44
+ if (content) return content;
45
+ }
46
+ return undefined;
47
+ }
48
+
49
+ /** A date as it is written in structured data (JSON-LD), when the page has one. */
50
+ function jsonLdDate(root: HtmlElement, key: "datePublished" | "dateModified"): string | undefined {
51
+ for (const script of findAll(root, (element) => element.name === "script" && (element.attrs.type ?? "").includes("ld+json"))) {
52
+ const source = script.children.map((child) => (child.type === "text" ? child.text : "")).join("");
53
+ const match = new RegExp(`"${key}"\\s*:\\s*"([^"]+)"`).exec(source);
54
+ if (match) return match[1];
55
+ }
56
+ return undefined;
57
+ }
58
+
59
+ /** A date trimmed to what a citation needs: the day (ISO), or the text as the page gives it. */
60
+ function day(value: string | undefined): string | undefined {
61
+ if (!value) return undefined;
62
+ const iso = /^(\d{4}-\d{2}-\d{2})/.exec(value.trim());
63
+ if (iso) return iso[1];
64
+ const parsed = Date.parse(value);
65
+ return Number.isNaN(parsed) ? value.trim().slice(0, 40) : new Date(parsed).toISOString().slice(0, 10);
66
+ }
67
+
68
+ export function pageMeta(root: HtmlElement): PageMeta {
69
+ const titleElement = findFirst(root, (element) => element.name === "title");
70
+ const title = metaContent(root, "og:title", "twitter:title") ?? (titleElement ? textOf(titleElement) : undefined);
71
+ const time = findFirst(root, (element) => element.name === "time" && Boolean(element.attrs.datetime));
72
+ const html = findFirst(root, (element) => element.name === "html");
73
+ const meta: PageMeta = {
74
+ title: title?.replace(/\s+/g, " ").trim() || undefined,
75
+ description: metaContent(root, "description", "og:description", "twitter:description"),
76
+ siteName: metaContent(root, "og:site_name", "application-name"),
77
+ published: day(metaContent(root, "article:published_time", "datepublished", "date", "dc.date", "dc.date.issued", "dcterms.created", "pubdate", "citation_publication_date", "citation_date", "sailthru.date") ?? jsonLdDate(root, "datePublished") ?? time?.attrs.datetime),
78
+ modified: day(metaContent(root, "article:modified_time", "og:updated_time", "datemodified", "last-modified", "dcterms.modified") ?? jsonLdDate(root, "dateModified")),
79
+ lang: html?.attrs.lang,
80
+ };
81
+ return Object.fromEntries(Object.entries(meta).filter(([, value]) => value !== undefined)) as PageMeta;
82
+ }
83
+
84
+ /** The part of the page that is its content: main, the one article, role=main, or the body. */
85
+ function contentRoot(root: HtmlElement): HtmlElement {
86
+ const main = findFirst(root, (element) => element.name === "main" || element.attrs.role === "main");
87
+ if (main && textOf(main).length > 200) return main;
88
+ const articles = findAll(root, (element) => element.name === "article");
89
+ if (articles.length === 1 && textOf(articles[0]!).length > 200) return articles[0]!;
90
+ const content = findFirst(root, (element) => ["content", "main-content", "maincontent"].includes((element.attrs.id ?? "").toLowerCase()));
91
+ if (content && textOf(content).length > 200) return content;
92
+ return findFirst(root, (element) => element.name === "body") ?? root;
93
+ }
94
+
95
+ /**
96
+ * Drop what is never content; page chrome goes too unless nothing else is
97
+ * left. Reading the whole body, its header (logo, site menu) is chrome as well.
98
+ */
99
+ function prune(node: HtmlElement, keepChrome: boolean, dropHeader: boolean): HtmlElement {
100
+ const children: HtmlNode[] = [];
101
+ for (const child of node.children) {
102
+ if (child.type === "text") children.push(child);
103
+ else if (DROP.has(child.name)) continue;
104
+ else if (!keepChrome && (CHROME.has(child.name) || isChrome(child) || (dropHeader && child.name === "header"))) continue;
105
+ else children.push(prune(child, keepChrome, dropHeader));
106
+ }
107
+ return { ...node, children };
108
+ }
109
+
110
+ interface Context {
111
+ /** Where relative links point from. */
112
+ base: URL | undefined;
113
+ }
114
+
115
+ function absolute(href: string | undefined, base: URL | undefined): string | undefined {
116
+ if (!href) return undefined;
117
+ const trimmed = href.trim();
118
+ if (!trimmed || trimmed.startsWith("#") || /^(javascript|mailto|tel|data):/i.test(trimmed)) return undefined;
119
+ try {
120
+ return new URL(trimmed, base).href;
121
+ } catch {
122
+ return undefined;
123
+ }
124
+ }
125
+
126
+ function fence(code: string): string {
127
+ const longest = Math.max(2, ...[...code.matchAll(/`+/g)].map((run) => run[0].length));
128
+ return "`".repeat(longest + 1);
129
+ }
130
+
131
+ function codeLanguage(element: HtmlElement): string {
132
+ const code = findFirst(element, (child) => child.name === "code");
133
+ const classes = `${element.attrs.class ?? ""} ${code?.attrs.class ?? ""} ${element.attrs["data-lang"] ?? ""}`;
134
+ const match = /(?:language|lang|highlight-source)-([a-z0-9+#-]+)/i.exec(classes) ?? /^\s*([a-z0-9+#-]+)\s*$/i.exec(element.attrs["data-lang"] ?? "");
135
+ return match?.[1]?.toLowerCase() ?? "";
136
+ }
137
+
138
+ function rawText(node: HtmlNode): string {
139
+ if (node.type === "text") return node.text;
140
+ if (node.name === "br") return "\n";
141
+ return node.children.map(rawText).join("");
142
+ }
143
+
144
+ function wrapInline(text: string, mark: string): string {
145
+ const trimmed = text.trim();
146
+ if (!trimmed) return text;
147
+ const lead = text.startsWith(" ") ? " " : "";
148
+ const trail = text.endsWith(" ") ? " " : "";
149
+ return `${lead}${mark}${trimmed}${mark}${trail}`;
150
+ }
151
+
152
+ /** A node's inline Markdown: text with emphasis, code, links and line breaks. */
153
+ function inline(node: HtmlNode, context: Context): string {
154
+ if (node.type === "text") return node.text.replace(/\s+/g, " ");
155
+ const inner = () => node.children.map((child) => inline(child, context)).join("");
156
+ switch (node.name) {
157
+ case "br":
158
+ return "\n";
159
+ case "img": {
160
+ const alt = (node.attrs.alt ?? "").replace(/\s+/g, " ").trim();
161
+ const src = absolute(node.attrs.src, context.base);
162
+ return alt && src ? `![${alt}](${src})` : alt;
163
+ }
164
+ case "a": {
165
+ const text = inner();
166
+ const href = absolute(node.attrs.href, context.base);
167
+ if (!text.trim()) return "";
168
+ return href ? `${text.startsWith(" ") ? " " : ""}[${text.trim()}](${href})${text.endsWith(" ") ? " " : ""}` : text;
169
+ }
170
+ case "strong":
171
+ case "b":
172
+ return wrapInline(inner(), "**");
173
+ case "em":
174
+ case "i":
175
+ case "cite":
176
+ return wrapInline(inner(), "*");
177
+ case "del":
178
+ case "s":
179
+ case "strike":
180
+ return wrapInline(inner(), "~~");
181
+ case "code":
182
+ case "kbd":
183
+ case "samp":
184
+ case "tt": {
185
+ const code = rawText(node).replace(/\s+/g, " ");
186
+ if (!code.trim()) return code;
187
+ const ticks = code.includes("`") ? "``" : "`";
188
+ return `${ticks}${ticks.length > 1 ? " " : ""}${code.trim()}${ticks.length > 1 ? " " : ""}${ticks}`;
189
+ }
190
+ case "sup":
191
+ return `^${inner().trim()}`;
192
+ default:
193
+ // A block inside inline content (a div in a link) reads as a spaced run.
194
+ return BLOCK.has(node.name) ? ` ${inner()} ` : inner();
195
+ }
196
+ }
197
+
198
+ function paragraph(text: string): string {
199
+ return text
200
+ .split("\n")
201
+ .map((line) => line.replace(/[ \t]+/g, " ").trim())
202
+ .join("\n")
203
+ .replace(/\n{3,}/g, "\n\n")
204
+ .trim();
205
+ }
206
+
207
+ function indent(text: string, prefix: string, first = prefix): string {
208
+ return text
209
+ .split("\n")
210
+ .map((line, index) => (index === 0 ? first : line ? prefix : "") + line)
211
+ .join("\n");
212
+ }
213
+
214
+ function table(element: HtmlElement, context: Context): string {
215
+ const rows = findAll(element, (child) => child.name === "tr").map((row) =>
216
+ row.children
217
+ .filter((cell): cell is HtmlElement => cell.type === "element" && (cell.name === "td" || cell.name === "th"))
218
+ .map((cell) => paragraph(inline(cell, context)).replace(/\n+/g, " ").replace(/\|/g, "\\|")),
219
+ );
220
+ const filled = rows.filter((row) => row.some((cell) => cell));
221
+ if (filled.length === 0) return "";
222
+ const width = Math.max(...filled.map((row) => row.length));
223
+ const line = (cells: string[]) => `| ${[...cells, ...Array(width - cells.length).fill("")].join(" | ")} |`;
224
+ return [line(filled[0]!), `|${" --- |".repeat(width)}`, ...filled.slice(1).map(line)].join("\n");
225
+ }
226
+
227
+ function list(element: HtmlElement, context: Context): string {
228
+ const ordered = element.name === "ol";
229
+ let number = Number.parseInt(element.attrs.start ?? "1", 10) || 1;
230
+ const items: string[] = [];
231
+ for (const child of element.children) {
232
+ if (child.type === "text") {
233
+ if (child.text.trim()) items.push(`- ${child.text.trim()}`);
234
+ continue;
235
+ }
236
+ const body = child.name === "li" ? blocks(child, context).join("\n") : blocks({ ...child, children: [child] }, context).join("\n");
237
+ if (!body.trim()) continue;
238
+ const marker = ordered ? `${number++}. ` : "- ";
239
+ items.push(indent(body, " ".repeat(marker.length), marker));
240
+ }
241
+ return items.join("\n");
242
+ }
243
+
244
+ /** A block element as Markdown blocks. */
245
+ function block(element: HtmlElement, context: Context): string[] {
246
+ switch (element.name) {
247
+ case "h1":
248
+ case "h2":
249
+ case "h3":
250
+ case "h4":
251
+ case "h5":
252
+ case "h6": {
253
+ const text = paragraph(inline(element, context)).replace(/\n+/g, " ");
254
+ return text ? [`${"#".repeat(Number(element.name[1]))} ${text}`] : [];
255
+ }
256
+ case "p":
257
+ case "summary":
258
+ case "figcaption":
259
+ case "dt": {
260
+ const text = paragraph(element.children.map((child) => inline(child, context)).join(""));
261
+ return text ? [element.name === "dt" ? `**${text}**` : text] : [];
262
+ }
263
+ case "pre": {
264
+ const code = rawText(element).replace(/^\n/, "").replace(/\s+$/, "");
265
+ if (!code.trim()) return [];
266
+ const marks = fence(code);
267
+ return [`${marks}${codeLanguage(element)}\n${code}\n${marks}`];
268
+ }
269
+ case "ul":
270
+ case "ol":
271
+ case "menu": {
272
+ const text = list(element, context);
273
+ return text ? [text] : [];
274
+ }
275
+ case "blockquote": {
276
+ const inner = blocks(element, context).join("\n\n");
277
+ return inner ? [indent(inner, "> ")] : [];
278
+ }
279
+ case "table": {
280
+ const text = table(element, context);
281
+ return text ? [text] : [];
282
+ }
283
+ case "hr":
284
+ return ["---"];
285
+ default:
286
+ return blocks(element, context);
287
+ }
288
+ }
289
+
290
+ /** The children of an element as Markdown blocks: inline runs become paragraphs. */
291
+ function blocks(element: HtmlElement, context: Context): string[] {
292
+ const out: string[] = [];
293
+ let run = "";
294
+ const flush = () => {
295
+ const text = paragraph(run);
296
+ if (text) out.push(text);
297
+ run = "";
298
+ };
299
+ for (const child of element.children) {
300
+ if (child.type === "element" && BLOCK.has(child.name)) {
301
+ flush();
302
+ out.push(...block(child, context));
303
+ } else {
304
+ run += inline(child, context);
305
+ }
306
+ }
307
+ flush();
308
+ return out;
309
+ }
310
+
311
+ /** An HTML page as its metadata plus its main content in Markdown. */
312
+ export function htmlToMarkdown(html: string, url?: string): ReadablePage {
313
+ const root = parseHtml(html);
314
+ const meta = pageMeta(root);
315
+ const baseHref = findFirst(root, (element) => element.name === "base")?.attrs.href;
316
+ let base: URL | undefined;
317
+ try {
318
+ base = url ? new URL(baseHref ?? "", url) : undefined;
319
+ } catch {
320
+ base = undefined;
321
+ }
322
+ const content = contentRoot(root);
323
+ const wholePage = content.name === "body" || content.name === "#root";
324
+ let pruned = prune(content, false, wholePage);
325
+ if (textOf(pruned).length < 80) pruned = prune(content, true, false);
326
+ const markdown = blocks(pruned, { base })
327
+ .join("\n\n")
328
+ .replace(/\n{3,}/g, "\n\n")
329
+ .trim();
330
+ return { ...meta, markdown };
331
+ }
332
+
@@ -0,0 +1,232 @@
1
+ /**
2
+ * Fetching a page for a model, safely: http(s) only, public addresses only
3
+ * (a page cannot send the researcher to localhost, the LAN or a cloud
4
+ * metadata endpoint, through a redirect either), a size cap, a time limit,
5
+ * and the body decoded by its declared charset.
6
+ * `BOT_LOBBY_WEB_ALLOW_PRIVATE=1` lifts the address rule (a local docs server).
7
+ */
8
+ import { lookup as dnsLookup } from "node:dns/promises";
9
+ import { isIP } from "node:net";
10
+
11
+ export type Fetcher = typeof fetch;
12
+ export type Lookup = (host: string) => Promise<string[]>;
13
+
14
+ export interface FetchOptions {
15
+ signal?: AbortSignal;
16
+ /** Bytes read at most; the rest is cut off. */
17
+ maxBytes?: number;
18
+ timeoutMs?: number;
19
+ accept?: string;
20
+ method?: "GET" | "POST";
21
+ body?: string;
22
+ headers?: Record<string, string>;
23
+ /** Swappable for tests. */
24
+ fetch?: Fetcher;
25
+ lookup?: Lookup;
26
+ allowPrivate?: boolean;
27
+ }
28
+
29
+ export interface FetchedPage {
30
+ /** Where the page was finally read from, after redirects. */
31
+ url: string;
32
+ status: number;
33
+ statusText: string;
34
+ contentType: string;
35
+ headers: Headers;
36
+ /** The body as text; empty for a binary one (a PDF, an image), which is not read. */
37
+ text: string;
38
+ bytes: number;
39
+ truncated: boolean;
40
+ binary: boolean;
41
+ redirects: string[];
42
+ }
43
+
44
+ export const MAX_BYTES = 3 * 1024 * 1024;
45
+ export const TIMEOUT_MS = 20_000;
46
+ const MAX_REDIRECTS = 5;
47
+ export const USER_AGENT = "Mozilla/5.0 (compatible; bot-lobby research; +https://github.com/a-t-h-i/bot-lobby)";
48
+ const ACCEPT = "text/html,application/xhtml+xml,text/markdown;q=0.9,text/plain;q=0.8,application/json;q=0.7,*/*;q=0.5";
49
+
50
+ /** True for an address that is not on the public internet. */
51
+ export function isPrivateAddress(address: string): boolean {
52
+ const ip = address.replace(/^\[|\]$/g, "").toLowerCase();
53
+ const version = isIP(ip);
54
+ if (version === 4) {
55
+ const [a = 0, b = 0] = ip.split(".").map(Number);
56
+ return a === 0 || a === 10 || a === 127 || a >= 224 || (a === 100 && b >= 64 && b <= 127) || (a === 169 && b === 254) || (a === 172 && b >= 16 && b <= 31) || (a === 192 && b === 168) || (a === 192 && b === 0) || (a === 198 && (b === 18 || b === 19));
57
+ }
58
+ if (version === 6) {
59
+ if (ip === "::" || ip === "::1") return true;
60
+ const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(ip) ?? /^64:ff9b::(\d+\.\d+\.\d+\.\d+)$/.exec(ip);
61
+ if (mapped) return isPrivateAddress(mapped[1]!);
62
+ if (/^::ffff:[0-9a-f]{1,4}:[0-9a-f]{1,4}$/.test(ip)) {
63
+ const [high = "0", low = "0"] = ip.slice(7).split(":");
64
+ const value = (Number.parseInt(high, 16) << 16) | Number.parseInt(low, 16);
65
+ return isPrivateAddress([value >>> 24, (value >>> 16) & 255, (value >>> 8) & 255, value & 255].join("."));
66
+ }
67
+ return /^(fc|fd|fe[89ab]|ff)/.test(ip);
68
+ }
69
+ return false;
70
+ }
71
+
72
+ const defaultLookup: Lookup = async (host) => (await dnsLookup(host, { all: true, verbatim: true })).map((entry) => entry.address);
73
+
74
+ /**
75
+ * Why a URL may not be fetched, or undefined when it may: not http(s), a
76
+ * private or local host, or a name that resolves to one. A name that does not
77
+ * resolve here is let through (a proxy may resolve it; the fetch then says).
78
+ */
79
+ export async function blockedReason(url: URL, lookup: Lookup = defaultLookup, allowPrivate = process.env.BOT_LOBBY_WEB_ALLOW_PRIVATE === "1"): Promise<string | undefined> {
80
+ if (url.protocol !== "http:" && url.protocol !== "https:") return `only http and https pages can be fetched, not ${url.protocol.replace(/:$/, "")}`;
81
+ if (url.username || url.password) return "URLs with credentials in them are not fetched";
82
+ if (allowPrivate) return undefined;
83
+ const host = url.hostname.replace(/^\[|\]$/g, "").toLowerCase().replace(/\.$/, "");
84
+ const local = "is a private or local address; only public pages are fetched (BOT_LOBBY_WEB_ALLOW_PRIVATE=1 allows it)";
85
+ if (isIP(host)) return isPrivateAddress(host) ? `${host} ${local}` : undefined;
86
+ if (host === "localhost" || /\.(localhost|local|internal|home\.arpa|lan)$/.test(host) || !host.includes(".")) return `${host} ${local}`;
87
+ let addresses: string[];
88
+ try {
89
+ addresses = await lookup(host);
90
+ } catch {
91
+ return undefined;
92
+ }
93
+ const inside = addresses.find(isPrivateAddress);
94
+ return inside ? `${host} resolves to ${inside}, which ${local}` : undefined;
95
+ }
96
+
97
+ function charsetOf(contentType: string, head: Uint8Array): string {
98
+ const declared = /charset\s*=\s*"?([\w.:-]+)/i.exec(contentType)?.[1];
99
+ if (declared) return declared;
100
+ if (!/html|xml/i.test(contentType)) return "utf-8";
101
+ const sniff = new TextDecoder("latin1").decode(head.subarray(0, 2048));
102
+ return /<meta[^>]+charset\s*=\s*["']?([\w.:-]+)/i.exec(sniff)?.[1] ?? "utf-8";
103
+ }
104
+
105
+ function decode(bytes: Uint8Array, charset: string): string {
106
+ try {
107
+ return new TextDecoder(charset).decode(bytes);
108
+ } catch {
109
+ return new TextDecoder("utf-8").decode(bytes);
110
+ }
111
+ }
112
+
113
+ async function readCapped(response: Response, maxBytes: number): Promise<{ bytes: Uint8Array; truncated: boolean }> {
114
+ if (!response.body) return { bytes: new Uint8Array(await response.arrayBuffer()).subarray(0, maxBytes), truncated: false };
115
+ const reader = response.body.getReader();
116
+ const chunks: Uint8Array[] = [];
117
+ let size = 0;
118
+ let truncated = false;
119
+ for (;;) {
120
+ const { done, value } = await reader.read();
121
+ if (done) break;
122
+ if (size + value.byteLength > maxBytes) {
123
+ chunks.push(value.subarray(0, maxBytes - size));
124
+ size = maxBytes;
125
+ truncated = true;
126
+ await reader.cancel().catch(() => {});
127
+ break;
128
+ }
129
+ chunks.push(value);
130
+ size += value.byteLength;
131
+ }
132
+ const bytes = new Uint8Array(size);
133
+ let offset = 0;
134
+ for (const chunk of chunks) {
135
+ bytes.set(chunk, offset);
136
+ offset += chunk.byteLength;
137
+ }
138
+ return { bytes, truncated };
139
+ }
140
+
141
+ const TEXTUAL = /^(text\/|application\/(json|xml|xhtml\+xml|javascript|x-javascript|ecmascript|x-yaml|yaml|toml)\b|[^;]*\+(json|xml)\b)/i;
142
+
143
+ /** True for a content type read as text (an unlabelled body counts). */
144
+ export function isTextual(contentType: string): boolean {
145
+ return !contentType.trim() || TEXTUAL.test(contentType.trim());
146
+ }
147
+
148
+ /** A friendlier message for why a fetch failed. */
149
+ export function fetchError(error: unknown, url: string): Error {
150
+ const cause = (error as { cause?: { code?: string; message?: string } })?.cause;
151
+ const name = (error as { name?: string })?.name;
152
+ if (name === "TimeoutError") return new Error(`${url} did not answer in time`);
153
+ if (name === "AbortError") return new Error(`fetching ${url} was stopped`);
154
+ const code = cause?.code ?? "";
155
+ const reason = code === "ENOTFOUND" ? "the host does not exist" : code === "ECONNREFUSED" ? "the connection was refused" : code === "CERT_HAS_EXPIRED" || /CERT|SSL|TLS/.test(code) ? "its TLS certificate is not valid" : cause?.message ?? (error instanceof Error ? error.message : String(error));
156
+ return new Error(`could not fetch ${url}: ${reason}`);
157
+ }
158
+
159
+ /**
160
+ * Fetch a public page (following up to five redirects, each checked like the
161
+ * first) and read at most `maxBytes` of it as text. Throws with a readable
162
+ * reason when the URL is refused or unreachable; an HTTP error status is a
163
+ * page like any other, for the caller to judge.
164
+ */
165
+ export async function fetchPage(input: string, options: FetchOptions = {}): Promise<FetchedPage> {
166
+ const fetcher = options.fetch ?? globalThis.fetch;
167
+ const lookup = options.lookup ?? defaultLookup;
168
+ const allowPrivate = options.allowPrivate ?? process.env.BOT_LOBBY_WEB_ALLOW_PRIVATE === "1";
169
+ let url: URL;
170
+ try {
171
+ url = new URL(input.trim());
172
+ } catch {
173
+ throw new Error(`"${input}" is not a URL`);
174
+ }
175
+ const timeout = AbortSignal.timeout(options.timeoutMs ?? TIMEOUT_MS);
176
+ const signal = options.signal ? AbortSignal.any([options.signal, timeout]) : timeout;
177
+ const redirects: string[] = [];
178
+ let method = options.method ?? "GET";
179
+ let body = options.body;
180
+ for (;;) {
181
+ const blocked = await blockedReason(url, lookup, allowPrivate);
182
+ if (blocked) throw new Error(`not fetched: ${blocked}`);
183
+ let response: Response;
184
+ try {
185
+ response = await fetcher(url.href, {
186
+ method,
187
+ ...(body !== undefined ? { body } : {}),
188
+ redirect: "manual",
189
+ signal,
190
+ headers: { "user-agent": USER_AGENT, accept: options.accept ?? ACCEPT, "accept-language": "en;q=0.9, *;q=0.5", ...options.headers },
191
+ });
192
+ } catch (error) {
193
+ throw fetchError(error, url.href);
194
+ }
195
+ const location = response.headers.get("location");
196
+ if (response.status >= 300 && response.status < 400 && location) {
197
+ await response.body?.cancel().catch(() => {});
198
+ if (redirects.length >= MAX_REDIRECTS) throw new Error(`${input} redirects more than ${MAX_REDIRECTS} times`);
199
+ redirects.push(url.href);
200
+ url = new URL(location, url);
201
+ if (response.status === 303 || ((response.status === 301 || response.status === 302) && method === "POST")) {
202
+ method = "GET";
203
+ body = undefined;
204
+ }
205
+ continue;
206
+ }
207
+ const contentType = response.headers.get("content-type") ?? "";
208
+ if (!isTextual(contentType)) {
209
+ await response.body?.cancel().catch(() => {});
210
+ const length = Number(response.headers.get("content-length") ?? 0);
211
+ return { url: url.href, status: response.status, statusText: response.statusText, contentType, headers: response.headers, text: "", bytes: Number.isFinite(length) ? length : 0, truncated: false, binary: true, redirects };
212
+ }
213
+ let read: { bytes: Uint8Array; truncated: boolean };
214
+ try {
215
+ read = await readCapped(response, options.maxBytes ?? MAX_BYTES);
216
+ } catch (error) {
217
+ throw fetchError(error, url.href);
218
+ }
219
+ return {
220
+ url: url.href,
221
+ status: response.status,
222
+ statusText: response.statusText,
223
+ contentType,
224
+ headers: response.headers,
225
+ text: decode(read.bytes, charsetOf(contentType, read.bytes)),
226
+ bytes: read.bytes.byteLength,
227
+ truncated: read.truncated,
228
+ binary: false,
229
+ redirects,
230
+ };
231
+ }
232
+ }