@a-t-h-i/bot-lobby 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +112 -11
- package/package.json +1 -4
- package/prompts/master.md +47 -1
- package/prompts/researcher.md +8 -2
- package/prompts/reviewer.md +42 -0
- package/prompts/worker.md +7 -0
- package/src/ask/dialog.ts +167 -0
- package/src/ask/image.ts +202 -0
- package/src/ask/png.ts +179 -0
- package/src/ask/relay.ts +89 -0
- package/src/ask/state.ts +160 -0
- package/src/ask/tool.ts +126 -0
- package/src/ask/types.ts +55 -0
- package/src/ask/view.ts +159 -0
- package/src/execution/agent-runner.ts +103 -11
- package/src/execution/git.ts +111 -14
- package/src/execution/pi-runner.ts +164 -11
- package/src/index.ts +9 -0
- package/src/lobby/ask.ts +10 -120
- package/src/lobby/feed.ts +76 -8
- package/src/lobby/layout.ts +38 -18
- package/src/lobby/markdown.ts +92 -19
- package/src/lobby/planner.ts +1 -1
- package/src/lobby/quickfix.ts +21 -0
- package/src/lobby/runtime.ts +51 -37
- package/src/lobby/session-files.ts +162 -25
- package/src/lobby/sessions.ts +21 -7
- package/src/lobby/tabs/home.ts +289 -67
- package/src/lobby/tabs/issues.ts +4 -3
- package/src/lobby/tabs/plan.ts +4 -4
- package/src/lobby/tabs/quickfix.ts +7 -1
- package/src/lobby/tabs/tasks.ts +14 -4
- package/src/lobby/theme.ts +30 -0
- package/src/lobby/view.ts +141 -25
- package/src/master/decisions.ts +1 -1
- package/src/master/master.ts +41 -4
- package/src/master/research.ts +5 -2
- package/src/pi/commands.ts +94 -10
- package/src/pi/events.ts +54 -12
- package/src/pi/fresh-context.ts +134 -0
- package/src/pi/owner.ts +19 -10
- package/src/pi/quiet.ts +22 -4
- package/src/pi/start-task.ts +11 -2
- package/src/pi/tools.ts +26 -8
- package/src/pi/ui.ts +9 -3
- package/src/pi/zen-large.ts +10 -10
- package/src/pi/zen-metrics.ts +13 -3
- package/src/pi/zen.ts +22 -15
- package/src/roles/reviewer.ts +23 -4
- package/src/roles/worker.ts +18 -0
- package/src/schemas/configuration.ts +8 -0
- package/src/schemas/findings.ts +14 -0
- package/src/schemas/task.ts +21 -0
- package/src/state/archive.ts +12 -3
- package/src/state/backlog.ts +13 -3
- package/src/state/budget.ts +274 -0
- package/src/state/changes.ts +231 -0
- package/src/state/file-cache.ts +62 -0
- package/src/state/metrics.ts +63 -13
- package/src/state/persistence.ts +36 -2
- package/src/text.ts +28 -2
- package/src/web/extract.ts +332 -0
- package/src/web/fetch.ts +232 -0
- package/src/web/html.ts +183 -0
- package/src/web/read.ts +113 -0
- package/src/web/search.ts +202 -0
- package/src/web/tools.ts +279 -0
- package/src/width.ts +102 -0
- package/src/workflow/workflow.ts +592 -42
package/src/web/fetch.ts
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fetching a page for a model, safely: http(s) only, public addresses only
|
|
3
|
+
* (a page cannot send the researcher to localhost, the LAN or a cloud
|
|
4
|
+
* metadata endpoint, through a redirect either), a size cap, a time limit,
|
|
5
|
+
* and the body decoded by its declared charset.
|
|
6
|
+
* `BOT_LOBBY_WEB_ALLOW_PRIVATE=1` lifts the address rule (a local docs server).
|
|
7
|
+
*/
|
|
8
|
+
import { lookup as dnsLookup } from "node:dns/promises";
|
|
9
|
+
import { isIP } from "node:net";
|
|
10
|
+
|
|
11
|
+
export type Fetcher = typeof fetch;
|
|
12
|
+
export type Lookup = (host: string) => Promise<string[]>;
|
|
13
|
+
|
|
14
|
+
export interface FetchOptions {
|
|
15
|
+
signal?: AbortSignal;
|
|
16
|
+
/** Bytes read at most; the rest is cut off. */
|
|
17
|
+
maxBytes?: number;
|
|
18
|
+
timeoutMs?: number;
|
|
19
|
+
accept?: string;
|
|
20
|
+
method?: "GET" | "POST";
|
|
21
|
+
body?: string;
|
|
22
|
+
headers?: Record<string, string>;
|
|
23
|
+
/** Swappable for tests. */
|
|
24
|
+
fetch?: Fetcher;
|
|
25
|
+
lookup?: Lookup;
|
|
26
|
+
allowPrivate?: boolean;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export interface FetchedPage {
|
|
30
|
+
/** Where the page was finally read from, after redirects. */
|
|
31
|
+
url: string;
|
|
32
|
+
status: number;
|
|
33
|
+
statusText: string;
|
|
34
|
+
contentType: string;
|
|
35
|
+
headers: Headers;
|
|
36
|
+
/** The body as text; empty for a binary one (a PDF, an image), which is not read. */
|
|
37
|
+
text: string;
|
|
38
|
+
bytes: number;
|
|
39
|
+
truncated: boolean;
|
|
40
|
+
binary: boolean;
|
|
41
|
+
redirects: string[];
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export const MAX_BYTES = 3 * 1024 * 1024;
|
|
45
|
+
export const TIMEOUT_MS = 20_000;
|
|
46
|
+
const MAX_REDIRECTS = 5;
|
|
47
|
+
export const USER_AGENT = "Mozilla/5.0 (compatible; bot-lobby research; +https://github.com/a-t-h-i/bot-lobby)";
|
|
48
|
+
const ACCEPT = "text/html,application/xhtml+xml,text/markdown;q=0.9,text/plain;q=0.8,application/json;q=0.7,*/*;q=0.5";
|
|
49
|
+
|
|
50
|
+
/** True for an address that is not on the public internet. */
|
|
51
|
+
export function isPrivateAddress(address: string): boolean {
|
|
52
|
+
const ip = address.replace(/^\[|\]$/g, "").toLowerCase();
|
|
53
|
+
const version = isIP(ip);
|
|
54
|
+
if (version === 4) {
|
|
55
|
+
const [a = 0, b = 0] = ip.split(".").map(Number);
|
|
56
|
+
return a === 0 || a === 10 || a === 127 || a >= 224 || (a === 100 && b >= 64 && b <= 127) || (a === 169 && b === 254) || (a === 172 && b >= 16 && b <= 31) || (a === 192 && b === 168) || (a === 192 && b === 0) || (a === 198 && (b === 18 || b === 19));
|
|
57
|
+
}
|
|
58
|
+
if (version === 6) {
|
|
59
|
+
if (ip === "::" || ip === "::1") return true;
|
|
60
|
+
const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(ip) ?? /^64:ff9b::(\d+\.\d+\.\d+\.\d+)$/.exec(ip);
|
|
61
|
+
if (mapped) return isPrivateAddress(mapped[1]!);
|
|
62
|
+
if (/^::ffff:[0-9a-f]{1,4}:[0-9a-f]{1,4}$/.test(ip)) {
|
|
63
|
+
const [high = "0", low = "0"] = ip.slice(7).split(":");
|
|
64
|
+
const value = (Number.parseInt(high, 16) << 16) | Number.parseInt(low, 16);
|
|
65
|
+
return isPrivateAddress([value >>> 24, (value >>> 16) & 255, (value >>> 8) & 255, value & 255].join("."));
|
|
66
|
+
}
|
|
67
|
+
return /^(fc|fd|fe[89ab]|ff)/.test(ip);
|
|
68
|
+
}
|
|
69
|
+
return false;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
const defaultLookup: Lookup = async (host) => (await dnsLookup(host, { all: true, verbatim: true })).map((entry) => entry.address);
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Why a URL may not be fetched, or undefined when it may: not http(s), a
|
|
76
|
+
* private or local host, or a name that resolves to one. A name that does not
|
|
77
|
+
* resolve here is let through (a proxy may resolve it; the fetch then says).
|
|
78
|
+
*/
|
|
79
|
+
export async function blockedReason(url: URL, lookup: Lookup = defaultLookup, allowPrivate = process.env.BOT_LOBBY_WEB_ALLOW_PRIVATE === "1"): Promise<string | undefined> {
|
|
80
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") return `only http and https pages can be fetched, not ${url.protocol.replace(/:$/, "")}`;
|
|
81
|
+
if (url.username || url.password) return "URLs with credentials in them are not fetched";
|
|
82
|
+
if (allowPrivate) return undefined;
|
|
83
|
+
const host = url.hostname.replace(/^\[|\]$/g, "").toLowerCase().replace(/\.$/, "");
|
|
84
|
+
const local = "is a private or local address; only public pages are fetched (BOT_LOBBY_WEB_ALLOW_PRIVATE=1 allows it)";
|
|
85
|
+
if (isIP(host)) return isPrivateAddress(host) ? `${host} ${local}` : undefined;
|
|
86
|
+
if (host === "localhost" || /\.(localhost|local|internal|home\.arpa|lan)$/.test(host) || !host.includes(".")) return `${host} ${local}`;
|
|
87
|
+
let addresses: string[];
|
|
88
|
+
try {
|
|
89
|
+
addresses = await lookup(host);
|
|
90
|
+
} catch {
|
|
91
|
+
return undefined;
|
|
92
|
+
}
|
|
93
|
+
const inside = addresses.find(isPrivateAddress);
|
|
94
|
+
return inside ? `${host} resolves to ${inside}, which ${local}` : undefined;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function charsetOf(contentType: string, head: Uint8Array): string {
|
|
98
|
+
const declared = /charset\s*=\s*"?([\w.:-]+)/i.exec(contentType)?.[1];
|
|
99
|
+
if (declared) return declared;
|
|
100
|
+
if (!/html|xml/i.test(contentType)) return "utf-8";
|
|
101
|
+
const sniff = new TextDecoder("latin1").decode(head.subarray(0, 2048));
|
|
102
|
+
return /<meta[^>]+charset\s*=\s*["']?([\w.:-]+)/i.exec(sniff)?.[1] ?? "utf-8";
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function decode(bytes: Uint8Array, charset: string): string {
|
|
106
|
+
try {
|
|
107
|
+
return new TextDecoder(charset).decode(bytes);
|
|
108
|
+
} catch {
|
|
109
|
+
return new TextDecoder("utf-8").decode(bytes);
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
async function readCapped(response: Response, maxBytes: number): Promise<{ bytes: Uint8Array; truncated: boolean }> {
|
|
114
|
+
if (!response.body) return { bytes: new Uint8Array(await response.arrayBuffer()).subarray(0, maxBytes), truncated: false };
|
|
115
|
+
const reader = response.body.getReader();
|
|
116
|
+
const chunks: Uint8Array[] = [];
|
|
117
|
+
let size = 0;
|
|
118
|
+
let truncated = false;
|
|
119
|
+
for (;;) {
|
|
120
|
+
const { done, value } = await reader.read();
|
|
121
|
+
if (done) break;
|
|
122
|
+
if (size + value.byteLength > maxBytes) {
|
|
123
|
+
chunks.push(value.subarray(0, maxBytes - size));
|
|
124
|
+
size = maxBytes;
|
|
125
|
+
truncated = true;
|
|
126
|
+
await reader.cancel().catch(() => {});
|
|
127
|
+
break;
|
|
128
|
+
}
|
|
129
|
+
chunks.push(value);
|
|
130
|
+
size += value.byteLength;
|
|
131
|
+
}
|
|
132
|
+
const bytes = new Uint8Array(size);
|
|
133
|
+
let offset = 0;
|
|
134
|
+
for (const chunk of chunks) {
|
|
135
|
+
bytes.set(chunk, offset);
|
|
136
|
+
offset += chunk.byteLength;
|
|
137
|
+
}
|
|
138
|
+
return { bytes, truncated };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const TEXTUAL = /^(text\/|application\/(json|xml|xhtml\+xml|javascript|x-javascript|ecmascript|x-yaml|yaml|toml)\b|[^;]*\+(json|xml)\b)/i;
|
|
142
|
+
|
|
143
|
+
/** True for a content type read as text (an unlabelled body counts). */
|
|
144
|
+
export function isTextual(contentType: string): boolean {
|
|
145
|
+
return !contentType.trim() || TEXTUAL.test(contentType.trim());
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/** A friendlier message for why a fetch failed. */
|
|
149
|
+
export function fetchError(error: unknown, url: string): Error {
|
|
150
|
+
const cause = (error as { cause?: { code?: string; message?: string } })?.cause;
|
|
151
|
+
const name = (error as { name?: string })?.name;
|
|
152
|
+
if (name === "TimeoutError") return new Error(`${url} did not answer in time`);
|
|
153
|
+
if (name === "AbortError") return new Error(`fetching ${url} was stopped`);
|
|
154
|
+
const code = cause?.code ?? "";
|
|
155
|
+
const reason = code === "ENOTFOUND" ? "the host does not exist" : code === "ECONNREFUSED" ? "the connection was refused" : code === "CERT_HAS_EXPIRED" || /CERT|SSL|TLS/.test(code) ? "its TLS certificate is not valid" : cause?.message ?? (error instanceof Error ? error.message : String(error));
|
|
156
|
+
return new Error(`could not fetch ${url}: ${reason}`);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Fetch a public page (following up to five redirects, each checked like the
|
|
161
|
+
* first) and read at most `maxBytes` of it as text. Throws with a readable
|
|
162
|
+
* reason when the URL is refused or unreachable; an HTTP error status is a
|
|
163
|
+
* page like any other, for the caller to judge.
|
|
164
|
+
*/
|
|
165
|
+
export async function fetchPage(input: string, options: FetchOptions = {}): Promise<FetchedPage> {
|
|
166
|
+
const fetcher = options.fetch ?? globalThis.fetch;
|
|
167
|
+
const lookup = options.lookup ?? defaultLookup;
|
|
168
|
+
const allowPrivate = options.allowPrivate ?? process.env.BOT_LOBBY_WEB_ALLOW_PRIVATE === "1";
|
|
169
|
+
let url: URL;
|
|
170
|
+
try {
|
|
171
|
+
url = new URL(input.trim());
|
|
172
|
+
} catch {
|
|
173
|
+
throw new Error(`"${input}" is not a URL`);
|
|
174
|
+
}
|
|
175
|
+
const timeout = AbortSignal.timeout(options.timeoutMs ?? TIMEOUT_MS);
|
|
176
|
+
const signal = options.signal ? AbortSignal.any([options.signal, timeout]) : timeout;
|
|
177
|
+
const redirects: string[] = [];
|
|
178
|
+
let method = options.method ?? "GET";
|
|
179
|
+
let body = options.body;
|
|
180
|
+
for (;;) {
|
|
181
|
+
const blocked = await blockedReason(url, lookup, allowPrivate);
|
|
182
|
+
if (blocked) throw new Error(`not fetched: ${blocked}`);
|
|
183
|
+
let response: Response;
|
|
184
|
+
try {
|
|
185
|
+
response = await fetcher(url.href, {
|
|
186
|
+
method,
|
|
187
|
+
...(body !== undefined ? { body } : {}),
|
|
188
|
+
redirect: "manual",
|
|
189
|
+
signal,
|
|
190
|
+
headers: { "user-agent": USER_AGENT, accept: options.accept ?? ACCEPT, "accept-language": "en;q=0.9, *;q=0.5", ...options.headers },
|
|
191
|
+
});
|
|
192
|
+
} catch (error) {
|
|
193
|
+
throw fetchError(error, url.href);
|
|
194
|
+
}
|
|
195
|
+
const location = response.headers.get("location");
|
|
196
|
+
if (response.status >= 300 && response.status < 400 && location) {
|
|
197
|
+
await response.body?.cancel().catch(() => {});
|
|
198
|
+
if (redirects.length >= MAX_REDIRECTS) throw new Error(`${input} redirects more than ${MAX_REDIRECTS} times`);
|
|
199
|
+
redirects.push(url.href);
|
|
200
|
+
url = new URL(location, url);
|
|
201
|
+
if (response.status === 303 || ((response.status === 301 || response.status === 302) && method === "POST")) {
|
|
202
|
+
method = "GET";
|
|
203
|
+
body = undefined;
|
|
204
|
+
}
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
const contentType = response.headers.get("content-type") ?? "";
|
|
208
|
+
if (!isTextual(contentType)) {
|
|
209
|
+
await response.body?.cancel().catch(() => {});
|
|
210
|
+
const length = Number(response.headers.get("content-length") ?? 0);
|
|
211
|
+
return { url: url.href, status: response.status, statusText: response.statusText, contentType, headers: response.headers, text: "", bytes: Number.isFinite(length) ? length : 0, truncated: false, binary: true, redirects };
|
|
212
|
+
}
|
|
213
|
+
let read: { bytes: Uint8Array; truncated: boolean };
|
|
214
|
+
try {
|
|
215
|
+
read = await readCapped(response, options.maxBytes ?? MAX_BYTES);
|
|
216
|
+
} catch (error) {
|
|
217
|
+
throw fetchError(error, url.href);
|
|
218
|
+
}
|
|
219
|
+
return {
|
|
220
|
+
url: url.href,
|
|
221
|
+
status: response.status,
|
|
222
|
+
statusText: response.statusText,
|
|
223
|
+
contentType,
|
|
224
|
+
headers: response.headers,
|
|
225
|
+
text: decode(read.bytes, charsetOf(contentType, read.bytes)),
|
|
226
|
+
bytes: read.bytes.byteLength,
|
|
227
|
+
truncated: read.truncated,
|
|
228
|
+
binary: false,
|
|
229
|
+
redirects,
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
}
|
package/src/web/html.ts
ADDED
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A forgiving HTML parser, just enough to read web pages: tags into a tree
|
|
3
|
+
* (void elements, raw text in script/style, implied closes for p, li, td and
|
|
4
|
+
* the like, stray end tags ignored), entities decoded. No dependency; not a
|
|
5
|
+
* spec parser, and it does not need to be.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
export interface HtmlText {
|
|
9
|
+
type: "text";
|
|
10
|
+
text: string;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export interface HtmlElement {
|
|
14
|
+
type: "element";
|
|
15
|
+
name: string;
|
|
16
|
+
attrs: Record<string, string>;
|
|
17
|
+
children: HtmlNode[];
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export type HtmlNode = HtmlText | HtmlElement;
|
|
21
|
+
|
|
22
|
+
const VOID = new Set(["area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr"]);
|
|
23
|
+
/** Elements whose content is text up to their end tag, never markup. */
|
|
24
|
+
const RAW = new Set(["script", "style", "textarea", "title", "xmp", "noscript", "template"]);
|
|
25
|
+
/** Opening one of these closes an open element of the listed names (up to the given boundaries). */
|
|
26
|
+
const IMPLIED: Record<string, { closes: string[]; within: string[] }> = {
|
|
27
|
+
p: { closes: ["p"], within: ["div", "section", "article", "main", "body", "td", "th", "li", "blockquote"] },
|
|
28
|
+
li: { closes: ["li"], within: ["ul", "ol", "menu"] },
|
|
29
|
+
dt: { closes: ["dt", "dd"], within: ["dl"] },
|
|
30
|
+
dd: { closes: ["dt", "dd"], within: ["dl"] },
|
|
31
|
+
tr: { closes: ["tr", "td", "th"], within: ["table", "thead", "tbody", "tfoot"] },
|
|
32
|
+
td: { closes: ["td", "th"], within: ["tr", "table"] },
|
|
33
|
+
th: { closes: ["td", "th"], within: ["tr", "table"] },
|
|
34
|
+
thead: { closes: ["thead", "tbody", "tr", "td", "th"], within: ["table"] },
|
|
35
|
+
tbody: { closes: ["thead", "tbody", "tr", "td", "th"], within: ["table"] },
|
|
36
|
+
tfoot: { closes: ["thead", "tbody", "tr", "td", "th"], within: ["table"] },
|
|
37
|
+
option: { closes: ["option"], within: ["select", "datalist"] },
|
|
38
|
+
};
|
|
39
|
+
/** Block elements that end an open paragraph. */
|
|
40
|
+
const CLOSES_P = new Set(["address", "article", "aside", "blockquote", "details", "div", "dl", "fieldset", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", "h6", "header", "hr", "main", "nav", "ol", "pre", "section", "table", "ul"]);
|
|
41
|
+
|
|
42
|
+
const NAMED: Record<string, string> = {
|
|
43
|
+
amp: "&", lt: "<", gt: ">", quot: "\"", apos: "'", nbsp: " ", ensp: " ", emsp: " ", thinsp: " ", shy: "",
|
|
44
|
+
copy: "©", reg: "®", trade: "™", hellip: "…", mdash: "—", ndash: "–", minus: "−",
|
|
45
|
+
lsquo: "‘", rsquo: "’", sbquo: "‚", ldquo: "“", rdquo: "”", bdquo: "„", laquo: "«", raquo: "»", lsaquo: "‹", rsaquo: "›",
|
|
46
|
+
bull: "•", middot: "·", times: "×", divide: "÷", deg: "°", plusmn: "±", para: "¶", sect: "§", dagger: "†", Dagger: "‡",
|
|
47
|
+
euro: "€", pound: "£", yen: "¥", cent: "¢", larr: "←", rarr: "→", uarr: "↑", darr: "↓", harr: "↔", rArr: "⇒", lArr: "⇐",
|
|
48
|
+
le: "≤", ge: "≥", ne: "≠", asymp: "≈", infin: "∞", micro: "µ", frac12: "½", frac14: "¼", frac34: "¾", sup2: "²", sup3: "³",
|
|
49
|
+
zwj: "", zwnj: "", lrm: "", rlm: "", check: "✓", star: "☆", hearts: "♥",
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/** Decode HTML character references (named ones pages commonly use, and every numeric one). */
|
|
53
|
+
export function decodeEntities(text: string): string {
|
|
54
|
+
if (!text.includes("&")) return text;
|
|
55
|
+
return text.replace(/&(#x[0-9a-f]+|#[0-9]+|[a-z][a-z0-9]*);?/gi, (match, ref: string) => {
|
|
56
|
+
if (ref[0] === "#") {
|
|
57
|
+
const code = ref[1] === "x" || ref[1] === "X" ? Number.parseInt(ref.slice(2), 16) : Number.parseInt(ref.slice(1), 10);
|
|
58
|
+
if (!Number.isFinite(code) || code <= 0 || code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) return "�";
|
|
59
|
+
return String.fromCodePoint(code);
|
|
60
|
+
}
|
|
61
|
+
const named = NAMED[ref] ?? NAMED[ref.toLowerCase()];
|
|
62
|
+
return named ?? match;
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function parseAttrs(source: string): Record<string, string> {
|
|
67
|
+
const attrs: Record<string, string> = {};
|
|
68
|
+
const pattern = /([^\s"'<>/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
|
|
69
|
+
for (const match of source.matchAll(pattern)) {
|
|
70
|
+
const name = match[1]!.toLowerCase();
|
|
71
|
+
if (name in attrs) continue;
|
|
72
|
+
attrs[name] = decodeEntities(match[2] ?? match[3] ?? match[4] ?? "");
|
|
73
|
+
}
|
|
74
|
+
return attrs;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Parse HTML into a tree under a synthetic root element named `#root`. */
|
|
78
|
+
export function parseHtml(html: string): HtmlElement {
|
|
79
|
+
const root: HtmlElement = { type: "element", name: "#root", attrs: {}, children: [] };
|
|
80
|
+
const stack: HtmlElement[] = [root];
|
|
81
|
+
const current = () => stack[stack.length - 1]!;
|
|
82
|
+
const openIndex = (names: string[], within: string[]): number => {
|
|
83
|
+
for (let index = stack.length - 1; index > 0; index -= 1) {
|
|
84
|
+
const name = stack[index]!.name;
|
|
85
|
+
if (names.includes(name)) return index;
|
|
86
|
+
if (within.includes(name)) return -1;
|
|
87
|
+
}
|
|
88
|
+
return -1;
|
|
89
|
+
};
|
|
90
|
+
const text = (value: string) => {
|
|
91
|
+
if (!value) return;
|
|
92
|
+
const parent = current();
|
|
93
|
+
const last = parent.children[parent.children.length - 1];
|
|
94
|
+
if (last?.type === "text") last.text += value;
|
|
95
|
+
else parent.children.push({ type: "text", text: value });
|
|
96
|
+
};
|
|
97
|
+
|
|
98
|
+
let position = 0;
|
|
99
|
+
const tag = /<(\/?)([a-zA-Z][a-zA-Z0-9:-]*)((?:[^>"']|"[^"]*"|'[^']*')*?)(\/?)>|<!--[\s\S]*?(?:-->|$)|<![^>]*>|<\?[^>]*>/g;
|
|
100
|
+
for (;;) {
|
|
101
|
+
tag.lastIndex = position;
|
|
102
|
+
const match = tag.exec(html);
|
|
103
|
+
if (!match) {
|
|
104
|
+
text(decodeEntities(html.slice(position)));
|
|
105
|
+
break;
|
|
106
|
+
}
|
|
107
|
+
text(decodeEntities(html.slice(position, match.index)));
|
|
108
|
+
position = match.index + match[0].length;
|
|
109
|
+
if (!match[2]) continue; // comment, doctype, processing instruction
|
|
110
|
+
const name = match[2].toLowerCase();
|
|
111
|
+
if (match[1]) {
|
|
112
|
+
// An end tag closes the nearest open element of that name; a stray one is ignored.
|
|
113
|
+
for (let index = stack.length - 1; index > 0; index -= 1) {
|
|
114
|
+
if (stack[index]!.name === name) {
|
|
115
|
+
stack.length = index;
|
|
116
|
+
break;
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
const implied = IMPLIED[name];
|
|
122
|
+
if (implied) {
|
|
123
|
+
const index = openIndex(implied.closes, implied.within);
|
|
124
|
+
if (index > 0) stack.length = index;
|
|
125
|
+
}
|
|
126
|
+
if (CLOSES_P.has(name)) {
|
|
127
|
+
const index = openIndex(["p"], ["div", "section", "article", "main", "body", "td", "th", "li", "blockquote", "button"]);
|
|
128
|
+
if (index > 0) stack.length = index;
|
|
129
|
+
}
|
|
130
|
+
const element: HtmlElement = { type: "element", name, attrs: parseAttrs(match[3] ?? ""), children: [] };
|
|
131
|
+
current().children.push(element);
|
|
132
|
+
if (RAW.has(name)) {
|
|
133
|
+
const end = html.toLowerCase().indexOf(`</${name}`, position);
|
|
134
|
+
const stop = end < 0 ? html.length : end;
|
|
135
|
+
const raw = html.slice(position, stop);
|
|
136
|
+
if (raw) element.children.push({ type: "text", text: name === "title" || name === "textarea" ? decodeEntities(raw) : raw });
|
|
137
|
+
const close = end < 0 ? html.length : html.indexOf(">", end);
|
|
138
|
+
position = close < 0 ? html.length : close + 1;
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
if (!VOID.has(name) && !match[4]) stack.push(element);
|
|
142
|
+
}
|
|
143
|
+
return root;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** Every element under `node` (depth first, document order) that passes `test`. */
|
|
147
|
+
export function findAll(node: HtmlElement, test: (element: HtmlElement) => boolean): HtmlElement[] {
|
|
148
|
+
const found: HtmlElement[] = [];
|
|
149
|
+
const walk = (element: HtmlElement) => {
|
|
150
|
+
for (const child of element.children) {
|
|
151
|
+
if (child.type !== "element") continue;
|
|
152
|
+
if (test(child)) found.push(child);
|
|
153
|
+
walk(child);
|
|
154
|
+
}
|
|
155
|
+
};
|
|
156
|
+
walk(node);
|
|
157
|
+
return found;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export function findFirst(node: HtmlElement, test: (element: HtmlElement) => boolean): HtmlElement | undefined {
|
|
161
|
+
for (const child of node.children) {
|
|
162
|
+
if (child.type !== "element") continue;
|
|
163
|
+
if (test(child)) return child;
|
|
164
|
+
const found = findFirst(child, test);
|
|
165
|
+
if (found) return found;
|
|
166
|
+
}
|
|
167
|
+
return undefined;
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/** The text inside a node, whitespace collapsed. */
|
|
171
|
+
export function textOf(node: HtmlNode): string {
|
|
172
|
+
const parts: string[] = [];
|
|
173
|
+
const walk = (current: HtmlNode) => {
|
|
174
|
+
if (current.type === "text") parts.push(current.text);
|
|
175
|
+
else if (current.name !== "script" && current.name !== "style") for (const child of current.children) walk(child);
|
|
176
|
+
};
|
|
177
|
+
walk(node);
|
|
178
|
+
return parts.join("").replace(/\s+/g, " ").trim();
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
export function hasClass(element: HtmlElement, name: string): boolean {
|
|
182
|
+
return (element.attrs.class ?? "").split(/\s+/).includes(name);
|
|
183
|
+
}
|
package/src/web/read.ts
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reading pages and checking sources: a fetched page as readable text with
|
|
3
|
+
* its title and dates (kept a few minutes, so reading on from an offset does
|
|
4
|
+
* not fetch it again), and a URL's standing as a citation (reachable, where
|
|
5
|
+
* it ends up, when it was published).
|
|
6
|
+
*/
|
|
7
|
+
import { fetchPage, type FetchOptions } from "./fetch.ts";
|
|
8
|
+
import { htmlToMarkdown, pageMeta, type PageMeta } from "./extract.ts";
|
|
9
|
+
import { parseHtml } from "./html.ts";
|
|
10
|
+
|
|
11
|
+
export interface ReadPage extends PageMeta {
|
|
12
|
+
url: string;
|
|
13
|
+
/** The URL asked for, when a redirect moved it. */
|
|
14
|
+
requested?: string;
|
|
15
|
+
status: number;
|
|
16
|
+
contentType: string;
|
|
17
|
+
/** The page as Markdown or text. */
|
|
18
|
+
text: string;
|
|
19
|
+
/** The body was longer than was read. */
|
|
20
|
+
cut: boolean;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export interface SourceStatus extends PageMeta {
|
|
24
|
+
url: string;
|
|
25
|
+
ok: boolean;
|
|
26
|
+
status?: number;
|
|
27
|
+
statusText?: string;
|
|
28
|
+
finalUrl?: string;
|
|
29
|
+
contentType?: string;
|
|
30
|
+
/** The server's Last-Modified header, when it sends one. */
|
|
31
|
+
lastModified?: string;
|
|
32
|
+
error?: string;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const CACHE_MS = 10 * 60 * 1000;
|
|
36
|
+
const CACHE_SIZE = 40;
|
|
37
|
+
const cache = new Map<string, { page: ReadPage; at: number }>();
|
|
38
|
+
|
|
39
|
+
export function clearPageCache(): void {
|
|
40
|
+
cache.clear();
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const HTML = /html|xml/i;
|
|
44
|
+
|
|
45
|
+
function looksLikeHtml(contentType: string, text: string): boolean {
|
|
46
|
+
if (/xhtml|html/i.test(contentType)) return true;
|
|
47
|
+
if (contentType.trim()) return false;
|
|
48
|
+
return /^\s*(<!doctype html|<html|<head|<body)/i.test(text);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function kilobytes(bytes: number): string {
|
|
52
|
+
return bytes >= 1024 * 1024 ? `${(bytes / 1024 / 1024).toFixed(1)} MB` : `${Math.max(1, Math.round(bytes / 1024))} KB`;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** A page as text: HTML to Markdown, JSON pretty, other text as it is. Binary files are refused. */
|
|
56
|
+
export async function readPage(url: string, options: FetchOptions = {}): Promise<ReadPage> {
|
|
57
|
+
const key = url.trim();
|
|
58
|
+
const hit = cache.get(key);
|
|
59
|
+
if (hit && Date.now() - hit.at < CACHE_MS) return hit.page;
|
|
60
|
+
const fetched = await fetchPage(key, options);
|
|
61
|
+
const type = fetched.contentType.split(";")[0]!.trim().toLowerCase();
|
|
62
|
+
if (fetched.binary) {
|
|
63
|
+
throw new Error(`${fetched.url} is ${type || "a binary file"}${fetched.bytes ? ` (${kilobytes(fetched.bytes)})` : ""}; only web pages and text can be read${type === "application/pdf" ? ". Look for an HTML version of it (an abstract page, docs, a release note)" : ""}`);
|
|
64
|
+
}
|
|
65
|
+
let page: ReadPage;
|
|
66
|
+
const base = { url: fetched.url, ...(fetched.redirects.length > 0 ? { requested: key } : {}), status: fetched.status, contentType: type, cut: fetched.truncated };
|
|
67
|
+
if (looksLikeHtml(type, fetched.text)) {
|
|
68
|
+
const readable = htmlToMarkdown(fetched.text, fetched.url);
|
|
69
|
+
const { markdown, ...meta } = readable;
|
|
70
|
+
page = { ...meta, ...base, text: markdown || readable.description || "" };
|
|
71
|
+
} else if (/json/.test(type)) {
|
|
72
|
+
let text = fetched.text;
|
|
73
|
+
try {
|
|
74
|
+
text = JSON.stringify(JSON.parse(fetched.text), null, 2);
|
|
75
|
+
} catch {
|
|
76
|
+
// Not valid JSON after all: shown as it came.
|
|
77
|
+
}
|
|
78
|
+
page = { ...base, text: `\`\`\`json\n${text}\n\`\`\`` };
|
|
79
|
+
} else {
|
|
80
|
+
page = { ...base, text: fetched.text.replace(/\r\n?/g, "\n") };
|
|
81
|
+
}
|
|
82
|
+
if (!page.modified) {
|
|
83
|
+
const lastModified = fetched.headers.get("last-modified");
|
|
84
|
+
if (lastModified && !Number.isNaN(Date.parse(lastModified))) page.modified = new Date(lastModified).toISOString().slice(0, 10);
|
|
85
|
+
}
|
|
86
|
+
if (fetched.status < 400) {
|
|
87
|
+
cache.set(key, { page, at: Date.now() });
|
|
88
|
+
if (cache.size > CACHE_SIZE) cache.delete(cache.keys().next().value!);
|
|
89
|
+
}
|
|
90
|
+
return page;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** How each source stands: reachable or not, where it ends up, its title and dates. */
|
|
94
|
+
export async function checkSource(url: string, options: FetchOptions = {}): Promise<SourceStatus> {
|
|
95
|
+
try {
|
|
96
|
+
const fetched = await fetchPage(url, { ...options, maxBytes: 512 * 1024 });
|
|
97
|
+
const type = fetched.contentType.split(";")[0]!.trim().toLowerCase();
|
|
98
|
+
const meta = HTML.test(type) || looksLikeHtml(type, fetched.text) ? pageMeta(parseHtml(fetched.text)) : {};
|
|
99
|
+
const lastModified = fetched.headers.get("last-modified") ?? undefined;
|
|
100
|
+
return {
|
|
101
|
+
url,
|
|
102
|
+
ok: fetched.status < 400,
|
|
103
|
+
status: fetched.status,
|
|
104
|
+
statusText: fetched.statusText,
|
|
105
|
+
...(fetched.redirects.length > 0 ? { finalUrl: fetched.url } : {}),
|
|
106
|
+
contentType: type,
|
|
107
|
+
...(lastModified ? { lastModified } : {}),
|
|
108
|
+
...meta,
|
|
109
|
+
};
|
|
110
|
+
} catch (error) {
|
|
111
|
+
return { url, ok: false, error: error instanceof Error ? error.message : String(error) };
|
|
112
|
+
}
|
|
113
|
+
}
|