@a-t-h-i/bot-lobby 0.6.2 → 0.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +112 -11
- package/package.json +1 -4
- package/prompts/master.md +47 -1
- package/prompts/researcher.md +8 -2
- package/prompts/reviewer.md +42 -0
- package/prompts/worker.md +7 -0
- package/src/ask/dialog.ts +167 -0
- package/src/ask/image.ts +202 -0
- package/src/ask/png.ts +179 -0
- package/src/ask/relay.ts +89 -0
- package/src/ask/state.ts +160 -0
- package/src/ask/tool.ts +126 -0
- package/src/ask/types.ts +55 -0
- package/src/ask/view.ts +159 -0
- package/src/execution/agent-runner.ts +103 -11
- package/src/execution/git.ts +111 -14
- package/src/execution/pi-runner.ts +164 -11
- package/src/index.ts +9 -0
- package/src/lobby/ask.ts +10 -120
- package/src/lobby/feed.ts +76 -8
- package/src/lobby/layout.ts +38 -18
- package/src/lobby/markdown.ts +92 -19
- package/src/lobby/planner.ts +1 -1
- package/src/lobby/quickfix.ts +21 -0
- package/src/lobby/runtime.ts +51 -37
- package/src/lobby/session-files.ts +162 -25
- package/src/lobby/sessions.ts +21 -7
- package/src/lobby/tabs/home.ts +289 -67
- package/src/lobby/tabs/issues.ts +4 -3
- package/src/lobby/tabs/plan.ts +4 -4
- package/src/lobby/tabs/quickfix.ts +7 -1
- package/src/lobby/tabs/tasks.ts +14 -4
- package/src/lobby/theme.ts +30 -0
- package/src/lobby/view.ts +141 -25
- package/src/master/decisions.ts +1 -1
- package/src/master/master.ts +41 -4
- package/src/master/research.ts +5 -2
- package/src/pi/commands.ts +94 -10
- package/src/pi/events.ts +54 -12
- package/src/pi/fresh-context.ts +134 -0
- package/src/pi/owner.ts +19 -10
- package/src/pi/quiet.ts +22 -4
- package/src/pi/start-task.ts +11 -2
- package/src/pi/tools.ts +26 -8
- package/src/pi/ui.ts +9 -3
- package/src/pi/zen-large.ts +10 -10
- package/src/pi/zen-metrics.ts +13 -3
- package/src/pi/zen.ts +22 -15
- package/src/roles/reviewer.ts +23 -4
- package/src/roles/worker.ts +18 -0
- package/src/schemas/configuration.ts +8 -0
- package/src/schemas/findings.ts +14 -0
- package/src/schemas/task.ts +21 -0
- package/src/state/archive.ts +12 -3
- package/src/state/backlog.ts +13 -3
- package/src/state/budget.ts +274 -0
- package/src/state/changes.ts +231 -0
- package/src/state/file-cache.ts +62 -0
- package/src/state/metrics.ts +63 -13
- package/src/state/persistence.ts +36 -2
- package/src/text.ts +28 -2
- package/src/web/extract.ts +332 -0
- package/src/web/fetch.ts +232 -0
- package/src/web/html.ts +183 -0
- package/src/web/read.ts +113 -0
- package/src/web/search.ts +202 -0
- package/src/web/tools.ts +279 -0
- package/src/width.ts +102 -0
- package/src/workflow/workflow.ts +592 -42
package/src/state/metrics.ts
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* thinking level, how often it succeeds and what it costs. Append-only JSON
|
|
6
6
|
* lines per project; reads keep the newest `MAX_READ` records.
|
|
7
7
|
*/
|
|
8
|
-
import { appendFileSync,
|
|
8
|
+
import { appendFileSync, closeSync, mkdirSync, openSync, readSync, statSync } from "node:fs";
|
|
9
9
|
import { dirname, join } from "node:path";
|
|
10
10
|
import type { AgentRun } from "../schemas/findings.ts";
|
|
11
11
|
import type { RunLogEntry, Task } from "../schemas/task.ts";
|
|
@@ -119,6 +119,20 @@ function isRecord(value: unknown): value is MetricRecord {
|
|
|
119
119
|
return Boolean(record && typeof record.id === "string" && typeof record.kind === "string" && typeof record.durationMs === "number");
|
|
120
120
|
}
|
|
121
121
|
|
|
122
|
+
/** How much of a long log's end the first read takes: far more than `MAX_READ` records need. */
|
|
123
|
+
const TAIL_BYTES = 4 * 1024 * 1024;
|
|
124
|
+
|
|
125
|
+
/** A metrics log as followed so far: the file it was, how far it was read, and its newest records. */
|
|
126
|
+
interface Followed {
|
|
127
|
+
ino: bigint;
|
|
128
|
+
offset: number;
|
|
129
|
+
records: MetricRecord[];
|
|
130
|
+
/** The first read starts inside the file, part way through a line. */
|
|
131
|
+
midLine: boolean;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
const followed = new Map<string, Followed>();
|
|
135
|
+
|
|
122
136
|
/** Agent runs for the Metrics tab; classifier calls, a few hundred milliseconds each, are read apart. */
|
|
123
137
|
export function readMetrics(root: string, configDir: string, limit = MAX_READ): MetricRecord[] {
|
|
124
138
|
return readRecords(root, configDir, limit).filter((record) => record.kind !== "classifier");
|
|
@@ -129,26 +143,62 @@ export function readClassifierMetrics(root: string, configDir: string, limit = M
|
|
|
129
143
|
return readRecords(root, configDir, limit).filter((record) => record.kind === "classifier");
|
|
130
144
|
}
|
|
131
145
|
|
|
146
|
+
/**
|
|
147
|
+
* The newest records of the project's log, agent runs and classifier calls
|
|
148
|
+
* alike. The log only grows, so after the first read (of its end only) each
|
|
149
|
+
* read parses just the lines appended since; a log that shrank or was
|
|
150
|
+
* replaced is read afresh.
|
|
151
|
+
*/
|
|
132
152
|
function readRecords(root: string, configDir: string, limit: number): MetricRecord[] {
|
|
133
153
|
const path = metricsPath(root, configDir);
|
|
134
|
-
|
|
135
|
-
let
|
|
154
|
+
let ino: bigint;
|
|
155
|
+
let size: number;
|
|
136
156
|
try {
|
|
137
|
-
|
|
157
|
+
const stat = statSync(path, { bigint: true });
|
|
158
|
+
ino = stat.ino;
|
|
159
|
+
size = Number(stat.size);
|
|
138
160
|
} catch {
|
|
161
|
+
followed.delete(path);
|
|
139
162
|
return [];
|
|
140
163
|
}
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
164
|
+
let log = followed.get(path);
|
|
165
|
+
if (!log || log.ino !== ino || size < log.offset) {
|
|
166
|
+
const offset = Math.max(0, size - TAIL_BYTES);
|
|
167
|
+
log = { ino, offset, records: [], midLine: offset > 0 };
|
|
168
|
+
followed.set(path, log);
|
|
169
|
+
}
|
|
170
|
+
if (size > log.offset) readAppended(path, log, size);
|
|
171
|
+
return log.records.length > limit ? log.records.slice(-limit) : [...log.records];
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
function readAppended(path: string, log: Followed, size: number): void {
|
|
175
|
+
let fd: number | undefined;
|
|
176
|
+
try {
|
|
177
|
+
fd = openSync(path, "r");
|
|
178
|
+
const buffer = Buffer.alloc(size - log.offset);
|
|
179
|
+
const read = readSync(fd, buffer, 0, buffer.length, log.offset);
|
|
180
|
+
// Whole lines only: one still being appended is read next time.
|
|
181
|
+
const end = buffer.lastIndexOf(0x0a, read - 1);
|
|
182
|
+
if (end < 0) return;
|
|
183
|
+
const lines = buffer.toString("utf8", 0, end).split("\n");
|
|
184
|
+
if (log.midLine) lines.shift();
|
|
185
|
+
log.midLine = false;
|
|
186
|
+
for (const line of lines) {
|
|
187
|
+
if (!line.trim()) continue;
|
|
188
|
+
try {
|
|
189
|
+
const value = JSON.parse(line) as unknown;
|
|
190
|
+
if (isRecord(value)) log.records.push(value);
|
|
191
|
+
} catch {
|
|
192
|
+
// Torn lines are skipped.
|
|
193
|
+
}
|
|
149
194
|
}
|
|
195
|
+
if (log.records.length > MAX_READ) log.records.splice(0, log.records.length - MAX_READ);
|
|
196
|
+
log.offset += end + 1;
|
|
197
|
+
} catch {
|
|
198
|
+
// Unreadable for now; the next read tries again.
|
|
199
|
+
} finally {
|
|
200
|
+
if (fd !== undefined) closeSync(fd);
|
|
150
201
|
}
|
|
151
|
-
return records.slice(-limit);
|
|
152
202
|
}
|
|
153
203
|
|
|
154
204
|
const AGENT_NAMES: Record<string, string> = { backend: "DEV", designer: "DESIGN", qa: "QA" };
|
package/src/state/persistence.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { existsSync, mkdirSync, readdirSync, readFileSync, rmSync } from "node:fs";
|
|
2
|
-
import { join } from "node:path";
|
|
2
|
+
import { join, sep } from "node:path";
|
|
3
3
|
import type { KnowledgeConfig } from "../schemas/configuration.ts";
|
|
4
4
|
import type { Domain } from "../schemas/agent.ts";
|
|
5
5
|
import { TERMINAL_STATES, isTaskState, type Task } from "../schemas/task.ts";
|
|
@@ -14,6 +14,7 @@ import {
|
|
|
14
14
|
type KnowledgeAgent,
|
|
15
15
|
} from "../knowledge/paths.ts";
|
|
16
16
|
import { DEFAULT_KNOWLEDGE_CONTENT, ensureFile, readFileOr, writeFileEnsured } from "../knowledge/store.ts";
|
|
17
|
+
import { forgetCached, forgetCachedUnder, readJsonCached } from "./file-cache.ts";
|
|
17
18
|
|
|
18
19
|
/** Idempotently create the full knowledge + tasks layout with seed files. */
|
|
19
20
|
export function ensureProjectStructure(root: string, configDir: string): void {
|
|
@@ -36,6 +37,7 @@ export function ensureProjectStructure(root: string, configDir: string): void {
|
|
|
36
37
|
export function createTaskDir(root: string, configDir: string, task: Task): void {
|
|
37
38
|
const dir = taskDir(dataRoot(root, configDir), task.id);
|
|
38
39
|
mkdirSync(dir, { recursive: true });
|
|
40
|
+
forgetCached(join(dir, "state.json"));
|
|
39
41
|
writeFileEnsured(join(dir, "state.json"), JSON.stringify(task, null, 2));
|
|
40
42
|
ensureFile(join(dir, "proposal.md"), "");
|
|
41
43
|
ensureFile(join(dir, "plan.md"), "");
|
|
@@ -77,7 +79,9 @@ export function readTaskArtifact(
|
|
|
77
79
|
}
|
|
78
80
|
|
|
79
81
|
export function saveTask(root: string, configDir: string, task: Task): void {
|
|
80
|
-
|
|
82
|
+
const path = join(taskDir(dataRoot(root, configDir), task.id), "state.json");
|
|
83
|
+
forgetCached(path);
|
|
84
|
+
writeFileEnsured(path, JSON.stringify(task, null, 2));
|
|
81
85
|
}
|
|
82
86
|
|
|
83
87
|
/** Read a task state: the bot-lobby copy wins, else the newest pre-rename copy. */
|
|
@@ -147,6 +151,36 @@ export function listTasks(root: string, configDir: string): Task[] {
|
|
|
147
151
|
return tasks.sort((a, b) => b.updatedAt.localeCompare(a.updatedAt));
|
|
148
152
|
}
|
|
149
153
|
|
|
154
|
+
/** Any parsed JSON object counts as a task, as `readTaskAt` has it. */
|
|
155
|
+
function isObject(value: unknown): value is Task {
|
|
156
|
+
return typeof value === "object" && value !== null;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* All tasks on disk, newest first, like `listTasks`, but a file is read only
|
|
161
|
+
* when it changed since the last look. The tasks are shared with every other
|
|
162
|
+
* caller, so they are for reading (the owner's clock, the lobby): whatever
|
|
163
|
+
* changes a task loads its own copy with `loadTask` and saves that.
|
|
164
|
+
*/
|
|
165
|
+
export function peekTasks(root: string, configDir: string): Task[] {
|
|
166
|
+
const tasks: Task[] = [];
|
|
167
|
+
const seen = new Set<string>();
|
|
168
|
+
for (const { dir } of taskEntries(root, configDir)) {
|
|
169
|
+
const path = join(dir, "state.json");
|
|
170
|
+
seen.add(path);
|
|
171
|
+
const task = readJsonCached(path, isObject);
|
|
172
|
+
if (task) tasks.push(task);
|
|
173
|
+
}
|
|
174
|
+
// Forget tasks that left these folders (deleted or archived).
|
|
175
|
+
for (const dr of readDataRoots(root, configDir)) forgetCachedUnder(`${tasksRoot(dr)}${sep}`, seen);
|
|
176
|
+
return tasks.sort((a, b) => b.updatedAt.localeCompare(a.updatedAt));
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/** The non-terminal task a session owns, from `peekTasks` (for reading only). */
|
|
180
|
+
export function peekOwnedTask(root: string, configDir: string, sessionId: string): Task | undefined {
|
|
181
|
+
return peekTasks(root, configDir).find((task) => !TERMINAL_STATES.includes(task.state) && task.ownerSessionId === sessionId);
|
|
182
|
+
}
|
|
183
|
+
|
|
150
184
|
/** Tasks on disk plus the ids whose state.json could not be read (§59). */
|
|
151
185
|
export function taskHealth(root: string, configDir: string): { tasks: Task[]; corrupted: string[] } {
|
|
152
186
|
const tasks: Task[] = [];
|
package/src/text.ts
CHANGED
|
@@ -5,6 +5,13 @@ export function truncate(text: string, maxChars: number): string {
|
|
|
5
5
|
return `${text.slice(0, maxChars)}\n[...${text.length - maxChars} characters omitted]`;
|
|
6
6
|
}
|
|
7
7
|
|
|
8
|
+
/** The end of `text` within a character budget (the newest entries of a log), marking how much was left out before it. */
|
|
9
|
+
export function tail(text: string, maxChars: number): string {
|
|
10
|
+
if (maxChars <= 0) return "";
|
|
11
|
+
if (text.length <= maxChars) return text;
|
|
12
|
+
return `[...${text.length - maxChars} earlier characters omitted]\n${text.slice(text.length - maxChars)}`;
|
|
13
|
+
}
|
|
14
|
+
|
|
8
15
|
/** Filler and imperative words that carry no topic signal in a request. */
|
|
9
16
|
const FILLER_WORDS = new Set([
|
|
10
17
|
"let's",
|
|
@@ -45,9 +52,28 @@ function fillerKey(word: string): string {
|
|
|
45
52
|
*/
|
|
46
53
|
export function shortTitle(request: string, maxWords = 3): string {
|
|
47
54
|
if (maxWords <= 0) return "";
|
|
48
|
-
const words = request
|
|
55
|
+
const words = titleText(request).split(/\s+/).filter(Boolean);
|
|
49
56
|
const content = words.filter((word) => !FILLER_WORDS.has(fillerKey(word)));
|
|
50
|
-
return (content.length > 0 ? content : words).slice(0, maxWords).join(" ");
|
|
57
|
+
return (content.length > 0 ? content : words).slice(0, maxWords).join(" ").replace(/[:;,]+$/, "");
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** Headings that only label a section (`### Objective`), not name the work. */
|
|
61
|
+
const LABEL_HEADING = /^#{1,6}\s*(objective|goal|goals|summary|overview|task|request|context|background|description|plan)\s*:?\s*$/i;
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* A request as words for a title: escaped line breaks (`\\n` written out by a
|
|
65
|
+
* model) read as breaks, label headings are skipped, and Markdown marks
|
|
66
|
+
* (`#`, `-`, `*`, `>`, backticks) are dropped.
|
|
67
|
+
*/
|
|
68
|
+
function titleText(request: string): string {
|
|
69
|
+
return request
|
|
70
|
+
.replace(/\\[nrt]/g, "\n")
|
|
71
|
+
.split("\n")
|
|
72
|
+
.filter((line) => !LABEL_HEADING.test(line.trim()))
|
|
73
|
+
.join(" ")
|
|
74
|
+
.replace(/[#*>`|]+/g, " ")
|
|
75
|
+
.replace(/(^|\s)[-+](?=\s|$)/g, " ")
|
|
76
|
+
.trim();
|
|
51
77
|
}
|
|
52
78
|
|
|
53
79
|
/** Compact duration such as "45s", "3m" or "2m 05s"; a non-finite input reads "0s". */
|
|
@@ -0,0 +1,332 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A web page as a model reads it best: its title and dates, then the main
|
|
3
|
+
* content as Markdown (headings, paragraphs, lists, tables, code, links made
|
|
4
|
+
* absolute), with navigation, scripts, forms, cookie banners and the like
|
|
5
|
+
* left out.
|
|
6
|
+
*/
|
|
7
|
+
import { findAll, findFirst, parseHtml, textOf, type HtmlElement, type HtmlNode } from "./html.ts";
|
|
8
|
+
|
|
9
|
+
export interface PageMeta {
|
|
10
|
+
title?: string;
|
|
11
|
+
description?: string;
|
|
12
|
+
siteName?: string;
|
|
13
|
+
published?: string;
|
|
14
|
+
modified?: string;
|
|
15
|
+
lang?: string;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export interface ReadablePage extends PageMeta {
|
|
19
|
+
markdown: string;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Elements never worth reading. */
|
|
23
|
+
const DROP = new Set(["script", "style", "noscript", "template", "svg", "canvas", "iframe", "object", "embed", "video", "audio", "map", "head", "link", "meta", "button", "input", "select", "textarea", "form", "dialog", "nav"]);
|
|
24
|
+
/** Page chrome, dropped unless it is all there is. */
|
|
25
|
+
const CHROME = new Set(["aside", "footer"]);
|
|
26
|
+
/** Class or id words that mark page chrome rather than content. */
|
|
27
|
+
const CHROME_HINT = /(?:^|[\s_-])(cookie|consent|gdpr|banner|newsletter|subscribe|signup|sidebar|share|sharing|social|advert|ads?|promo|popup|modal|breadcrumbs?|related|comments?|skip-link|toc-mobile)(?:$|[\s_-])/i;
|
|
28
|
+
const BLOCK = new Set(["address", "article", "aside", "blockquote", "body", "center", "dd", "details", "dialog", "div", "dl", "dt", "fieldset", "figcaption", "figure", "footer", "h1", "h2", "h3", "h4", "h5", "h6", "header", "hgroup", "hr", "html", "li", "main", "ol", "p", "pre", "section", "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", "#root"]);
|
|
29
|
+
|
|
30
|
+
function isChrome(element: HtmlElement): boolean {
|
|
31
|
+
if (element.attrs.hidden !== undefined || element.attrs["aria-hidden"] === "true") return true;
|
|
32
|
+
if (element.attrs.role === "navigation" || element.attrs.role === "banner" || element.attrs.role === "contentinfo" || element.attrs.role === "dialog") return true;
|
|
33
|
+
const style = element.attrs.style ?? "";
|
|
34
|
+
if (/display\s*:\s*none|visibility\s*:\s*hidden/i.test(style)) return true;
|
|
35
|
+
if (!["div", "section", "aside", "ul", "span", "p", "header", "footer"].includes(element.name)) return false;
|
|
36
|
+
return CHROME_HINT.test(`${element.attrs.class ?? ""} ${element.attrs.id ?? ""}`);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function metaContent(root: HtmlElement, ...keys: string[]): string | undefined {
|
|
40
|
+
const metas = findAll(root, (element) => element.name === "meta");
|
|
41
|
+
for (const key of keys) {
|
|
42
|
+
const meta = metas.find((element) => (element.attrs.property ?? element.attrs.name ?? element.attrs.itemprop ?? "").toLowerCase() === key);
|
|
43
|
+
const content = meta?.attrs.content?.trim();
|
|
44
|
+
if (content) return content;
|
|
45
|
+
}
|
|
46
|
+
return undefined;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** A date as it is written in structured data (JSON-LD), when the page has one. */
|
|
50
|
+
function jsonLdDate(root: HtmlElement, key: "datePublished" | "dateModified"): string | undefined {
|
|
51
|
+
for (const script of findAll(root, (element) => element.name === "script" && (element.attrs.type ?? "").includes("ld+json"))) {
|
|
52
|
+
const source = script.children.map((child) => (child.type === "text" ? child.text : "")).join("");
|
|
53
|
+
const match = new RegExp(`"${key}"\\s*:\\s*"([^"]+)"`).exec(source);
|
|
54
|
+
if (match) return match[1];
|
|
55
|
+
}
|
|
56
|
+
return undefined;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** A date trimmed to what a citation needs: the day (ISO), or the text as the page gives it. */
|
|
60
|
+
function day(value: string | undefined): string | undefined {
|
|
61
|
+
if (!value) return undefined;
|
|
62
|
+
const iso = /^(\d{4}-\d{2}-\d{2})/.exec(value.trim());
|
|
63
|
+
if (iso) return iso[1];
|
|
64
|
+
const parsed = Date.parse(value);
|
|
65
|
+
return Number.isNaN(parsed) ? value.trim().slice(0, 40) : new Date(parsed).toISOString().slice(0, 10);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function pageMeta(root: HtmlElement): PageMeta {
|
|
69
|
+
const titleElement = findFirst(root, (element) => element.name === "title");
|
|
70
|
+
const title = metaContent(root, "og:title", "twitter:title") ?? (titleElement ? textOf(titleElement) : undefined);
|
|
71
|
+
const time = findFirst(root, (element) => element.name === "time" && Boolean(element.attrs.datetime));
|
|
72
|
+
const html = findFirst(root, (element) => element.name === "html");
|
|
73
|
+
const meta: PageMeta = {
|
|
74
|
+
title: title?.replace(/\s+/g, " ").trim() || undefined,
|
|
75
|
+
description: metaContent(root, "description", "og:description", "twitter:description"),
|
|
76
|
+
siteName: metaContent(root, "og:site_name", "application-name"),
|
|
77
|
+
published: day(metaContent(root, "article:published_time", "datepublished", "date", "dc.date", "dc.date.issued", "dcterms.created", "pubdate", "citation_publication_date", "citation_date", "sailthru.date") ?? jsonLdDate(root, "datePublished") ?? time?.attrs.datetime),
|
|
78
|
+
modified: day(metaContent(root, "article:modified_time", "og:updated_time", "datemodified", "last-modified", "dcterms.modified") ?? jsonLdDate(root, "dateModified")),
|
|
79
|
+
lang: html?.attrs.lang,
|
|
80
|
+
};
|
|
81
|
+
return Object.fromEntries(Object.entries(meta).filter(([, value]) => value !== undefined)) as PageMeta;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** The part of the page that is its content: main, the one article, role=main, or the body. */
|
|
85
|
+
function contentRoot(root: HtmlElement): HtmlElement {
|
|
86
|
+
const main = findFirst(root, (element) => element.name === "main" || element.attrs.role === "main");
|
|
87
|
+
if (main && textOf(main).length > 200) return main;
|
|
88
|
+
const articles = findAll(root, (element) => element.name === "article");
|
|
89
|
+
if (articles.length === 1 && textOf(articles[0]!).length > 200) return articles[0]!;
|
|
90
|
+
const content = findFirst(root, (element) => ["content", "main-content", "maincontent"].includes((element.attrs.id ?? "").toLowerCase()));
|
|
91
|
+
if (content && textOf(content).length > 200) return content;
|
|
92
|
+
return findFirst(root, (element) => element.name === "body") ?? root;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Drop what is never content; page chrome goes too unless nothing else is
|
|
97
|
+
* left. Reading the whole body, its header (logo, site menu) is chrome as well.
|
|
98
|
+
*/
|
|
99
|
+
function prune(node: HtmlElement, keepChrome: boolean, dropHeader: boolean): HtmlElement {
|
|
100
|
+
const children: HtmlNode[] = [];
|
|
101
|
+
for (const child of node.children) {
|
|
102
|
+
if (child.type === "text") children.push(child);
|
|
103
|
+
else if (DROP.has(child.name)) continue;
|
|
104
|
+
else if (!keepChrome && (CHROME.has(child.name) || isChrome(child) || (dropHeader && child.name === "header"))) continue;
|
|
105
|
+
else children.push(prune(child, keepChrome, dropHeader));
|
|
106
|
+
}
|
|
107
|
+
return { ...node, children };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
interface Context {
|
|
111
|
+
/** Where relative links point from. */
|
|
112
|
+
base: URL | undefined;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
function absolute(href: string | undefined, base: URL | undefined): string | undefined {
|
|
116
|
+
if (!href) return undefined;
|
|
117
|
+
const trimmed = href.trim();
|
|
118
|
+
if (!trimmed || trimmed.startsWith("#") || /^(javascript|mailto|tel|data):/i.test(trimmed)) return undefined;
|
|
119
|
+
try {
|
|
120
|
+
return new URL(trimmed, base).href;
|
|
121
|
+
} catch {
|
|
122
|
+
return undefined;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function fence(code: string): string {
|
|
127
|
+
const longest = Math.max(2, ...[...code.matchAll(/`+/g)].map((run) => run[0].length));
|
|
128
|
+
return "`".repeat(longest + 1);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function codeLanguage(element: HtmlElement): string {
|
|
132
|
+
const code = findFirst(element, (child) => child.name === "code");
|
|
133
|
+
const classes = `${element.attrs.class ?? ""} ${code?.attrs.class ?? ""} ${element.attrs["data-lang"] ?? ""}`;
|
|
134
|
+
const match = /(?:language|lang|highlight-source)-([a-z0-9+#-]+)/i.exec(classes) ?? /^\s*([a-z0-9+#-]+)\s*$/i.exec(element.attrs["data-lang"] ?? "");
|
|
135
|
+
return match?.[1]?.toLowerCase() ?? "";
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function rawText(node: HtmlNode): string {
|
|
139
|
+
if (node.type === "text") return node.text;
|
|
140
|
+
if (node.name === "br") return "\n";
|
|
141
|
+
return node.children.map(rawText).join("");
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
function wrapInline(text: string, mark: string): string {
|
|
145
|
+
const trimmed = text.trim();
|
|
146
|
+
if (!trimmed) return text;
|
|
147
|
+
const lead = text.startsWith(" ") ? " " : "";
|
|
148
|
+
const trail = text.endsWith(" ") ? " " : "";
|
|
149
|
+
return `${lead}${mark}${trimmed}${mark}${trail}`;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/** A node's inline Markdown: text with emphasis, code, links and line breaks. */
|
|
153
|
+
function inline(node: HtmlNode, context: Context): string {
|
|
154
|
+
if (node.type === "text") return node.text.replace(/\s+/g, " ");
|
|
155
|
+
const inner = () => node.children.map((child) => inline(child, context)).join("");
|
|
156
|
+
switch (node.name) {
|
|
157
|
+
case "br":
|
|
158
|
+
return "\n";
|
|
159
|
+
case "img": {
|
|
160
|
+
const alt = (node.attrs.alt ?? "").replace(/\s+/g, " ").trim();
|
|
161
|
+
const src = absolute(node.attrs.src, context.base);
|
|
162
|
+
return alt && src ? `` : alt;
|
|
163
|
+
}
|
|
164
|
+
case "a": {
|
|
165
|
+
const text = inner();
|
|
166
|
+
const href = absolute(node.attrs.href, context.base);
|
|
167
|
+
if (!text.trim()) return "";
|
|
168
|
+
return href ? `${text.startsWith(" ") ? " " : ""}[${text.trim()}](${href})${text.endsWith(" ") ? " " : ""}` : text;
|
|
169
|
+
}
|
|
170
|
+
case "strong":
|
|
171
|
+
case "b":
|
|
172
|
+
return wrapInline(inner(), "**");
|
|
173
|
+
case "em":
|
|
174
|
+
case "i":
|
|
175
|
+
case "cite":
|
|
176
|
+
return wrapInline(inner(), "*");
|
|
177
|
+
case "del":
|
|
178
|
+
case "s":
|
|
179
|
+
case "strike":
|
|
180
|
+
return wrapInline(inner(), "~~");
|
|
181
|
+
case "code":
|
|
182
|
+
case "kbd":
|
|
183
|
+
case "samp":
|
|
184
|
+
case "tt": {
|
|
185
|
+
const code = rawText(node).replace(/\s+/g, " ");
|
|
186
|
+
if (!code.trim()) return code;
|
|
187
|
+
const ticks = code.includes("`") ? "``" : "`";
|
|
188
|
+
return `${ticks}${ticks.length > 1 ? " " : ""}${code.trim()}${ticks.length > 1 ? " " : ""}${ticks}`;
|
|
189
|
+
}
|
|
190
|
+
case "sup":
|
|
191
|
+
return `^${inner().trim()}`;
|
|
192
|
+
default:
|
|
193
|
+
// A block inside inline content (a div in a link) reads as a spaced run.
|
|
194
|
+
return BLOCK.has(node.name) ? ` ${inner()} ` : inner();
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function paragraph(text: string): string {
|
|
199
|
+
return text
|
|
200
|
+
.split("\n")
|
|
201
|
+
.map((line) => line.replace(/[ \t]+/g, " ").trim())
|
|
202
|
+
.join("\n")
|
|
203
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
204
|
+
.trim();
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function indent(text: string, prefix: string, first = prefix): string {
|
|
208
|
+
return text
|
|
209
|
+
.split("\n")
|
|
210
|
+
.map((line, index) => (index === 0 ? first : line ? prefix : "") + line)
|
|
211
|
+
.join("\n");
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
function table(element: HtmlElement, context: Context): string {
|
|
215
|
+
const rows = findAll(element, (child) => child.name === "tr").map((row) =>
|
|
216
|
+
row.children
|
|
217
|
+
.filter((cell): cell is HtmlElement => cell.type === "element" && (cell.name === "td" || cell.name === "th"))
|
|
218
|
+
.map((cell) => paragraph(inline(cell, context)).replace(/\n+/g, " ").replace(/\|/g, "\\|")),
|
|
219
|
+
);
|
|
220
|
+
const filled = rows.filter((row) => row.some((cell) => cell));
|
|
221
|
+
if (filled.length === 0) return "";
|
|
222
|
+
const width = Math.max(...filled.map((row) => row.length));
|
|
223
|
+
const line = (cells: string[]) => `| ${[...cells, ...Array(width - cells.length).fill("")].join(" | ")} |`;
|
|
224
|
+
return [line(filled[0]!), `|${" --- |".repeat(width)}`, ...filled.slice(1).map(line)].join("\n");
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
function list(element: HtmlElement, context: Context): string {
|
|
228
|
+
const ordered = element.name === "ol";
|
|
229
|
+
let number = Number.parseInt(element.attrs.start ?? "1", 10) || 1;
|
|
230
|
+
const items: string[] = [];
|
|
231
|
+
for (const child of element.children) {
|
|
232
|
+
if (child.type === "text") {
|
|
233
|
+
if (child.text.trim()) items.push(`- ${child.text.trim()}`);
|
|
234
|
+
continue;
|
|
235
|
+
}
|
|
236
|
+
const body = child.name === "li" ? blocks(child, context).join("\n") : blocks({ ...child, children: [child] }, context).join("\n");
|
|
237
|
+
if (!body.trim()) continue;
|
|
238
|
+
const marker = ordered ? `${number++}. ` : "- ";
|
|
239
|
+
items.push(indent(body, " ".repeat(marker.length), marker));
|
|
240
|
+
}
|
|
241
|
+
return items.join("\n");
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/** A block element as Markdown blocks. */
|
|
245
|
+
function block(element: HtmlElement, context: Context): string[] {
|
|
246
|
+
switch (element.name) {
|
|
247
|
+
case "h1":
|
|
248
|
+
case "h2":
|
|
249
|
+
case "h3":
|
|
250
|
+
case "h4":
|
|
251
|
+
case "h5":
|
|
252
|
+
case "h6": {
|
|
253
|
+
const text = paragraph(inline(element, context)).replace(/\n+/g, " ");
|
|
254
|
+
return text ? [`${"#".repeat(Number(element.name[1]))} ${text}`] : [];
|
|
255
|
+
}
|
|
256
|
+
case "p":
|
|
257
|
+
case "summary":
|
|
258
|
+
case "figcaption":
|
|
259
|
+
case "dt": {
|
|
260
|
+
const text = paragraph(element.children.map((child) => inline(child, context)).join(""));
|
|
261
|
+
return text ? [element.name === "dt" ? `**${text}**` : text] : [];
|
|
262
|
+
}
|
|
263
|
+
case "pre": {
|
|
264
|
+
const code = rawText(element).replace(/^\n/, "").replace(/\s+$/, "");
|
|
265
|
+
if (!code.trim()) return [];
|
|
266
|
+
const marks = fence(code);
|
|
267
|
+
return [`${marks}${codeLanguage(element)}\n${code}\n${marks}`];
|
|
268
|
+
}
|
|
269
|
+
case "ul":
|
|
270
|
+
case "ol":
|
|
271
|
+
case "menu": {
|
|
272
|
+
const text = list(element, context);
|
|
273
|
+
return text ? [text] : [];
|
|
274
|
+
}
|
|
275
|
+
case "blockquote": {
|
|
276
|
+
const inner = blocks(element, context).join("\n\n");
|
|
277
|
+
return inner ? [indent(inner, "> ")] : [];
|
|
278
|
+
}
|
|
279
|
+
case "table": {
|
|
280
|
+
const text = table(element, context);
|
|
281
|
+
return text ? [text] : [];
|
|
282
|
+
}
|
|
283
|
+
case "hr":
|
|
284
|
+
return ["---"];
|
|
285
|
+
default:
|
|
286
|
+
return blocks(element, context);
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
/** The children of an element as Markdown blocks: inline runs become paragraphs. */
|
|
291
|
+
function blocks(element: HtmlElement, context: Context): string[] {
|
|
292
|
+
const out: string[] = [];
|
|
293
|
+
let run = "";
|
|
294
|
+
const flush = () => {
|
|
295
|
+
const text = paragraph(run);
|
|
296
|
+
if (text) out.push(text);
|
|
297
|
+
run = "";
|
|
298
|
+
};
|
|
299
|
+
for (const child of element.children) {
|
|
300
|
+
if (child.type === "element" && BLOCK.has(child.name)) {
|
|
301
|
+
flush();
|
|
302
|
+
out.push(...block(child, context));
|
|
303
|
+
} else {
|
|
304
|
+
run += inline(child, context);
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
flush();
|
|
308
|
+
return out;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/** An HTML page as its metadata plus its main content in Markdown. */
|
|
312
|
+
export function htmlToMarkdown(html: string, url?: string): ReadablePage {
|
|
313
|
+
const root = parseHtml(html);
|
|
314
|
+
const meta = pageMeta(root);
|
|
315
|
+
const baseHref = findFirst(root, (element) => element.name === "base")?.attrs.href;
|
|
316
|
+
let base: URL | undefined;
|
|
317
|
+
try {
|
|
318
|
+
base = url ? new URL(baseHref ?? "", url) : undefined;
|
|
319
|
+
} catch {
|
|
320
|
+
base = undefined;
|
|
321
|
+
}
|
|
322
|
+
const content = contentRoot(root);
|
|
323
|
+
const wholePage = content.name === "body" || content.name === "#root";
|
|
324
|
+
let pruned = prune(content, false, wholePage);
|
|
325
|
+
if (textOf(pruned).length < 80) pruned = prune(content, true, false);
|
|
326
|
+
const markdown = blocks(pruned, { base })
|
|
327
|
+
.join("\n\n")
|
|
328
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
329
|
+
.trim();
|
|
330
|
+
return { ...meta, markdown };
|
|
331
|
+
}
|
|
332
|
+
|