mioku-plugin-chat 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +24 -0
- package/configs/base.ts +12 -0
- package/configs/personalization.ts +84 -0
- package/configs/settings.ts +54 -0
- package/context.ts +98 -0
- package/core/base.ts +936 -0
- package/core/chat-engine.ts +779 -0
- package/core/external-skills.ts +94 -0
- package/core/feature-prompts.ts +47 -0
- package/core/media/audio.ts +73 -0
- package/core/media/gif-extractor.ts +118 -0
- package/core/media/history-media.ts +751 -0
- package/core/media/image-analyzer.ts +447 -0
- package/core/media/markdown-message.ts +190 -0
- package/core/multimodal.ts +180 -0
- package/core/prompt.ts +827 -0
- package/core/tools.ts +862 -0
- package/core/web/searxng.ts +184 -0
- package/core/web/web-reader.ts +792 -0
- package/db.ts +849 -0
- package/humanize/emoji-agent.ts +397 -0
- package/humanize/expression.ts +152 -0
- package/humanize/index.ts +40 -0
- package/humanize/memory.ts +131 -0
- package/humanize/planner.ts +249 -0
- package/humanize/topic.ts +205 -0
- package/humanize/utils.ts +28 -0
- package/index.ts +1830 -0
- package/manage/cooldown.ts +445 -0
- package/manage/group-structured-history.ts +195 -0
- package/manage/idle-check.ts +186 -0
- package/manage/queue-processor.ts +228 -0
- package/manage/rate-limiter.ts +312 -0
- package/manage/session.ts +96 -0
- package/manage/skill-session.ts +173 -0
- package/manage/types.ts +138 -0
- package/package.json +51 -0
- package/tsconfig.json +7 -0
- package/types.ts +381 -0
- package/utils/index.ts +9 -0
- package/utils/message.ts +496 -0
- package/utils/queue.ts +186 -0
|
@@ -0,0 +1,792 @@
|
|
|
1
|
+
import { logger } from "mioki";
|
|
2
|
+
import puppeteer from "puppeteer";
|
|
3
|
+
import type { AIInstance } from "mioku";
|
|
4
|
+
import type { WebReaderConfig } from "../../types";
|
|
5
|
+
|
|
6
|
+
type ReadMode = "fetch" | "browser";
|
|
7
|
+
|
|
8
|
+
export interface WebReadArgs {
|
|
9
|
+
url?: string;
|
|
10
|
+
render_js?: boolean;
|
|
11
|
+
question?: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
interface ExtractedPage {
|
|
15
|
+
finalUrl: string;
|
|
16
|
+
title?: string;
|
|
17
|
+
metaDescription?: string;
|
|
18
|
+
headings: string[];
|
|
19
|
+
text: string;
|
|
20
|
+
contentType?: string;
|
|
21
|
+
statusCode?: number;
|
|
22
|
+
sourceBytes?: number;
|
|
23
|
+
warnings: string[];
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
interface PageSummary {
|
|
27
|
+
title?: string;
|
|
28
|
+
content: string;
|
|
29
|
+
warnings: string[];
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
const DEFAULT_USER_AGENT =
|
|
33
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36";
|
|
34
|
+
|
|
35
|
+
const HTML_ENTITY_MAP: Record<string, string> = {
|
|
36
|
+
nbsp: " ",
|
|
37
|
+
amp: "&",
|
|
38
|
+
lt: "<",
|
|
39
|
+
gt: ">",
|
|
40
|
+
quot: '"',
|
|
41
|
+
apos: "'",
|
|
42
|
+
"#39": "'",
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
const BLOCK_TAGS = [
|
|
46
|
+
"address",
|
|
47
|
+
"article",
|
|
48
|
+
"aside",
|
|
49
|
+
"blockquote",
|
|
50
|
+
"br",
|
|
51
|
+
"caption",
|
|
52
|
+
"dd",
|
|
53
|
+
"div",
|
|
54
|
+
"dl",
|
|
55
|
+
"dt",
|
|
56
|
+
"figcaption",
|
|
57
|
+
"figure",
|
|
58
|
+
"footer",
|
|
59
|
+
"h1",
|
|
60
|
+
"h2",
|
|
61
|
+
"h3",
|
|
62
|
+
"h4",
|
|
63
|
+
"h5",
|
|
64
|
+
"h6",
|
|
65
|
+
"header",
|
|
66
|
+
"hr",
|
|
67
|
+
"li",
|
|
68
|
+
"main",
|
|
69
|
+
"nav",
|
|
70
|
+
"ol",
|
|
71
|
+
"p",
|
|
72
|
+
"pre",
|
|
73
|
+
"section",
|
|
74
|
+
"table",
|
|
75
|
+
"td",
|
|
76
|
+
"th",
|
|
77
|
+
"tr",
|
|
78
|
+
"ul",
|
|
79
|
+
];
|
|
80
|
+
|
|
81
|
+
const NOISE_TAGS = [
|
|
82
|
+
"script",
|
|
83
|
+
"style",
|
|
84
|
+
"noscript",
|
|
85
|
+
"svg",
|
|
86
|
+
"canvas",
|
|
87
|
+
"iframe",
|
|
88
|
+
"form",
|
|
89
|
+
"button",
|
|
90
|
+
"input",
|
|
91
|
+
"select",
|
|
92
|
+
"textarea",
|
|
93
|
+
"template",
|
|
94
|
+
];
|
|
95
|
+
|
|
96
|
+
const BOILERPLATE_TAGS = ["nav", "footer", "aside", "header"];
|
|
97
|
+
|
|
98
|
+
function clampText(text: string, maxChars: number): string {
|
|
99
|
+
if (maxChars <= 0 || text.length <= maxChars) {
|
|
100
|
+
return text;
|
|
101
|
+
}
|
|
102
|
+
return `${text.slice(0, maxChars).trim()}...`;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function trimList(
|
|
106
|
+
values: string[],
|
|
107
|
+
maxItems: number,
|
|
108
|
+
maxChars: number,
|
|
109
|
+
): string[] {
|
|
110
|
+
const deduped = [
|
|
111
|
+
...new Set(values.map((item) => normalizeText(item)).filter(Boolean)),
|
|
112
|
+
];
|
|
113
|
+
return deduped.slice(0, maxItems).map((item) => clampText(item, maxChars));
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function normalizeText(text: string): string {
|
|
117
|
+
return text
|
|
118
|
+
.replace(/\r/g, "")
|
|
119
|
+
.replace(/[ \t\f\v\u00a0]+/g, " ")
|
|
120
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
121
|
+
.split("\n")
|
|
122
|
+
.map((line) => line.trim())
|
|
123
|
+
.filter(Boolean)
|
|
124
|
+
.join("\n")
|
|
125
|
+
.trim();
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function decodeHtmlEntities(text: string): string {
|
|
129
|
+
return text.replace(
|
|
130
|
+
/&(#x?[0-9a-fA-F]+|[a-zA-Z]+);/g,
|
|
131
|
+
(entity, body: string) => {
|
|
132
|
+
const mapped = HTML_ENTITY_MAP[body];
|
|
133
|
+
if (mapped != null) {
|
|
134
|
+
return mapped;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
if (body.startsWith("#x") || body.startsWith("#X")) {
|
|
138
|
+
const codePoint = Number.parseInt(body.slice(2), 16);
|
|
139
|
+
if (Number.isFinite(codePoint)) {
|
|
140
|
+
return String.fromCodePoint(codePoint);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
if (body.startsWith("#")) {
|
|
145
|
+
const codePoint = Number.parseInt(body.slice(1), 10);
|
|
146
|
+
if (Number.isFinite(codePoint)) {
|
|
147
|
+
return String.fromCodePoint(codePoint);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
return entity;
|
|
152
|
+
},
|
|
153
|
+
);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
function stripTagBlock(html: string, tagName: string): string {
|
|
157
|
+
const pattern = new RegExp(
|
|
158
|
+
`<${tagName}\\b[^>]*>[\\s\\S]*?<\\/${tagName}>`,
|
|
159
|
+
"gi",
|
|
160
|
+
);
|
|
161
|
+
return html.replace(pattern, "\n");
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
function stripHtmlTags(html: string): string {
|
|
165
|
+
let result = html;
|
|
166
|
+
|
|
167
|
+
for (const tag of NOISE_TAGS) {
|
|
168
|
+
result = stripTagBlock(result, tag);
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
for (const tag of BOILERPLATE_TAGS) {
|
|
172
|
+
result = stripTagBlock(result, tag);
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
result = result.replace(/<!--[\s\S]*?-->/g, "\n");
|
|
176
|
+
result = result.replace(/<br\s*\/?>/gi, "\n");
|
|
177
|
+
|
|
178
|
+
for (const tag of BLOCK_TAGS) {
|
|
179
|
+
const openPattern = new RegExp(`<${tag}\\b[^>]*>`, "gi");
|
|
180
|
+
const closePattern = new RegExp(`<\\/${tag}>`, "gi");
|
|
181
|
+
result = result.replace(openPattern, "\n");
|
|
182
|
+
result = result.replace(closePattern, "\n");
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
result = result.replace(/<[^>]+>/g, " ");
|
|
186
|
+
result = decodeHtmlEntities(result);
|
|
187
|
+
|
|
188
|
+
return normalizeText(result);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
function extractTitleFromHtml(html: string): string | undefined {
|
|
192
|
+
const candidates = [
|
|
193
|
+
/<meta\b[^>]*property=["']og:title["'][^>]*content=["']([^"']+)["'][^>]*>/i,
|
|
194
|
+
/<meta\b[^>]*content=["']([^"']+)["'][^>]*property=["']og:title["'][^>]*>/i,
|
|
195
|
+
/<title\b[^>]*>([\s\S]*?)<\/title>/i,
|
|
196
|
+
];
|
|
197
|
+
|
|
198
|
+
for (const pattern of candidates) {
|
|
199
|
+
const match = html.match(pattern);
|
|
200
|
+
if (match?.[1]) {
|
|
201
|
+
const text = normalizeText(decodeHtmlEntities(match[1]));
|
|
202
|
+
if (text) {
|
|
203
|
+
return text;
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
return undefined;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
function extractMetaDescription(html: string): string | undefined {
|
|
212
|
+
const patterns = [
|
|
213
|
+
/<meta\b[^>]*name=["']description["'][^>]*content=["']([^"']+)["'][^>]*>/i,
|
|
214
|
+
/<meta\b[^>]*content=["']([^"']+)["'][^>]*name=["']description["'][^>]*>/i,
|
|
215
|
+
/<meta\b[^>]*property=["']og:description["'][^>]*content=["']([^"']+)["'][^>]*>/i,
|
|
216
|
+
/<meta\b[^>]*content=["']([^"']+)["'][^>]*property=["']og:description["'][^>]*>/i,
|
|
217
|
+
];
|
|
218
|
+
|
|
219
|
+
for (const pattern of patterns) {
|
|
220
|
+
const match = html.match(pattern);
|
|
221
|
+
if (match?.[1]) {
|
|
222
|
+
const text = normalizeText(decodeHtmlEntities(match[1]));
|
|
223
|
+
if (text) {
|
|
224
|
+
return text;
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
return undefined;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
function extractHeadings(html: string): string[] {
|
|
233
|
+
const headings: string[] = [];
|
|
234
|
+
const pattern = /<(h1|h2|h3)\b[^>]*>([\s\S]*?)<\/\1>/gi;
|
|
235
|
+
|
|
236
|
+
let match: RegExpExecArray | null;
|
|
237
|
+
while ((match = pattern.exec(html))) {
|
|
238
|
+
const text = stripHtmlTags(match[2]);
|
|
239
|
+
if (text) {
|
|
240
|
+
headings.push(text);
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
return trimList(headings, 10, 120);
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
function pickBestParagraphs(text: string, maxChars: number): string {
|
|
248
|
+
const paragraphs = text
|
|
249
|
+
.split(/\n{1,2}/)
|
|
250
|
+
.map((paragraph, index) => ({
|
|
251
|
+
index,
|
|
252
|
+
text: paragraph.trim(),
|
|
253
|
+
}))
|
|
254
|
+
.filter((paragraph) => paragraph.text.length >= 12);
|
|
255
|
+
|
|
256
|
+
if (paragraphs.length === 0) {
|
|
257
|
+
return clampText(text, maxChars);
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
if (paragraphs.length <= 8) {
|
|
261
|
+
return clampText(
|
|
262
|
+
paragraphs.map((paragraph) => paragraph.text).join("\n\n"),
|
|
263
|
+
maxChars,
|
|
264
|
+
);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
const scored = paragraphs.map((paragraph) => {
|
|
268
|
+
const value = paragraph.text;
|
|
269
|
+
const punctuationCount = (value.match(/[,。!?;:,.!?;:]/g) || [])
|
|
270
|
+
.length;
|
|
271
|
+
const sentenceCount = (value.match(/[。!?.!?]/g) || []).length;
|
|
272
|
+
const linkLikeCount = (value.match(/https?:\/\/|www\./g) || []).length;
|
|
273
|
+
const separatorCount = (value.match(/[|>•·]/g) || []).length;
|
|
274
|
+
|
|
275
|
+
let score = value.length + punctuationCount * 12 + sentenceCount * 20;
|
|
276
|
+
if (value.length < 24) score -= 40;
|
|
277
|
+
score -= linkLikeCount * 40;
|
|
278
|
+
score -= separatorCount * 8;
|
|
279
|
+
|
|
280
|
+
return {
|
|
281
|
+
...paragraph,
|
|
282
|
+
score,
|
|
283
|
+
};
|
|
284
|
+
});
|
|
285
|
+
|
|
286
|
+
const selected = scored
|
|
287
|
+
.sort((a, b) => b.score - a.score)
|
|
288
|
+
.slice(0, 18)
|
|
289
|
+
.sort((a, b) => a.index - b.index)
|
|
290
|
+
.map((item) => item.text);
|
|
291
|
+
|
|
292
|
+
return clampText(selected.join("\n\n"), maxChars);
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
function extractHtmlContent(
|
|
296
|
+
html: string,
|
|
297
|
+
cfg: WebReaderConfig,
|
|
298
|
+
): Omit<
|
|
299
|
+
ExtractedPage,
|
|
300
|
+
"finalUrl" | "contentType" | "statusCode" | "sourceBytes"
|
|
301
|
+
> {
|
|
302
|
+
const title = extractTitleFromHtml(html);
|
|
303
|
+
const metaDescription = extractMetaDescription(html);
|
|
304
|
+
const headings = extractHeadings(html);
|
|
305
|
+
|
|
306
|
+
let workingHtml = html;
|
|
307
|
+
for (const tag of NOISE_TAGS) {
|
|
308
|
+
workingHtml = stripTagBlock(workingHtml, tag);
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
const bodyMatch = workingHtml.match(/<body\b[^>]*>([\s\S]*?)<\/body>/i);
|
|
312
|
+
const bodyHtml = bodyMatch?.[1] || workingHtml;
|
|
313
|
+
const plainText = stripHtmlTags(bodyHtml);
|
|
314
|
+
const pickedText = pickBestParagraphs(plainText, cfg.maxExtractedChars);
|
|
315
|
+
const warnings: string[] = [];
|
|
316
|
+
|
|
317
|
+
if (pickedText.length < 180) {
|
|
318
|
+
warnings.push(
|
|
319
|
+
"Extracted content is sparse. The page may rely on JavaScript rendering.",
|
|
320
|
+
);
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
return {
|
|
324
|
+
title,
|
|
325
|
+
metaDescription,
|
|
326
|
+
headings,
|
|
327
|
+
text: pickedText,
|
|
328
|
+
warnings,
|
|
329
|
+
};
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
function normalizeUrl(input: string): string {
|
|
333
|
+
const url = new URL(input);
|
|
334
|
+
if (!["http:", "https:"].includes(url.protocol)) {
|
|
335
|
+
throw new Error("Only http and https URLs are supported");
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
const hostname = url.hostname.toLowerCase();
|
|
339
|
+
if (
|
|
340
|
+
hostname === "localhost" ||
|
|
341
|
+
hostname === "127.0.0.1" ||
|
|
342
|
+
hostname === "::1" ||
|
|
343
|
+
hostname.endsWith(".local")
|
|
344
|
+
) {
|
|
345
|
+
throw new Error("Local network URLs are not allowed");
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
return url.toString();
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
function parseCharset(contentType: string | null): string | null {
|
|
352
|
+
if (!contentType) return null;
|
|
353
|
+
const match = contentType.match(/charset=([^;]+)/i);
|
|
354
|
+
return match?.[1]?.trim() || null;
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
function createTextDecoder(charset: string | null) {
|
|
358
|
+
if (!charset) {
|
|
359
|
+
return new TextDecoder("utf-8");
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
try {
|
|
363
|
+
return new TextDecoder(charset);
|
|
364
|
+
} catch {
|
|
365
|
+
return new TextDecoder("utf-8");
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
async function readResponseText(
|
|
370
|
+
response: Response,
|
|
371
|
+
maxBytes: number,
|
|
372
|
+
): Promise<{ text: string; bytes: number }> {
|
|
373
|
+
const reader = response.body?.getReader();
|
|
374
|
+
if (!reader) {
|
|
375
|
+
const text = await response.text();
|
|
376
|
+
const bytes = Buffer.byteLength(text);
|
|
377
|
+
if (bytes > maxBytes) {
|
|
378
|
+
throw new Error(`Response body exceeds ${maxBytes} bytes`);
|
|
379
|
+
}
|
|
380
|
+
return { text, bytes };
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
const decoder = createTextDecoder(
|
|
384
|
+
parseCharset(response.headers.get("content-type")),
|
|
385
|
+
);
|
|
386
|
+
let bytes = 0;
|
|
387
|
+
let text = "";
|
|
388
|
+
|
|
389
|
+
while (true) {
|
|
390
|
+
const { done, value } = await reader.read();
|
|
391
|
+
if (done) {
|
|
392
|
+
break;
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
bytes += value.byteLength;
|
|
396
|
+
if (bytes > maxBytes) {
|
|
397
|
+
await reader.cancel();
|
|
398
|
+
throw new Error(`Response body exceeds ${maxBytes} bytes`);
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
text += decoder.decode(value, { stream: true });
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
text += decoder.decode();
|
|
405
|
+
return { text, bytes };
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
function isAllowedContentType(
|
|
409
|
+
contentType: string | null,
|
|
410
|
+
cfg: WebReaderConfig,
|
|
411
|
+
): boolean {
|
|
412
|
+
if (!contentType) {
|
|
413
|
+
return true;
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
const normalized = contentType.toLowerCase();
|
|
417
|
+
return cfg.allowedContentTypes.some((allowed) =>
|
|
418
|
+
normalized.startsWith(allowed.toLowerCase()),
|
|
419
|
+
);
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
async function fetchPage(
|
|
423
|
+
url: string,
|
|
424
|
+
cfg: WebReaderConfig,
|
|
425
|
+
): Promise<ExtractedPage> {
|
|
426
|
+
const controller = new AbortController();
|
|
427
|
+
const timeout = setTimeout(() => controller.abort(), cfg.timeoutMs);
|
|
428
|
+
|
|
429
|
+
try {
|
|
430
|
+
const response = await fetch(url, {
|
|
431
|
+
method: "GET",
|
|
432
|
+
headers: {
|
|
433
|
+
Accept: "text/html,application/xhtml+xml,text/plain;q=0.9,*/*;q=0.5",
|
|
434
|
+
"User-Agent": DEFAULT_USER_AGENT,
|
|
435
|
+
},
|
|
436
|
+
redirect: "follow",
|
|
437
|
+
signal: controller.signal,
|
|
438
|
+
});
|
|
439
|
+
|
|
440
|
+
if (!response.ok) {
|
|
441
|
+
throw new Error(`HTTP ${response.status}`);
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
const contentType = response.headers.get("content-type");
|
|
445
|
+
if (!isAllowedContentType(contentType, cfg)) {
|
|
446
|
+
throw new Error(`Unsupported content-type: ${contentType || "unknown"}`);
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
const finalUrl = response.url || url;
|
|
450
|
+
const { text, bytes } = await readResponseText(response, cfg.maxHtmlBytes);
|
|
451
|
+
const normalizedContentType = (contentType || "").toLowerCase();
|
|
452
|
+
|
|
453
|
+
if (normalizedContentType.startsWith("text/plain")) {
|
|
454
|
+
const normalizedText = clampText(
|
|
455
|
+
normalizeText(text),
|
|
456
|
+
cfg.maxExtractedChars,
|
|
457
|
+
);
|
|
458
|
+
return {
|
|
459
|
+
finalUrl,
|
|
460
|
+
title: undefined,
|
|
461
|
+
metaDescription: undefined,
|
|
462
|
+
headings: [],
|
|
463
|
+
text: normalizedText,
|
|
464
|
+
contentType: contentType || undefined,
|
|
465
|
+
statusCode: response.status,
|
|
466
|
+
sourceBytes: bytes,
|
|
467
|
+
warnings:
|
|
468
|
+
normalizedText.length < 180 ? ["Extracted content is sparse."] : [],
|
|
469
|
+
};
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
const extracted = extractHtmlContent(text, cfg);
|
|
473
|
+
return {
|
|
474
|
+
finalUrl,
|
|
475
|
+
contentType: contentType || undefined,
|
|
476
|
+
statusCode: response.status,
|
|
477
|
+
sourceBytes: bytes,
|
|
478
|
+
...extracted,
|
|
479
|
+
};
|
|
480
|
+
} catch (err) {
|
|
481
|
+
const isAbort = err instanceof Error && err.name === "AbortError";
|
|
482
|
+
throw new Error(
|
|
483
|
+
isAbort ? `Request timeout after ${cfg.timeoutMs}ms` : String(err),
|
|
484
|
+
);
|
|
485
|
+
} finally {
|
|
486
|
+
clearTimeout(timeout);
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
async function renderPage(
|
|
491
|
+
url: string,
|
|
492
|
+
cfg: WebReaderConfig,
|
|
493
|
+
): Promise<ExtractedPage> {
|
|
494
|
+
const browser = await puppeteer.launch({
|
|
495
|
+
headless: true,
|
|
496
|
+
args: ["--no-sandbox", "--disable-setuid-sandbox"],
|
|
497
|
+
});
|
|
498
|
+
|
|
499
|
+
let page;
|
|
500
|
+
try {
|
|
501
|
+
page = await browser.newPage();
|
|
502
|
+
await page.setUserAgent(DEFAULT_USER_AGENT);
|
|
503
|
+
await page.goto(url, {
|
|
504
|
+
waitUntil: "domcontentloaded",
|
|
505
|
+
timeout: cfg.browserTimeoutMs,
|
|
506
|
+
});
|
|
507
|
+
await page
|
|
508
|
+
.waitForNetworkIdle({
|
|
509
|
+
idleTime: 500,
|
|
510
|
+
timeout: Math.min(cfg.browserTimeoutMs, 3000),
|
|
511
|
+
})
|
|
512
|
+
.catch(() => undefined);
|
|
513
|
+
|
|
514
|
+
const pageUrl = page.url();
|
|
515
|
+
const contentType = await page.evaluate(() => {
|
|
516
|
+
const doc = (globalThis as any).document;
|
|
517
|
+
const contentTypeValue = doc?.contentType;
|
|
518
|
+
return typeof contentTypeValue === "string" ? contentTypeValue : "";
|
|
519
|
+
});
|
|
520
|
+
|
|
521
|
+
const extracted = await page.evaluate((maxChars: number) => {
|
|
522
|
+
const doc = (globalThis as any).document;
|
|
523
|
+
const textOf = (node: any): string =>
|
|
524
|
+
String(node?.innerText || node?.textContent || "")
|
|
525
|
+
.replace(/\r/g, "")
|
|
526
|
+
.replace(/[ \t\f\v\u00a0]+/g, " ")
|
|
527
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
528
|
+
.split("\n")
|
|
529
|
+
.map((line: string) => line.trim())
|
|
530
|
+
.filter(Boolean)
|
|
531
|
+
.join("\n")
|
|
532
|
+
.trim();
|
|
533
|
+
|
|
534
|
+
const metaDescription =
|
|
535
|
+
doc
|
|
536
|
+
?.querySelector(
|
|
537
|
+
'meta[name="description"], meta[property="og:description"]',
|
|
538
|
+
)
|
|
539
|
+
?.getAttribute("content") || "";
|
|
540
|
+
|
|
541
|
+
const headings = Array.from(doc?.querySelectorAll("h1, h2, h3") || [])
|
|
542
|
+
.map((node: any) => textOf(node))
|
|
543
|
+
.filter(Boolean)
|
|
544
|
+
.slice(0, 10);
|
|
545
|
+
|
|
546
|
+
const candidates = Array.from(
|
|
547
|
+
doc?.querySelectorAll("article, main, [role='main'], section, div") ||
|
|
548
|
+
[],
|
|
549
|
+
).slice(0, 300) as any[];
|
|
550
|
+
|
|
551
|
+
let bestNode = doc?.querySelector("article, main, [role='main']") || null;
|
|
552
|
+
let bestScore = -1;
|
|
553
|
+
|
|
554
|
+
for (const node of candidates) {
|
|
555
|
+
const text = textOf(node);
|
|
556
|
+
if (text.length < 120) continue;
|
|
557
|
+
|
|
558
|
+
const paragraphCount = node.querySelectorAll
|
|
559
|
+
? node.querySelectorAll("p").length
|
|
560
|
+
: 0;
|
|
561
|
+
const linkTextLength = Array.from(node.querySelectorAll?.("a") || [])
|
|
562
|
+
.map((link: any) => textOf(link).length)
|
|
563
|
+
.reduce((sum: number, item: number) => sum + item, 0);
|
|
564
|
+
|
|
565
|
+
const linkDensity = text.length > 0 ? linkTextLength / text.length : 1;
|
|
566
|
+
const score =
|
|
567
|
+
text.length + paragraphCount * 80 - Math.round(linkDensity * 600);
|
|
568
|
+
|
|
569
|
+
if (score > bestScore) {
|
|
570
|
+
bestScore = score;
|
|
571
|
+
bestNode = node;
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
const bestText = textOf(bestNode || doc?.body);
|
|
576
|
+
return {
|
|
577
|
+
title: String(doc?.title || "").trim(),
|
|
578
|
+
metaDescription: String(metaDescription).trim(),
|
|
579
|
+
headings,
|
|
580
|
+
text:
|
|
581
|
+
bestText.length > maxChars
|
|
582
|
+
? `${bestText.slice(0, maxChars).trim()}...`
|
|
583
|
+
: bestText,
|
|
584
|
+
};
|
|
585
|
+
}, cfg.maxExtractedChars);
|
|
586
|
+
|
|
587
|
+
const warnings: string[] = [];
|
|
588
|
+
if (extracted.text.length < 180) {
|
|
589
|
+
warnings.push("Rendered page still contains little readable text.");
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
return {
|
|
593
|
+
finalUrl: pageUrl || url,
|
|
594
|
+
title: normalizeText(extracted.title),
|
|
595
|
+
metaDescription: normalizeText(extracted.metaDescription),
|
|
596
|
+
headings: trimList(extracted.headings, 10, 120),
|
|
597
|
+
text: normalizeText(extracted.text),
|
|
598
|
+
contentType: contentType || "text/html",
|
|
599
|
+
sourceBytes: undefined,
|
|
600
|
+
statusCode: undefined,
|
|
601
|
+
warnings,
|
|
602
|
+
};
|
|
603
|
+
} catch (err) {
|
|
604
|
+
throw new Error(
|
|
605
|
+
`Browser rendering failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
606
|
+
);
|
|
607
|
+
} finally {
|
|
608
|
+
if (page) {
|
|
609
|
+
await page.close().catch(() => undefined);
|
|
610
|
+
}
|
|
611
|
+
await browser.close().catch(() => undefined);
|
|
612
|
+
}
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
function buildSummarizerPrompt(): string {
|
|
616
|
+
return `You clean webpage content for another LLM.
|
|
617
|
+
|
|
618
|
+
Return strict JSON only:
|
|
619
|
+
{
|
|
620
|
+
"title": "optional better title",
|
|
621
|
+
"content": "cleaned main webpage content",
|
|
622
|
+
"warnings": ["warning 1"]
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
Rules:
|
|
626
|
+
- Do not summarize, compress, or rewrite the page into an abstract overview.
|
|
627
|
+
- Your primary job is to remove irrelevant material while preserving the main body content as fully as possible.
|
|
628
|
+
- Keep the page's original facts, details, narrative flow, examples, and conclusions whenever they belong to the main content.
|
|
629
|
+
- Keep names, numbers, dates, versions, entities, relationships, conclusions, examples, and caveats whenever they appear in the source.
|
|
630
|
+
- Remove ads, navigation, repeated boilerplate, cookie text, subscription prompts, decorative text, and other non-content noise first.
|
|
631
|
+
- Remove obviously duplicated text, but otherwise prefer retaining content over shortening it.
|
|
632
|
+
- If a question is provided, make sure relevant content is preserved, but do not drop other important main-page information just to focus on that question.
|
|
633
|
+
- warnings: include uncertainty, missing context, paywall/login issues, sparse extraction, or likely JS-rendering gaps.
|
|
634
|
+
- Do not use markdown. Output valid JSON only.`;
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
function parseJsonContent(content: string): any {
|
|
638
|
+
const jsonMatch = content.match(/\{[\s\S]*\}/);
|
|
639
|
+
const payload = jsonMatch?.[0] || content;
|
|
640
|
+
return JSON.parse(payload);
|
|
641
|
+
}
|
|
642
|
+
|
|
643
|
+
async function summarizePage(
|
|
644
|
+
ai: AIInstance,
|
|
645
|
+
model: string,
|
|
646
|
+
page: ExtractedPage,
|
|
647
|
+
cfg: WebReaderConfig,
|
|
648
|
+
question?: string,
|
|
649
|
+
): Promise<PageSummary> {
|
|
650
|
+
const contentParts = [
|
|
651
|
+
page.title ? `Title: ${page.title}` : "",
|
|
652
|
+
page.metaDescription ? `Meta description: ${page.metaDescription}` : "",
|
|
653
|
+
page.headings.length > 0 ? `Headings:\n${page.headings.join("\n")}` : "",
|
|
654
|
+
question ? `Question focus: ${question.trim()}` : "",
|
|
655
|
+
`Readable content:\n${page.text}`,
|
|
656
|
+
].filter(Boolean);
|
|
657
|
+
|
|
658
|
+
const userContent = contentParts.join("\n\n");
|
|
659
|
+
|
|
660
|
+
try {
|
|
661
|
+
const response = await ai.complete({
|
|
662
|
+
model,
|
|
663
|
+
messages: [
|
|
664
|
+
{
|
|
665
|
+
role: "system",
|
|
666
|
+
content: buildSummarizerPrompt(),
|
|
667
|
+
},
|
|
668
|
+
{
|
|
669
|
+
role: "user",
|
|
670
|
+
content: userContent,
|
|
671
|
+
},
|
|
672
|
+
],
|
|
673
|
+
temperature: 0.2,
|
|
674
|
+
});
|
|
675
|
+
|
|
676
|
+
if (!response.content) {
|
|
677
|
+
throw new Error("Model returned empty response");
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
const parsed = parseJsonContent(response.content);
|
|
681
|
+
const title = normalizeText(String(parsed.title || page.title || ""));
|
|
682
|
+
const content = normalizeText(
|
|
683
|
+
String(parsed.content || parsed.summary || ""),
|
|
684
|
+
);
|
|
685
|
+
const warnings = trimList(
|
|
686
|
+
[
|
|
687
|
+
...(Array.isArray(parsed.warnings) ? parsed.warnings.map(String) : []),
|
|
688
|
+
...page.warnings,
|
|
689
|
+
],
|
|
690
|
+
6,
|
|
691
|
+
180,
|
|
692
|
+
);
|
|
693
|
+
|
|
694
|
+
return {
|
|
695
|
+
title: title || page.title,
|
|
696
|
+
content: content || page.text,
|
|
697
|
+
warnings,
|
|
698
|
+
};
|
|
699
|
+
} catch (err) {
|
|
700
|
+
logger.warn(`[web-reader] Failed to summarize with model: ${err}`);
|
|
701
|
+
return {
|
|
702
|
+
title: page.title,
|
|
703
|
+
content: page.text,
|
|
704
|
+
warnings: trimList(
|
|
705
|
+
[
|
|
706
|
+
...page.warnings,
|
|
707
|
+
"Model summarization failed; returned fallback extraction.",
|
|
708
|
+
],
|
|
709
|
+
6,
|
|
710
|
+
180,
|
|
711
|
+
),
|
|
712
|
+
};
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
export async function readWebPage(
|
|
717
|
+
ai: AIInstance | undefined,
|
|
718
|
+
model: string,
|
|
719
|
+
cfg: WebReaderConfig,
|
|
720
|
+
args: WebReadArgs,
|
|
721
|
+
): Promise<Record<string, unknown>> {
|
|
722
|
+
if (!cfg.enabled) {
|
|
723
|
+
return {
|
|
724
|
+
success: false,
|
|
725
|
+
error: "Web reader is disabled in config",
|
|
726
|
+
};
|
|
727
|
+
}
|
|
728
|
+
|
|
729
|
+
const rawUrl = String(args.url || "").trim();
|
|
730
|
+
if (!rawUrl) {
|
|
731
|
+
return {
|
|
732
|
+
success: false,
|
|
733
|
+
error: "url is required",
|
|
734
|
+
};
|
|
735
|
+
}
|
|
736
|
+
|
|
737
|
+
try {
|
|
738
|
+
const normalizedUrl = normalizeUrl(rawUrl);
|
|
739
|
+
const mode: ReadMode = args.render_js ? "browser" : "fetch";
|
|
740
|
+
|
|
741
|
+
logger.info(`[web-reader] Reading ${normalizedUrl} via ${mode}`);
|
|
742
|
+
|
|
743
|
+
const page =
|
|
744
|
+
mode === "browser"
|
|
745
|
+
? await renderPage(normalizedUrl, cfg)
|
|
746
|
+
: await fetchPage(normalizedUrl, cfg);
|
|
747
|
+
|
|
748
|
+
if (!page.text) {
|
|
749
|
+
return {
|
|
750
|
+
success: false,
|
|
751
|
+
url: normalizedUrl,
|
|
752
|
+
finalUrl: page.finalUrl,
|
|
753
|
+
mode,
|
|
754
|
+
error: "No readable content extracted from page",
|
|
755
|
+
};
|
|
756
|
+
}
|
|
757
|
+
|
|
758
|
+
const summary =
|
|
759
|
+
cfg.useWorkingModel && ai
|
|
760
|
+
? await summarizePage(ai, model, page, cfg, args.question)
|
|
761
|
+
: {
|
|
762
|
+
title: page.title,
|
|
763
|
+
content: page.text,
|
|
764
|
+
warnings: page.warnings,
|
|
765
|
+
};
|
|
766
|
+
|
|
767
|
+
return {
|
|
768
|
+
success: true,
|
|
769
|
+
url: normalizedUrl,
|
|
770
|
+
finalUrl: page.finalUrl,
|
|
771
|
+
mode,
|
|
772
|
+
title: summary.title || page.title,
|
|
773
|
+
content: summary.content,
|
|
774
|
+
headings: page.headings,
|
|
775
|
+
meta_description: page.metaDescription,
|
|
776
|
+
warnings: summary.warnings,
|
|
777
|
+
content_type: page.contentType,
|
|
778
|
+
status_code: page.statusCode,
|
|
779
|
+
content_stats: {
|
|
780
|
+
source_bytes: page.sourceBytes,
|
|
781
|
+
extracted_chars: page.text.length,
|
|
782
|
+
processed_chars: summary.content.length,
|
|
783
|
+
processed_by_working_model: Boolean(cfg.useWorkingModel && ai),
|
|
784
|
+
},
|
|
785
|
+
};
|
|
786
|
+
} catch (err) {
|
|
787
|
+
return {
|
|
788
|
+
success: false,
|
|
789
|
+
error: err instanceof Error ? err.message : String(err),
|
|
790
|
+
};
|
|
791
|
+
}
|
|
792
|
+
}
|