extract-webpage 1.2.109 → 1.2.111
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +2 -2
- package/dist/extract-webpage.es.js.map +1 -1
- package/dist/html-to-content/html-utils.d.ts +21 -0
- package/package.json +4 -4
- package/src/html-to-content/html-utils.ts +210 -77
- package/src/url-to-content/url-to-html.ts +5 -2
|
@@ -55,6 +55,27 @@ export declare function convertURLToAbsoluteURL(base: any, relative: any): any;
|
|
|
55
55
|
* // <ul><li>List item 1</li><li>List item 2</li></ul>
|
|
56
56
|
*/
|
|
57
57
|
export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): any;
|
|
58
|
+
/**
|
|
59
|
+
* Convert a Markdown document to formatted HTML using regular expressions to
|
|
60
|
+
* detect Markdown syntax. Unlike {@link convertMarkdownToHTML} (which relies on
|
|
61
|
+
* the `marked` library), this is a dependency-free, self-contained converter
|
|
62
|
+
* intended for post-processing content returned as Markdown (e.g. from the
|
|
63
|
+
* JINA reader fallback in the scraper).
|
|
64
|
+
*
|
|
65
|
+
* Supported block elements: ATX headers (`#`..`######`), fenced code blocks
|
|
66
|
+
* (```lang), blockquotes (`>`), unordered lists (`-`, `*`, `+`), ordered lists
|
|
67
|
+
* (`1.`, `1)`), horizontal rules (`---`, `***`, `___`) and paragraphs.
|
|
68
|
+
* Supported inline elements: bold, italic, strikethrough, inline code, images
|
|
69
|
+
* and links.
|
|
70
|
+
*
|
|
71
|
+
* @param {string} markdown - The Markdown content to convert.
|
|
72
|
+
* @returns {string} The resulting formatted HTML string.
|
|
73
|
+
* @category HTML Utilities
|
|
74
|
+
* @example
|
|
75
|
+
* convertMarkdownToFormattedHTML("# Title\n\nSome **bold** text.");
|
|
76
|
+
* // => "<h1>Title</h1>\n<p>Some <strong>bold</strong> text.</p>"
|
|
77
|
+
*/
|
|
78
|
+
export declare function convertMarkdownToFormattedHTML(markdown: any): string;
|
|
58
79
|
export declare function convertHTMLToMarkdown(html: any): any;
|
|
59
80
|
/**
|
|
60
81
|
* Copy HTML to clipboard. When pasting into rich text field,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-webpage",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.111",
|
|
4
4
|
"module": "./dist/extract-webpage.es.js",
|
|
5
5
|
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
6
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
@@ -72,11 +72,11 @@
|
|
|
72
72
|
"dependencies": {
|
|
73
73
|
"@huggingface/transformers": "^3.8.1",
|
|
74
74
|
"ai": "^5.0.0",
|
|
75
|
-
"chat-agent-toolkit": "^1.2.
|
|
75
|
+
"chat-agent-toolkit": "^1.2.111",
|
|
76
76
|
"chrono-node": "^2.9.0",
|
|
77
77
|
"drizzle-orm": "^0.45.1",
|
|
78
|
-
"extract-pdf": "^0.1.
|
|
79
|
-
"extract-youtube": "^1.0.
|
|
78
|
+
"extract-pdf": "^0.1.98",
|
|
79
|
+
"extract-youtube": "^1.0.96",
|
|
80
80
|
"html-entities": "^2.6.0",
|
|
81
81
|
"js-yaml": "^4.1.1",
|
|
82
82
|
"jsdom": "^28.1.0",
|
|
@@ -206,96 +206,229 @@ export function convertMarkdownToHTML(content, toHtml = true) {
|
|
|
206
206
|
if (!toHtml) return convertHTMLToMarkdown(content);
|
|
207
207
|
|
|
208
208
|
return content?.length ? marked.parse(content) : "";
|
|
209
|
+
}
|
|
209
210
|
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
211
|
+
/**
|
|
212
|
+
* Escape the HTML-significant characters in a raw string so it can be
|
|
213
|
+
* safely embedded inside generated HTML (used for code spans/blocks).
|
|
214
|
+
* @param {string} str
|
|
215
|
+
* @returns {string}
|
|
216
|
+
*/
|
|
217
|
+
function escapeHTMLChars(str) {
|
|
218
|
+
return String(str)
|
|
219
|
+
.replace(/&/g, "&")
|
|
220
|
+
.replace(/</g, "<")
|
|
221
|
+
.replace(/>/g, ">");
|
|
222
|
+
}
|
|
216
223
|
|
|
217
|
-
|
|
218
|
-
|
|
224
|
+
/**
|
|
225
|
+
* Apply inline-level Markdown regexp replacements (images, links, bold,
|
|
226
|
+
* italic, strikethrough) to a single already-block-parsed line of text.
|
|
227
|
+
* Inline code spans are expected to already be swapped out for placeholders
|
|
228
|
+
* so their contents are never touched here.
|
|
229
|
+
* @param {string} text
|
|
230
|
+
* @returns {string}
|
|
231
|
+
*/
|
|
232
|
+
function applyInlineMarkdown(text) {
|
|
233
|
+
return text
|
|
234
|
+
// Images:  -- must run before links
|
|
235
|
+
.replace(
|
|
236
|
+
/!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/g,
|
|
237
|
+
(_m, alt, src, title) =>
|
|
238
|
+
`<img src="${src}" alt="${alt}"${title ? ` title="${title}"` : ""} />`
|
|
239
|
+
)
|
|
240
|
+
// Links: [text](href "title")
|
|
241
|
+
.replace(
|
|
242
|
+
/\[([^\]]+)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/g,
|
|
243
|
+
(_m, label, href, title) =>
|
|
244
|
+
`<a href="${href}"${title ? ` title="${title}"` : ""}>${label}</a>`
|
|
245
|
+
)
|
|
246
|
+
// Bold: **text** or __text__
|
|
247
|
+
.replace(/\*\*([^*]+)\*\*/g, "<strong>$1</strong>")
|
|
248
|
+
.replace(/__([^_]+)__/g, "<strong>$1</strong>")
|
|
249
|
+
// Italic: *text* or _text_ (avoid matching inside words for `_`)
|
|
250
|
+
.replace(/\*([^*\n]+)\*/g, "<em>$1</em>")
|
|
251
|
+
.replace(/(^|[^A-Za-z0-9_])_([^_\n]+)_(?=[^A-Za-z0-9_]|$)/g, "$1<em>$2</em>")
|
|
252
|
+
// Strikethrough: ~~text~~
|
|
253
|
+
.replace(/~~([^~]+)~~/g, "<del>$1</del>");
|
|
254
|
+
}
|
|
219
255
|
|
|
220
|
-
|
|
221
|
-
|
|
256
|
+
/**
|
|
257
|
+
* Convert a Markdown document to formatted HTML using regular expressions to
|
|
258
|
+
* detect Markdown syntax. Unlike {@link convertMarkdownToHTML} (which relies on
|
|
259
|
+
* the `marked` library), this is a dependency-free, self-contained converter
|
|
260
|
+
* intended for post-processing content returned as Markdown (e.g. from the
|
|
261
|
+
* JINA reader fallback in the scraper).
|
|
262
|
+
*
|
|
263
|
+
* Supported block elements: ATX headers (`#`..`######`), fenced code blocks
|
|
264
|
+
* (```lang), blockquotes (`>`), unordered lists (`-`, `*`, `+`), ordered lists
|
|
265
|
+
* (`1.`, `1)`), horizontal rules (`---`, `***`, `___`) and paragraphs.
|
|
266
|
+
* Supported inline elements: bold, italic, strikethrough, inline code, images
|
|
267
|
+
* and links.
|
|
268
|
+
*
|
|
269
|
+
* @param {string} markdown - The Markdown content to convert.
|
|
270
|
+
* @returns {string} The resulting formatted HTML string.
|
|
271
|
+
* @category HTML Utilities
|
|
272
|
+
* @example
|
|
273
|
+
* convertMarkdownToFormattedHTML("# Title\n\nSome **bold** text.");
|
|
274
|
+
* // => "<h1>Title</h1>\n<p>Some <strong>bold</strong> text.</p>"
|
|
275
|
+
*/
|
|
276
|
+
export function convertMarkdownToFormattedHTML(markdown) {
|
|
277
|
+
if (!markdown || typeof markdown !== "string") return "";
|
|
278
|
+
|
|
279
|
+
let text = markdown.replace(/\r\n?/g, "\n");
|
|
280
|
+
|
|
281
|
+
// 1. Pull fenced code blocks out first so their contents are never parsed
|
|
282
|
+
// as Markdown. Each is replaced by a placeholder restored at the end.
|
|
283
|
+
const codeBlocks = [];
|
|
284
|
+
text = text.replace(
|
|
285
|
+
/```([^\n`]*)\n([\s\S]*?)```/g,
|
|
286
|
+
(_m, lang, code) => {
|
|
287
|
+
const language = (lang || "").trim();
|
|
288
|
+
const cls = language ? ` class="language-${language}"` : "";
|
|
289
|
+
const body = escapeHTMLChars(code.replace(/\n$/, ""));
|
|
290
|
+
codeBlocks.push(`<pre><code${cls}>${body}</code></pre>`);
|
|
291
|
+
return `\u0000CB${codeBlocks.length - 1}\u0000`;
|
|
292
|
+
}
|
|
293
|
+
);
|
|
222
294
|
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
295
|
+
// 2. Pull inline code spans out next for the same reason.
|
|
296
|
+
const inlineCodes = [];
|
|
297
|
+
text = text.replace(/`([^`\n]+)`/g, (_m, code) => {
|
|
298
|
+
inlineCodes.push(`<code>${escapeHTMLChars(code)}</code>`);
|
|
299
|
+
return `\u0000IC${inlineCodes.length - 1}\u0000`;
|
|
300
|
+
});
|
|
226
301
|
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
.
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
302
|
+
const lines = text.split("\n");
|
|
303
|
+
const out = [];
|
|
304
|
+
let inUl = false;
|
|
305
|
+
let inOl = false;
|
|
306
|
+
let inBlockquote = false;
|
|
307
|
+
let paragraph = [];
|
|
308
|
+
|
|
309
|
+
const flushParagraph = () => {
|
|
310
|
+
if (paragraph.length) {
|
|
311
|
+
out.push(`<p>${applyInlineMarkdown(paragraph.join(" "))}</p>`);
|
|
312
|
+
paragraph = [];
|
|
313
|
+
}
|
|
314
|
+
};
|
|
315
|
+
const closeLists = () => {
|
|
316
|
+
if (inUl) {
|
|
317
|
+
out.push("</ul>");
|
|
318
|
+
inUl = false;
|
|
319
|
+
}
|
|
320
|
+
if (inOl) {
|
|
321
|
+
out.push("</ol>");
|
|
322
|
+
inOl = false;
|
|
323
|
+
}
|
|
324
|
+
};
|
|
325
|
+
const closeBlockquote = () => {
|
|
326
|
+
if (inBlockquote) {
|
|
327
|
+
out.push("</blockquote>");
|
|
328
|
+
inBlockquote = false;
|
|
329
|
+
}
|
|
330
|
+
};
|
|
253
331
|
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
.replace(/&/g, "&")
|
|
266
|
-
.replace(/</g, "<")
|
|
267
|
-
.replace(/>/g, ">")
|
|
268
|
-
.replace(/"/g, """)
|
|
269
|
-
.replace(/'/g, "'");
|
|
270
|
-
|
|
271
|
-
return pre + "<code>" + code + "</code>";
|
|
272
|
-
}
|
|
273
|
-
)
|
|
332
|
+
for (const line of lines) {
|
|
333
|
+
// Standalone fenced-code-block placeholder line
|
|
334
|
+
const cb = line.match(/^\u0000CB(\d+)\u0000$/);
|
|
335
|
+
if (cb) {
|
|
336
|
+
flushParagraph();
|
|
337
|
+
closeLists();
|
|
338
|
+
closeBlockquote();
|
|
339
|
+
out.push(codeBlocks[Number(cb[1])]);
|
|
340
|
+
continue;
|
|
341
|
+
}
|
|
274
342
|
|
|
275
|
-
//
|
|
276
|
-
|
|
343
|
+
// Blank line closes open blocks
|
|
344
|
+
if (/^\s*$/.test(line)) {
|
|
345
|
+
flushParagraph();
|
|
346
|
+
closeLists();
|
|
347
|
+
closeBlockquote();
|
|
348
|
+
continue;
|
|
349
|
+
}
|
|
277
350
|
|
|
278
|
-
//
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
351
|
+
// Horizontal rule: ---, ***, ___ (3+)
|
|
352
|
+
if (/^\s*([-*_])(?:\s*\1){2,}\s*$/.test(line)) {
|
|
353
|
+
flushParagraph();
|
|
354
|
+
closeLists();
|
|
355
|
+
closeBlockquote();
|
|
356
|
+
out.push("<hr>");
|
|
357
|
+
continue;
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
// ATX header: # .. ######
|
|
361
|
+
const header = line.match(/^\s*(#{1,6})\s+(.+?)\s*#*\s*$/);
|
|
362
|
+
if (header) {
|
|
363
|
+
flushParagraph();
|
|
364
|
+
closeLists();
|
|
365
|
+
closeBlockquote();
|
|
366
|
+
const level = header[1].length;
|
|
367
|
+
out.push(`<h${level}>${applyInlineMarkdown(header[2])}</h${level}>`);
|
|
368
|
+
continue;
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
// Blockquote: > text
|
|
372
|
+
const bq = line.match(/^\s*>\s?(.*)$/);
|
|
373
|
+
if (bq) {
|
|
374
|
+
flushParagraph();
|
|
375
|
+
closeLists();
|
|
376
|
+
if (!inBlockquote) {
|
|
377
|
+
out.push("<blockquote>");
|
|
378
|
+
inBlockquote = true;
|
|
283
379
|
}
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
380
|
+
out.push(`<p>${applyInlineMarkdown(bq[1])}</p>`);
|
|
381
|
+
continue;
|
|
382
|
+
}
|
|
383
|
+
closeBlockquote();
|
|
384
|
+
|
|
385
|
+
// Unordered list item: -, *, +
|
|
386
|
+
const ul = line.match(/^\s*[-*+]\s+(.+)$/);
|
|
387
|
+
if (ul) {
|
|
388
|
+
flushParagraph();
|
|
389
|
+
if (inOl) {
|
|
390
|
+
out.push("</ol>");
|
|
391
|
+
inOl = false;
|
|
392
|
+
}
|
|
393
|
+
if (!inUl) {
|
|
394
|
+
out.push("<ul>");
|
|
395
|
+
inUl = true;
|
|
396
|
+
}
|
|
397
|
+
out.push(`<li>${applyInlineMarkdown(ul[1])}</li>`);
|
|
398
|
+
continue;
|
|
399
|
+
}
|
|
287
400
|
|
|
288
|
-
//
|
|
289
|
-
.
|
|
401
|
+
// Ordered list item: 1. or 1)
|
|
402
|
+
const ol = line.match(/^\s*\d+[.)]\s+(.+)$/);
|
|
403
|
+
if (ol) {
|
|
404
|
+
flushParagraph();
|
|
405
|
+
if (inUl) {
|
|
406
|
+
out.push("</ul>");
|
|
407
|
+
inUl = false;
|
|
408
|
+
}
|
|
409
|
+
if (!inOl) {
|
|
410
|
+
out.push("<ol>");
|
|
411
|
+
inOl = true;
|
|
412
|
+
}
|
|
413
|
+
out.push(`<li>${applyInlineMarkdown(ol[1])}</li>`);
|
|
414
|
+
continue;
|
|
415
|
+
}
|
|
290
416
|
|
|
291
|
-
//
|
|
292
|
-
|
|
417
|
+
// Otherwise accumulate into the current paragraph
|
|
418
|
+
closeLists();
|
|
419
|
+
paragraph.push(line.trim());
|
|
420
|
+
}
|
|
293
421
|
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
422
|
+
flushParagraph();
|
|
423
|
+
closeLists();
|
|
424
|
+
closeBlockquote();
|
|
425
|
+
|
|
426
|
+
let html = out.join("\n");
|
|
427
|
+
|
|
428
|
+
// Restore inline code placeholders
|
|
429
|
+
html = html.replace(/\u0000IC(\d+)\u0000/g, (_m, i) => inlineCodes[Number(i)]);
|
|
297
430
|
|
|
298
|
-
return html;
|
|
431
|
+
return html.trim();
|
|
299
432
|
}
|
|
300
433
|
|
|
301
434
|
export function convertHTMLToMarkdown(html) {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// @ts-nocheck
|
|
2
2
|
import { convertHTMLToBasicHTML } from "../html-to-content/html-to-basic-html";
|
|
3
|
-
import {
|
|
3
|
+
import { convertMarkdownToFormattedHTML } from "../html-to-content/html-utils";
|
|
4
4
|
import grab from "../utils/grab";
|
|
5
5
|
|
|
6
6
|
/**
|
|
@@ -299,7 +299,10 @@ export async function scrapeJINA(url) {
|
|
|
299
299
|
var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
|
|
300
300
|
articleExtract = match ? match[1] : articleExtract;
|
|
301
301
|
|
|
302
|
-
|
|
302
|
+
// JINA returns the article body as Markdown. Convert it to formatted HTML
|
|
303
|
+
// using regexp-based Markdown detection so headers, lists, links, emphasis,
|
|
304
|
+
// and code render correctly downstream.
|
|
305
|
+
articleExtract = convertMarkdownToFormattedHTML(articleExtract);
|
|
303
306
|
|
|
304
307
|
if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
|
|
305
308
|
|