extract-webpage 1.2.109 → 1.2.111

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -55,6 +55,27 @@ export declare function convertURLToAbsoluteURL(base: any, relative: any): any;
55
55
  * // <ul><li>List item 1</li><li>List item 2</li></ul>
56
56
  */
57
57
  export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): any;
58
+ /**
59
+ * Convert a Markdown document to formatted HTML using regular expressions to
60
+ * detect Markdown syntax. Unlike {@link convertMarkdownToHTML} (which relies on
61
+ * the `marked` library), this is a dependency-free, self-contained converter
62
+ * intended for post-processing content returned as Markdown (e.g. from the
63
+ * JINA reader fallback in the scraper).
64
+ *
65
+ * Supported block elements: ATX headers (`#`..`######`), fenced code blocks
66
+ * (```lang), blockquotes (`>`), unordered lists (`-`, `*`, `+`), ordered lists
67
+ * (`1.`, `1)`), horizontal rules (`---`, `***`, `___`) and paragraphs.
68
+ * Supported inline elements: bold, italic, strikethrough, inline code, images
69
+ * and links.
70
+ *
71
+ * @param {string} markdown - The Markdown content to convert.
72
+ * @returns {string} The resulting formatted HTML string.
73
+ * @category HTML Utilities
74
+ * @example
75
+ * convertMarkdownToFormattedHTML("# Title\n\nSome **bold** text.");
76
+ * // => "<h1>Title</h1>\n<p>Some <strong>bold</strong> text.</p>"
77
+ */
78
+ export declare function convertMarkdownToFormattedHTML(markdown: any): string;
58
79
  export declare function convertHTMLToMarkdown(html: any): any;
59
80
  /**
60
81
  * Copy HTML to clipboard. When pasting into rich text field,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.109",
3
+ "version": "1.2.111",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -72,11 +72,11 @@
72
72
  "dependencies": {
73
73
  "@huggingface/transformers": "^3.8.1",
74
74
  "ai": "^5.0.0",
75
- "chat-agent-toolkit": "^1.2.109",
75
+ "chat-agent-toolkit": "^1.2.111",
76
76
  "chrono-node": "^2.9.0",
77
77
  "drizzle-orm": "^0.45.1",
78
- "extract-pdf": "^0.1.96",
79
- "extract-youtube": "^1.0.94",
78
+ "extract-pdf": "^0.1.98",
79
+ "extract-youtube": "^1.0.96",
80
80
  "html-entities": "^2.6.0",
81
81
  "js-yaml": "^4.1.1",
82
82
  "jsdom": "^28.1.0",
@@ -206,96 +206,229 @@ export function convertMarkdownToHTML(content, toHtml = true) {
206
206
  if (!toHtml) return convertHTMLToMarkdown(content);
207
207
 
208
208
  return content?.length ? marked.parse(content) : "";
209
+ }
209
210
 
210
- var html = contentconvertMarkdownToHTML
211
- // Convert headers
212
- .replace(/^(#{1,6})\s(.+)$/gm, (match, hashes, content) => {
213
- const level = hashes.length;
214
- return `<h${level}>${content.trim()}</h${level}>`;
215
- })
211
+ /**
212
+ * Escape the HTML-significant characters in a raw string so it can be
213
+ * safely embedded inside generated HTML (used for code spans/blocks).
214
+ * @param {string} str
215
+ * @returns {string}
216
+ */
217
+ function escapeHTMLChars(str) {
218
+ return String(str)
219
+ .replace(/&/g, "&amp;")
220
+ .replace(/</g, "&lt;")
221
+ .replace(/>/g, "&gt;");
222
+ }
216
223
 
217
- // Convert bold text
218
- .replace(/\*\*(.+?)\*\*/g, "<b>$1</b>")
224
+ /**
225
+ * Apply inline-level Markdown regexp replacements (images, links, bold,
226
+ * italic, strikethrough) to a single already-block-parsed line of text.
227
+ * Inline code spans are expected to already be swapped out for placeholders
228
+ * so their contents are never touched here.
229
+ * @param {string} text
230
+ * @returns {string}
231
+ */
232
+ function applyInlineMarkdown(text) {
233
+ return text
234
+ // Images: ![alt](src "title") -- must run before links
235
+ .replace(
236
+ /!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/g,
237
+ (_m, alt, src, title) =>
238
+ `<img src="${src}" alt="${alt}"${title ? ` title="${title}"` : ""} />`
239
+ )
240
+ // Links: [text](href "title")
241
+ .replace(
242
+ /\[([^\]]+)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/g,
243
+ (_m, label, href, title) =>
244
+ `<a href="${href}"${title ? ` title="${title}"` : ""}>${label}</a>`
245
+ )
246
+ // Bold: **text** or __text__
247
+ .replace(/\*\*([^*]+)\*\*/g, "<strong>$1</strong>")
248
+ .replace(/__([^_]+)__/g, "<strong>$1</strong>")
249
+ // Italic: *text* or _text_ (avoid matching inside words for `_`)
250
+ .replace(/\*([^*\n]+)\*/g, "<em>$1</em>")
251
+ .replace(/(^|[^A-Za-z0-9_])_([^_\n]+)_(?=[^A-Za-z0-9_]|$)/g, "$1<em>$2</em>")
252
+ // Strikethrough: ~~text~~
253
+ .replace(/~~([^~]+)~~/g, "<del>$1</del>");
254
+ }
219
255
 
220
- // Convert italic text
221
- .replace(/\*(.+?)\*/g, "<em>$1</em>")
256
+ /**
257
+ * Convert a Markdown document to formatted HTML using regular expressions to
258
+ * detect Markdown syntax. Unlike {@link convertMarkdownToHTML} (which relies on
259
+ * the `marked` library), this is a dependency-free, self-contained converter
260
+ * intended for post-processing content returned as Markdown (e.g. from the
261
+ * JINA reader fallback in the scraper).
262
+ *
263
+ * Supported block elements: ATX headers (`#`..`######`), fenced code blocks
264
+ * (```lang), blockquotes (`>`), unordered lists (`-`, `*`, `+`), ordered lists
265
+ * (`1.`, `1)`), horizontal rules (`---`, `***`, `___`) and paragraphs.
266
+ * Supported inline elements: bold, italic, strikethrough, inline code, images
267
+ * and links.
268
+ *
269
+ * @param {string} markdown - The Markdown content to convert.
270
+ * @returns {string} The resulting formatted HTML string.
271
+ * @category HTML Utilities
272
+ * @example
273
+ * convertMarkdownToFormattedHTML("# Title\n\nSome **bold** text.");
274
+ * // => "<h1>Title</h1>\n<p>Some <strong>bold</strong> text.</p>"
275
+ */
276
+ export function convertMarkdownToFormattedHTML(markdown) {
277
+ if (!markdown || typeof markdown !== "string") return "";
278
+
279
+ let text = markdown.replace(/\r\n?/g, "\n");
280
+
281
+ // 1. Pull fenced code blocks out first so their contents are never parsed
282
+ // as Markdown. Each is replaced by a placeholder restored at the end.
283
+ const codeBlocks = [];
284
+ text = text.replace(
285
+ /```([^\n`]*)\n([\s\S]*?)```/g,
286
+ (_m, lang, code) => {
287
+ const language = (lang || "").trim();
288
+ const cls = language ? ` class="language-${language}"` : "";
289
+ const body = escapeHTMLChars(code.replace(/\n$/, ""));
290
+ codeBlocks.push(`<pre><code${cls}>${body}</code></pre>`);
291
+ return `\u0000CB${codeBlocks.length - 1}\u0000`;
292
+ }
293
+ );
222
294
 
223
- // Convert unordered lists
224
- .replace(/^\s*\*\s(.+)$/gm, "<li>$1</li>")
225
- .replace(/(<li>.*<\/li>)/s, "<ul>$1</ul>")
295
+ // 2. Pull inline code spans out next for the same reason.
296
+ const inlineCodes = [];
297
+ text = text.replace(/`([^`\n]+)`/g, (_m, code) => {
298
+ inlineCodes.push(`<code>${escapeHTMLChars(code)}</code>`);
299
+ return `\u0000IC${inlineCodes.length - 1}\u0000`;
300
+ });
226
301
 
227
- // Convert ordered lists
228
- .replace(/^\s*\d+\.\s(.+)$/gm, "<li>$1</li>")
229
- .replace(/(<li>.*<\/li>)/s, "<ol>$1</ol>")
230
-
231
- // Convert horizontal rules (---, ___, ***)
232
- .replace(/^[-_*]{3,}\s*$/gm, "<hr>")
233
-
234
- // Convert code blocks (```)
235
- .replace(/```([^`]+)```/g, "<code>$1</code>")
236
-
237
- .replace(/```(\w*)\n([\s\S]*?)```/g, function (match, lang, code) {
238
- code = code
239
- .trim()
240
- // Remove leading whitespace from each line while preserving relative indentation
241
- .replace(/^[ \t]*/gm, "")
242
- // Encode HTML special characters
243
- .replace(/&/g, "&amp;")
244
- .replace(/</g, "&lt;")
245
- .replace(/>/g, "&gt;")
246
- .replace(/"/g, "&quot;")
247
- .replace(/'/g, "&#39;");
248
-
249
- return lang
250
- ? `<code class="language-${lang}">${code}</code>`
251
- : `<code>${code}</code>`;
252
- })
302
+ const lines = text.split("\n");
303
+ const out = [];
304
+ let inUl = false;
305
+ let inOl = false;
306
+ let inBlockquote = false;
307
+ let paragraph = [];
308
+
309
+ const flushParagraph = () => {
310
+ if (paragraph.length) {
311
+ out.push(`<p>${applyInlineMarkdown(paragraph.join(" "))}</p>`);
312
+ paragraph = [];
313
+ }
314
+ };
315
+ const closeLists = () => {
316
+ if (inUl) {
317
+ out.push("</ul>");
318
+ inUl = false;
319
+ }
320
+ if (inOl) {
321
+ out.push("</ol>");
322
+ inOl = false;
323
+ }
324
+ };
325
+ const closeBlockquote = () => {
326
+ if (inBlockquote) {
327
+ out.push("</blockquote>");
328
+ inBlockquote = false;
329
+ }
330
+ };
253
331
 
254
- // Handle inline code blocks
255
- .replace(
256
- /(^|[^\\])(`+)([^\r]*?[^`])\2(?!`)/gm,
257
- function (match, pre, backticks, code) {
258
- code = code
259
- .trim()
260
- // Remove leading and trailing whitespace
261
- .replace(/^[ \t]*/g, "")
262
- .replace(/[ \t]*$/g, "")
263
- // Encode HTML special characters
264
-
265
- .replace(/&/g, "&amp;")
266
- .replace(/</g, "&lt;")
267
- .replace(/>/g, "&gt;")
268
- .replace(/"/g, "&quot;")
269
- .replace(/'/g, "&#39;");
270
-
271
- return pre + "<code>" + code + "</code>";
272
- }
273
- )
332
+ for (const line of lines) {
333
+ // Standalone fenced-code-block placeholder line
334
+ const cb = line.match(/^\u0000CB(\d+)\u0000$/);
335
+ if (cb) {
336
+ flushParagraph();
337
+ closeLists();
338
+ closeBlockquote();
339
+ out.push(codeBlocks[Number(cb[1])]);
340
+ continue;
341
+ }
274
342
 
275
- // Convert inline code (`)
276
- .replace(/`([^`]+)`/g, "<code>$1</code>")
343
+ // Blank line closes open blocks
344
+ if (/^\s*$/.test(line)) {
345
+ flushParagraph();
346
+ closeLists();
347
+ closeBlockquote();
348
+ continue;
349
+ }
277
350
 
278
- // Convert paragraphs
279
- .split("\n\n")
280
- .map((para) => {
281
- if (!para.startsWith("<")) {
282
- return `<p>${para.trim()}</p>`;
351
+ // Horizontal rule: ---, ***, ___ (3+)
352
+ if (/^\s*([-*_])(?:\s*\1){2,}\s*$/.test(line)) {
353
+ flushParagraph();
354
+ closeLists();
355
+ closeBlockquote();
356
+ out.push("<hr>");
357
+ continue;
358
+ }
359
+
360
+ // ATX header: # .. ######
361
+ const header = line.match(/^\s*(#{1,6})\s+(.+?)\s*#*\s*$/);
362
+ if (header) {
363
+ flushParagraph();
364
+ closeLists();
365
+ closeBlockquote();
366
+ const level = header[1].length;
367
+ out.push(`<h${level}>${applyInlineMarkdown(header[2])}</h${level}>`);
368
+ continue;
369
+ }
370
+
371
+ // Blockquote: > text
372
+ const bq = line.match(/^\s*>\s?(.*)$/);
373
+ if (bq) {
374
+ flushParagraph();
375
+ closeLists();
376
+ if (!inBlockquote) {
377
+ out.push("<blockquote>");
378
+ inBlockquote = true;
283
379
  }
284
- return para;
285
- })
286
- .join("\n")
380
+ out.push(`<p>${applyInlineMarkdown(bq[1])}</p>`);
381
+ continue;
382
+ }
383
+ closeBlockquote();
384
+
385
+ // Unordered list item: -, *, +
386
+ const ul = line.match(/^\s*[-*+]\s+(.+)$/);
387
+ if (ul) {
388
+ flushParagraph();
389
+ if (inOl) {
390
+ out.push("</ol>");
391
+ inOl = false;
392
+ }
393
+ if (!inUl) {
394
+ out.push("<ul>");
395
+ inUl = true;
396
+ }
397
+ out.push(`<li>${applyInlineMarkdown(ul[1])}</li>`);
398
+ continue;
399
+ }
287
400
 
288
- // Convert images
289
- .replace(/\!\[(.*?)\]\((.*?)\)/g, '<img src="$2" alt="$1" />')
401
+ // Ordered list item: 1. or 1)
402
+ const ol = line.match(/^\s*\d+[.)]\s+(.+)$/);
403
+ if (ol) {
404
+ flushParagraph();
405
+ if (inUl) {
406
+ out.push("</ul>");
407
+ inUl = false;
408
+ }
409
+ if (!inOl) {
410
+ out.push("<ol>");
411
+ inOl = true;
412
+ }
413
+ out.push(`<li>${applyInlineMarkdown(ol[1])}</li>`);
414
+ continue;
415
+ }
290
416
 
291
- // Convert links
292
- .replace(/\[(.*?)\]\((.*?)\)/g, '<a href="$2">$1</a>')
417
+ // Otherwise accumulate into the current paragraph
418
+ closeLists();
419
+ paragraph.push(line.trim());
420
+ }
293
421
 
294
- // Clean up extra newlines
295
- .replace(/\n\s*\n/g, "\n")
296
- .trim();
422
+ flushParagraph();
423
+ closeLists();
424
+ closeBlockquote();
425
+
426
+ let html = out.join("\n");
427
+
428
+ // Restore inline code placeholders
429
+ html = html.replace(/\u0000IC(\d+)\u0000/g, (_m, i) => inlineCodes[Number(i)]);
297
430
 
298
- return html;
431
+ return html.trim();
299
432
  }
300
433
 
301
434
  export function convertHTMLToMarkdown(html) {
@@ -1,6 +1,6 @@
1
1
  // @ts-nocheck
2
2
  import { convertHTMLToBasicHTML } from "../html-to-content/html-to-basic-html";
3
- import { convertMarkdownToHTML } from "../html-to-content/html-utils";
3
+ import { convertMarkdownToFormattedHTML } from "../html-to-content/html-utils";
4
4
  import grab from "../utils/grab";
5
5
 
6
6
  /**
@@ -299,7 +299,10 @@ export async function scrapeJINA(url) {
299
299
  var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
300
300
  articleExtract = match ? match[1] : articleExtract;
301
301
 
302
- articleExtract = convertMarkdownToHTML(articleExtract);
302
+ // JINA returns the article body as Markdown. Convert it to formatted HTML
303
+ // using regexp-based Markdown detection so headers, lists, links, emphasis,
304
+ // and code render correctly downstream.
305
+ articleExtract = convertMarkdownToFormattedHTML(articleExtract);
303
306
 
304
307
  if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
305
308