extract-webpage 1.2.270 → 1.2.272

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,10 @@
4
4
  * with bot-detection handling and Cloudflare/JINA fallbacks, plus robots.txt checking.
5
5
  */
6
6
  import { convertHTMLToBasicHTML } from "../html-to-content/html-to-basic-html";
7
- import { convertMarkdownToFormattedHTML } from "../html-to-content/html-utils";
7
+ import {
8
+ convertMarkdownToFormattedHTML,
9
+ removeMarkdownNavigation,
10
+ } from "../html-to-content/html-utils";
8
11
  import grab from "../utils/grab";
9
12
 
10
13
  /**
@@ -312,9 +315,11 @@ export async function scrapeJINA(url) {
312
315
  var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
313
316
  articleExtract = match ? match[1] : articleExtract;
314
317
 
315
- // JINA returns the article body as Markdown. Convert it to formatted HTML
316
- // using regexp-based Markdown detection so headers, lists, links, emphasis,
318
+ // JINA returns the article body as Markdown. Strip reader metadata and
319
+ // navigation-only link blocks first, then convert to formatted HTML using
320
+ // regexp-based Markdown detection so headers, lists, links, emphasis,
317
321
  // and code render correctly downstream.
322
+ articleExtract = removeMarkdownNavigation(articleExtract);
318
323
  articleExtract = convertMarkdownToFormattedHTML(articleExtract);
319
324
 
320
325
  if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;