extract-webpage 1.2.270 → 1.2.272
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +2 -2
- package/dist/extract-webpage.es.js.map +1 -1
- package/dist/html-to-content/html-utils.d.ts +39 -5
- package/package.json +4 -4
- package/src/html-to-content/__tests__/html-utils.test.ts +220 -0
- package/src/html-to-content/html-to-content.ts +13 -0
- package/src/html-to-content/html-utils.ts +238 -22
- package/src/url-to-content/url-to-content.ts +6 -2
- package/src/url-to-content/url-to-html.ts +8 -3
|
@@ -4,7 +4,10 @@
|
|
|
4
4
|
* with bot-detection handling and Cloudflare/JINA fallbacks, plus robots.txt checking.
|
|
5
5
|
*/
|
|
6
6
|
import { convertHTMLToBasicHTML } from "../html-to-content/html-to-basic-html";
|
|
7
|
-
import {
|
|
7
|
+
import {
|
|
8
|
+
convertMarkdownToFormattedHTML,
|
|
9
|
+
removeMarkdownNavigation,
|
|
10
|
+
} from "../html-to-content/html-utils";
|
|
8
11
|
import grab from "../utils/grab";
|
|
9
12
|
|
|
10
13
|
/**
|
|
@@ -312,9 +315,11 @@ export async function scrapeJINA(url) {
|
|
|
312
315
|
var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
|
|
313
316
|
articleExtract = match ? match[1] : articleExtract;
|
|
314
317
|
|
|
315
|
-
// JINA returns the article body as Markdown.
|
|
316
|
-
//
|
|
318
|
+
// JINA returns the article body as Markdown. Strip reader metadata and
|
|
319
|
+
// navigation-only link blocks first, then convert to formatted HTML using
|
|
320
|
+
// regexp-based Markdown detection so headers, lists, links, emphasis,
|
|
317
321
|
// and code render correctly downstream.
|
|
322
|
+
articleExtract = removeMarkdownNavigation(articleExtract);
|
|
318
323
|
articleExtract = convertMarkdownToFormattedHTML(articleExtract);
|
|
319
324
|
|
|
320
325
|
if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
|