extract-webpage 1.2.269 → 1.2.271
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +2 -2
- package/dist/extract-webpage.es.js.map +1 -1
- package/dist/html-to-content/html-utils.d.ts +39 -5
- package/package.json +4 -4
- package/src/html-to-content/__tests__/html-utils.test.ts +220 -0
- package/src/html-to-content/html-to-content.ts +13 -0
- package/src/html-to-content/html-utils.ts +238 -22
- package/src/url-to-content/url-to-content.ts +6 -2
- package/src/url-to-content/url-to-html.ts +8 -3
|
@@ -55,6 +55,39 @@ export declare function convertURLToAbsoluteURL(base: any, relative: any): any;
|
|
|
55
55
|
* // <ul><li>List item 1</li><li>List item 2</li></ul>
|
|
56
56
|
*/
|
|
57
57
|
export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): any;
|
|
58
|
+
/**
|
|
59
|
+
* Detect whether a string is Markdown (rather than HTML or plain text) using
|
|
60
|
+
* regexp checks. Content that is dominated by HTML tags is never treated as
|
|
61
|
+
* Markdown, so real scraped pages pass through untouched; text needs at least
|
|
62
|
+
* two distinct Markdown syntax signals (or several links/images in Markdown
|
|
63
|
+
* form) to qualify. Used to catch scraper responses (e.g. the JINA reader or
|
|
64
|
+
* proxies wrapping it) that return Markdown in place of HTML, so it can be
|
|
65
|
+
* converted before main-content extraction — otherwise the article panel
|
|
66
|
+
* renders raw `[text](url)` syntax.
|
|
67
|
+
*
|
|
68
|
+
* @param {string} text - The content to test.
|
|
69
|
+
* @returns {boolean} True when the content should be parsed as Markdown.
|
|
70
|
+
* @category HTML Utilities
|
|
71
|
+
* @example
|
|
72
|
+
* detectMarkdown("# Title\n\nSome **bold** text.") // true
|
|
73
|
+
* detectMarkdown("<html><body><p>Hi</p></body></html>") // false
|
|
74
|
+
*/
|
|
75
|
+
export declare function detectMarkdown(text: any): boolean;
|
|
76
|
+
/**
|
|
77
|
+
* Remove extra non-article content from a Markdown extraction using regexp
|
|
78
|
+
* checks: JINA reader metadata lines, cookie/consent and navigation phrases,
|
|
79
|
+
* and runs of consecutive link-only lines (menus, breadcrumbs, "related"
|
|
80
|
+
* link farms) whose targets are mostly relative site navigation. Standalone
|
|
81
|
+
* links inside prose are kept.
|
|
82
|
+
*
|
|
83
|
+
* @param {string} markdown - The Markdown content to clean.
|
|
84
|
+
* @returns {string} The cleaned Markdown.
|
|
85
|
+
* @category HTML Utilities
|
|
86
|
+
* @example
|
|
87
|
+
* removeMarkdownNavigation("Title: Page\n[Skip to content](#main)\nReal text")
|
|
88
|
+
* // => "Real text"
|
|
89
|
+
*/
|
|
90
|
+
export declare function removeMarkdownNavigation(markdown: any): string;
|
|
58
91
|
/**
|
|
59
92
|
* Convert a Markdown document to formatted HTML using regular expressions to
|
|
60
93
|
* detect Markdown syntax. Unlike {@link convertMarkdownToHTML} (which relies on
|
|
@@ -62,11 +95,12 @@ export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): a
|
|
|
62
95
|
* intended for post-processing content returned as Markdown (e.g. from the
|
|
63
96
|
* JINA reader fallback in the scraper).
|
|
64
97
|
*
|
|
65
|
-
* Supported block elements: ATX headers (`#`..`######`),
|
|
66
|
-
* (
|
|
67
|
-
* (
|
|
68
|
-
*
|
|
69
|
-
*
|
|
98
|
+
* Supported block elements: ATX headers (`#`..`######`), setext headers
|
|
99
|
+
* (`===`/`---` underlines), fenced code blocks (```lang), blockquotes (`>`),
|
|
100
|
+
* unordered lists (`-`, `*`, `+`), ordered lists (`1.`, `1)`), pipe tables,
|
|
101
|
+
* horizontal rules (`---`, `***`, `___`) and paragraphs.
|
|
102
|
+
* Supported inline elements: bold, italic, strikethrough, inline code, images,
|
|
103
|
+
* links, linked images (`[](href)`) and autolinks (`<https://…>`).
|
|
70
104
|
*
|
|
71
105
|
* @param {string} markdown - The Markdown content to convert.
|
|
72
106
|
* @returns {string} The resulting formatted HTML string.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-webpage",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.271",
|
|
4
4
|
"module": "./dist/extract-webpage.es.js",
|
|
5
5
|
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
6
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
@@ -74,11 +74,11 @@
|
|
|
74
74
|
"dependencies": {
|
|
75
75
|
"@huggingface/transformers": "^3.8.1",
|
|
76
76
|
"ai": "^5.0.0",
|
|
77
|
-
"chat-agent-toolkit": "^1.2.
|
|
77
|
+
"chat-agent-toolkit": "^1.2.274",
|
|
78
78
|
"chrono-node": "^2.9.0",
|
|
79
79
|
"drizzle-orm": "^0.45.1",
|
|
80
|
-
"extract-pdf": "^0.1.
|
|
81
|
-
"extract-youtube": "^1.0.
|
|
80
|
+
"extract-pdf": "^0.1.261",
|
|
81
|
+
"extract-youtube": "^1.0.256",
|
|
82
82
|
"html-entities": "^2.6.0",
|
|
83
83
|
"js-yaml": "^4.1.1",
|
|
84
84
|
"jsdom": "^28.1.0",
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Tests for regexp-based Markdown detection, navigation/noise
|
|
3
|
+
* removal, and Markdown-to-HTML conversion used on JINA-style extractions.
|
|
4
|
+
*/
|
|
5
|
+
import { describe, expect, it } from "vitest";
|
|
6
|
+
import {
|
|
7
|
+
convertMarkdownToFormattedHTML,
|
|
8
|
+
detectMarkdown,
|
|
9
|
+
removeMarkdownNavigation,
|
|
10
|
+
} from "../html-utils";
|
|
11
|
+
|
|
12
|
+
describe("detectMarkdown", () => {
|
|
13
|
+
it("detects typical markdown documents", () => {
|
|
14
|
+
expect(
|
|
15
|
+
detectMarkdown("# Title\n\nSome **bold** text.\n\n- item 1\n- item 2")
|
|
16
|
+
).toBe(true);
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
it("detects JINA reader output", () => {
|
|
20
|
+
const jina = [
|
|
21
|
+
"Markdown Content:",
|
|
22
|
+
"Michael Jackson",
|
|
23
|
+
"===============",
|
|
24
|
+
"",
|
|
25
|
+
"[](/album/1)",
|
|
26
|
+
"[Thriller](/album/1C2h7mLntPSeVYciMRTF4a)",
|
|
27
|
+
"[Bad 25th Anniversary](/album/24TAupSNVWSAHL0R7n71vm)",
|
|
28
|
+
].join("\n");
|
|
29
|
+
expect(detectMarkdown(jina)).toBe(true);
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
it("detects link-heavy markdown with a single syntax signal", () => {
|
|
33
|
+
const links = [
|
|
34
|
+
"[Dangerous](/album/0oX4SealMgNXrvRDhqqOKg) some text",
|
|
35
|
+
"more [Invincible](/album/52E4RP7XDzalpIrOgSTgiQ) text",
|
|
36
|
+
"and [HIStory](/album/3OBhnTLrvkoEEETjFA3Qfk) too",
|
|
37
|
+
].join("\n");
|
|
38
|
+
expect(detectMarkdown(links)).toBe(true);
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
it("does not flag HTML documents", () => {
|
|
42
|
+
expect(
|
|
43
|
+
detectMarkdown(
|
|
44
|
+
"<html><body><h1>Title</h1><p>Some **bold** text</p><ul><li>a</li></ul></body></html>"
|
|
45
|
+
)
|
|
46
|
+
).toBe(false);
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
it("does not flag HTML fragments with several closing tags", () => {
|
|
50
|
+
expect(
|
|
51
|
+
detectMarkdown(
|
|
52
|
+
"<div><p>One - two</p><p>Three</p><p>- looks like a list</p></div>"
|
|
53
|
+
)
|
|
54
|
+
).toBe(false);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
it("does not flag plain text", () => {
|
|
58
|
+
expect(
|
|
59
|
+
detectMarkdown("Just a plain sentence.\nAnother plain sentence here.")
|
|
60
|
+
).toBe(false);
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it("handles empty and non-string input", () => {
|
|
64
|
+
expect(detectMarkdown("")).toBe(false);
|
|
65
|
+
expect(detectMarkdown(null as any)).toBe(false);
|
|
66
|
+
expect(detectMarkdown(undefined as any)).toBe(false);
|
|
67
|
+
});
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
describe("removeMarkdownNavigation", () => {
|
|
71
|
+
it("removes JINA reader metadata lines", () => {
|
|
72
|
+
const input = [
|
|
73
|
+
"Title: Some Page",
|
|
74
|
+
"URL Source: https://example.com",
|
|
75
|
+
"Published Time: 2024-01-01",
|
|
76
|
+
"Markdown Content:",
|
|
77
|
+
"",
|
|
78
|
+
"Real article text here.",
|
|
79
|
+
].join("\n");
|
|
80
|
+
const output = removeMarkdownNavigation(input);
|
|
81
|
+
expect(output).toBe("Real article text here.");
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
it("removes navigation phrase lines", () => {
|
|
85
|
+
const input = [
|
|
86
|
+
"[Skip to content](#main)",
|
|
87
|
+
"[Sign in](/login)",
|
|
88
|
+
"Real paragraph of the article.",
|
|
89
|
+
"[Back to top](#top)",
|
|
90
|
+
].join("\n");
|
|
91
|
+
const output = removeMarkdownNavigation(input);
|
|
92
|
+
expect(output).toBe("Real paragraph of the article.");
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
it("removes runs of 3+ link-only lines pointing at relative URLs", () => {
|
|
96
|
+
const input = [
|
|
97
|
+
"Intro paragraph.",
|
|
98
|
+
"",
|
|
99
|
+
"[](/album/1)",
|
|
100
|
+
"[Thriller](/album/1)",
|
|
101
|
+
"[Bad](/album/2)",
|
|
102
|
+
"[Dangerous](/album/3)",
|
|
103
|
+
"",
|
|
104
|
+
"Closing paragraph.",
|
|
105
|
+
].join("\n");
|
|
106
|
+
const output = removeMarkdownNavigation(input);
|
|
107
|
+
expect(output).toContain("Intro paragraph.");
|
|
108
|
+
expect(output).toContain("Closing paragraph.");
|
|
109
|
+
expect(output).not.toContain("/album/");
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
it("keeps short link-only runs and external citation links", () => {
|
|
113
|
+
const input = [
|
|
114
|
+
"See sources:",
|
|
115
|
+
"[Source A](https://a.example.com)",
|
|
116
|
+
"[Source B](https://b.example.com)",
|
|
117
|
+
"[Source C](https://c.example.com)",
|
|
118
|
+
].join("\n");
|
|
119
|
+
const output = removeMarkdownNavigation(input);
|
|
120
|
+
expect(output).toContain("Source A");
|
|
121
|
+
expect(output).toContain("Source C");
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
it("keeps links inside prose", () => {
|
|
125
|
+
const input = "This mentions [a page](/internal/path) inside a sentence.";
|
|
126
|
+
expect(removeMarkdownNavigation(input)).toBe(input);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
it("leaves fenced code blocks untouched", () => {
|
|
130
|
+
const input = [
|
|
131
|
+
"```",
|
|
132
|
+
"Title: not metadata, just code",
|
|
133
|
+
"[a](/x)",
|
|
134
|
+
"[b](/y)",
|
|
135
|
+
"[c](/z)",
|
|
136
|
+
"```",
|
|
137
|
+
].join("\n");
|
|
138
|
+
const output = removeMarkdownNavigation(input);
|
|
139
|
+
expect(output).toContain("Title: not metadata, just code");
|
|
140
|
+
expect(output).toContain("[b](/y)");
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
it("handles empty and non-string input", () => {
|
|
144
|
+
expect(removeMarkdownNavigation("")).toBe("");
|
|
145
|
+
expect(removeMarkdownNavigation(null as any)).toBe("");
|
|
146
|
+
});
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
describe("convertMarkdownToFormattedHTML", () => {
|
|
150
|
+
it("converts linked images [](href)", () => {
|
|
151
|
+
const html = convertMarkdownToFormattedHTML(
|
|
152
|
+
"[](https://example.com/album/1)"
|
|
153
|
+
);
|
|
154
|
+
expect(html).toContain(
|
|
155
|
+
'<a href="https://example.com/album/1"><img src="https://i.scdn.co/image/abc" alt="Cover" /></a>'
|
|
156
|
+
);
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it("converts images with empty alt text", () => {
|
|
160
|
+
const html = convertMarkdownToFormattedHTML(
|
|
161
|
+
""
|
|
162
|
+
);
|
|
163
|
+
expect(html).toContain('<img src="https://i.scdn.co/image/abc" alt="" />');
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
it("converts setext headers", () => {
|
|
167
|
+
const html = convertMarkdownToFormattedHTML(
|
|
168
|
+
"Page Title\n===============\n\nSection\n-------\n\nBody text."
|
|
169
|
+
);
|
|
170
|
+
expect(html).toContain("<h1>Page Title</h1>");
|
|
171
|
+
expect(html).toContain("<h2>Section</h2>");
|
|
172
|
+
expect(html).toContain("<p>Body text.</p>");
|
|
173
|
+
});
|
|
174
|
+
|
|
175
|
+
it("still treats --- after a blank line as a horizontal rule", () => {
|
|
176
|
+
const html = convertMarkdownToFormattedHTML("Some text.\n\n---\n\nMore.");
|
|
177
|
+
expect(html).toContain("<hr>");
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
it("converts autolinks", () => {
|
|
181
|
+
const html = convertMarkdownToFormattedHTML("Visit <https://example.com> now.");
|
|
182
|
+
expect(html).toContain('<a href="https://example.com">https://example.com</a>');
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
it("converts pipe tables", () => {
|
|
186
|
+
const html = convertMarkdownToFormattedHTML(
|
|
187
|
+
"| Name | Year |\n| --- | --- |\n| Thriller | 1982 |\n| Bad | 1987 |"
|
|
188
|
+
);
|
|
189
|
+
expect(html).toContain("<table>");
|
|
190
|
+
expect(html).toContain("<th>Name</th>");
|
|
191
|
+
expect(html).toContain("<td>Thriller</td>");
|
|
192
|
+
expect(html).toContain("<td>1987</td>");
|
|
193
|
+
expect(html).toContain("</table>");
|
|
194
|
+
});
|
|
195
|
+
|
|
196
|
+
it("applies inline markdown inside table cells", () => {
|
|
197
|
+
const html = convertMarkdownToFormattedHTML(
|
|
198
|
+
"| Album | Link |\n| --- | --- |\n| **Bad** | [play](/album/2) |"
|
|
199
|
+
);
|
|
200
|
+
expect(html).toContain("<strong>Bad</strong>");
|
|
201
|
+
expect(html).toContain('<a href="/album/2">play</a>');
|
|
202
|
+
});
|
|
203
|
+
|
|
204
|
+
it("converts a JINA-style extraction end to end", () => {
|
|
205
|
+
const jina = [
|
|
206
|
+
"Michael Jackson",
|
|
207
|
+
"===============",
|
|
208
|
+
"",
|
|
209
|
+
"The **King of Pop** released many albums.",
|
|
210
|
+
"",
|
|
211
|
+
"- [Thriller](https://open.spotify.com/album/1)",
|
|
212
|
+
"- [Bad](https://open.spotify.com/album/2)",
|
|
213
|
+
].join("\n");
|
|
214
|
+
const html = convertMarkdownToFormattedHTML(jina);
|
|
215
|
+
expect(html).toContain("<h1>Michael Jackson</h1>");
|
|
216
|
+
expect(html).toContain("<strong>King of Pop</strong>");
|
|
217
|
+
expect(html).toContain('<li><a href="https://open.spotify.com/album/1">Thriller</a></li>');
|
|
218
|
+
expect(html).not.toContain("](");
|
|
219
|
+
});
|
|
220
|
+
});
|
|
@@ -6,6 +6,11 @@
|
|
|
6
6
|
import { parseHTML } from "linkedom";
|
|
7
7
|
import { extractCite } from "../html-to-cite/extract-cite";
|
|
8
8
|
import { convertHTMLToBasicHTML } from "./html-to-basic-html";
|
|
9
|
+
import {
|
|
10
|
+
convertMarkdownToFormattedHTML,
|
|
11
|
+
detectMarkdown,
|
|
12
|
+
removeMarkdownNavigation,
|
|
13
|
+
} from "./html-utils";
|
|
9
14
|
import { extractHumanName } from "../html-to-cite/human-names-recognize";
|
|
10
15
|
import { extractMainContentFromHTML } from "./extract-content/extract-content-readability";
|
|
11
16
|
import { extractMainContentFromHTML2 } from "./extract-content/extract-content-mercury";
|
|
@@ -48,6 +53,14 @@ export function extractContentAndCite(documentOrHTML, options = {}) {
|
|
|
48
53
|
|
|
49
54
|
if (!html) return { error: "No HTML found" };
|
|
50
55
|
|
|
56
|
+
// Some scrape paths (the JINA reader, or proxies wrapping it) return the
|
|
57
|
+
// article as Markdown instead of HTML. Without conversion the sidebar
|
|
58
|
+
// renders raw `[text](url)` syntax, so use regexp checks to detect
|
|
59
|
+
// Markdown, strip navigation/reader-metadata noise, and convert it to
|
|
60
|
+
// formatted HTML before main-content extraction runs.
|
|
61
|
+
if (detectMarkdown(html))
|
|
62
|
+
html = convertMarkdownToFormattedHTML(removeMarkdownNavigation(html));
|
|
63
|
+
|
|
51
64
|
try {
|
|
52
65
|
var content1 = extractMainContentFromHTML(html, options);
|
|
53
66
|
} catch (e) {
|
|
@@ -140,23 +140,23 @@ export function convertURLToAbsoluteURL(base, relative) {
|
|
|
140
140
|
|
|
141
141
|
import { marked } from "marked";
|
|
142
142
|
import Prism from "prismjs";
|
|
143
|
-
import "prismjs/components/prism-markup";
|
|
144
|
-
import "prismjs/components/prism-css";
|
|
145
|
-
import "prismjs/components/prism-javascript";
|
|
146
|
-
import "prismjs/components/prism-typescript";
|
|
147
|
-
import "prismjs/components/prism-jsx";
|
|
148
|
-
import "prismjs/components/prism-tsx";
|
|
149
|
-
import "prismjs/components/prism-python";
|
|
150
|
-
import "prismjs/components/prism-bash";
|
|
151
|
-
import "prismjs/components/prism-json";
|
|
152
|
-
import "prismjs/components/prism-yaml";
|
|
153
|
-
import "prismjs/components/prism-markdown";
|
|
154
|
-
import "prismjs/components/prism-sql";
|
|
155
|
-
import "prismjs/components/prism-rust";
|
|
156
|
-
import "prismjs/components/prism-go";
|
|
157
|
-
import "prismjs/components/prism-java";
|
|
158
|
-
import "prismjs/components/prism-c";
|
|
159
|
-
import "prismjs/components/prism-cpp";
|
|
143
|
+
import "prismjs/components/prism-markup.js";
|
|
144
|
+
import "prismjs/components/prism-css.js";
|
|
145
|
+
import "prismjs/components/prism-javascript.js";
|
|
146
|
+
import "prismjs/components/prism-typescript.js";
|
|
147
|
+
import "prismjs/components/prism-jsx.js";
|
|
148
|
+
import "prismjs/components/prism-tsx.js";
|
|
149
|
+
import "prismjs/components/prism-python.js";
|
|
150
|
+
import "prismjs/components/prism-bash.js";
|
|
151
|
+
import "prismjs/components/prism-json.js";
|
|
152
|
+
import "prismjs/components/prism-yaml.js";
|
|
153
|
+
import "prismjs/components/prism-markdown.js";
|
|
154
|
+
import "prismjs/components/prism-sql.js";
|
|
155
|
+
import "prismjs/components/prism-rust.js";
|
|
156
|
+
import "prismjs/components/prism-go.js";
|
|
157
|
+
import "prismjs/components/prism-java.js";
|
|
158
|
+
import "prismjs/components/prism-c.js";
|
|
159
|
+
import "prismjs/components/prism-cpp.js";
|
|
160
160
|
|
|
161
161
|
// Configure marked once at module load with Prism.js syntax highlighting.
|
|
162
162
|
// marked v17 removed the `highlight` option from setOptions, so highlighting
|
|
@@ -208,6 +208,159 @@ export function convertMarkdownToHTML(content, toHtml = true) {
|
|
|
208
208
|
return content?.length ? marked.parse(content) : "";
|
|
209
209
|
}
|
|
210
210
|
|
|
211
|
+
/**
|
|
212
|
+
* Regexp patterns that identify Markdown syntax. Each entry is a signal that
|
|
213
|
+
* the text is Markdown rather than plain text or HTML — {@link detectMarkdown}
|
|
214
|
+
* counts how many distinct signals match.
|
|
215
|
+
* @private
|
|
216
|
+
*/
|
|
217
|
+
const MARKDOWN_SYNTAX_PATTERNS = [
|
|
218
|
+
/^#{1,6}\s+\S/m, // ATX header: # Title
|
|
219
|
+
/^[^\n]{1,120}\n(?:={3,}|-{3,})[ \t]*$/m, // setext header underline
|
|
220
|
+
/!\[[^\]]*\]\([^)]+\)/, // image: 
|
|
221
|
+
/(?<!!)\[[^\]]+\]\([^)]+\)/, // link: [text](href)
|
|
222
|
+
/^\s{0,3}[-*+]\s+\S/m, // unordered list item
|
|
223
|
+
/^\s{0,3}\d+[.)]\s+\S/m, // ordered list item
|
|
224
|
+
/^```/m, // fenced code block
|
|
225
|
+
/^\s{0,3}>\s+\S/m, // blockquote
|
|
226
|
+
/\*\*[^*\n]+\*\*|__[^_\n]+__/, // bold
|
|
227
|
+
/^\|?[^\n|]*\|[^\n]*\n\|?[\s:]*-{2,}[\s|:-]*$/m, // pipe table header + separator
|
|
228
|
+
/^Markdown Content:$/m, // JINA reader preamble
|
|
229
|
+
];
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Detect whether a string is Markdown (rather than HTML or plain text) using
|
|
233
|
+
* regexp checks. Content that is dominated by HTML tags is never treated as
|
|
234
|
+
* Markdown, so real scraped pages pass through untouched; text needs at least
|
|
235
|
+
* two distinct Markdown syntax signals (or several links/images in Markdown
|
|
236
|
+
* form) to qualify. Used to catch scraper responses (e.g. the JINA reader or
|
|
237
|
+
* proxies wrapping it) that return Markdown in place of HTML, so it can be
|
|
238
|
+
* converted before main-content extraction — otherwise the article panel
|
|
239
|
+
* renders raw `[text](url)` syntax.
|
|
240
|
+
*
|
|
241
|
+
* @param {string} text - The content to test.
|
|
242
|
+
* @returns {boolean} True when the content should be parsed as Markdown.
|
|
243
|
+
* @category HTML Utilities
|
|
244
|
+
* @example
|
|
245
|
+
* detectMarkdown("# Title\n\nSome **bold** text.") // true
|
|
246
|
+
* detectMarkdown("<html><body><p>Hi</p></body></html>") // false
|
|
247
|
+
*/
|
|
248
|
+
export function detectMarkdown(text) {
|
|
249
|
+
if (!text || typeof text !== "string") return false;
|
|
250
|
+
|
|
251
|
+
const sample = text.slice(0, 20000);
|
|
252
|
+
|
|
253
|
+
// HTML dominance check: a document skeleton or a meaningful density of
|
|
254
|
+
// closing tags means this is HTML, not Markdown.
|
|
255
|
+
if (/<!doctype\s+html|<html[\s>]|<body[\s>]/i.test(sample)) return false;
|
|
256
|
+
const closingTags = sample.match(
|
|
257
|
+
/<\/(?:p|div|a|span|h[1-6]|li|ul|ol|table|section|article|nav|em|strong|b|i)>/gi
|
|
258
|
+
);
|
|
259
|
+
if (closingTags && closingTags.length >= 3) return false;
|
|
260
|
+
|
|
261
|
+
let signals = 0;
|
|
262
|
+
for (const pattern of MARKDOWN_SYNTAX_PATTERNS)
|
|
263
|
+
if (pattern.test(sample)) signals++;
|
|
264
|
+
|
|
265
|
+
if (signals >= 2) return true;
|
|
266
|
+
|
|
267
|
+
// A single signal still counts when Markdown links/images repeat — the
|
|
268
|
+
// signature of JINA reader output for link-heavy pages.
|
|
269
|
+
const mdLinks = sample.match(/!?\[[^\]]*\]\([^)]+\)/g);
|
|
270
|
+
return signals >= 1 && !!mdLinks && mdLinks.length >= 3;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Line-level regexps for boilerplate that JINA-style Markdown extractions
|
|
275
|
+
* carry along with the article: reader-preamble metadata and common
|
|
276
|
+
* navigation/chrome phrases rendered as standalone links.
|
|
277
|
+
* @private
|
|
278
|
+
*/
|
|
279
|
+
const MARKDOWN_NOISE_LINE_PATTERNS = [
|
|
280
|
+
/^(?:Title|URL Source|Published Time|Markdown Content|Warning|Links\/Buttons):.*$/i, // JINA metadata
|
|
281
|
+
/^\[?\s*(?:skip to (?:main )?content|main menu|jump to (?:content|navigation)|menu|navigation|sign (?:in|up)|log ?in|register|subscribe(?: now)?|share(?: this)?|tweet|print|download app|open app|back to top|show all|see all|see more|view all|load more|read more|previous|next|home)\s*\]?\s*(?:\([^)]*\))?\s*$/i, // nav phrases, bare or as a single link
|
|
282
|
+
/^(?:\W*\s*)?(?:accept(?: all)?(?: cookies)?|we use cookies.*|cookie (?:policy|settings|preferences)|privacy policy|terms of (?:use|service))\s*(?:\([^)]*\))?\]?\s*$/i, // cookie/legal chrome
|
|
283
|
+
];
|
|
284
|
+
|
|
285
|
+
/**
|
|
286
|
+
* Matches a line that carries no prose of its own — only Markdown links,
|
|
287
|
+
* images, bullets, pipes and punctuation. Runs of these are navigation menus,
|
|
288
|
+
* breadcrumbs, and tag/related-content lists.
|
|
289
|
+
* @private
|
|
290
|
+
*/
|
|
291
|
+
const MARKDOWN_LINK_ONLY_LINE =
|
|
292
|
+
/^\s*(?:[-*+>|]\s*)?(?:(?:\[!\[[^\]]*\]\([^)]*\)\]\([^)]*\)|!?\[[^\]]*\]\([^)]*\))\s*(?:[|•·,/>-]\s*)?)+[.\s]*$/;
|
|
293
|
+
|
|
294
|
+
/**
|
|
295
|
+
* Remove extra non-article content from a Markdown extraction using regexp
|
|
296
|
+
* checks: JINA reader metadata lines, cookie/consent and navigation phrases,
|
|
297
|
+
* and runs of consecutive link-only lines (menus, breadcrumbs, "related"
|
|
298
|
+
* link farms) whose targets are mostly relative site navigation. Standalone
|
|
299
|
+
* links inside prose are kept.
|
|
300
|
+
*
|
|
301
|
+
* @param {string} markdown - The Markdown content to clean.
|
|
302
|
+
* @returns {string} The cleaned Markdown.
|
|
303
|
+
* @category HTML Utilities
|
|
304
|
+
* @example
|
|
305
|
+
* removeMarkdownNavigation("Title: Page\n[Skip to content](#main)\nReal text")
|
|
306
|
+
* // => "Real text"
|
|
307
|
+
*/
|
|
308
|
+
export function removeMarkdownNavigation(markdown) {
|
|
309
|
+
if (!markdown || typeof markdown !== "string") return "";
|
|
310
|
+
|
|
311
|
+
const lines = markdown.replace(/\r\n?/g, "\n").split("\n");
|
|
312
|
+
const kept = [];
|
|
313
|
+
|
|
314
|
+
// First pass: drop noise lines outside fenced code blocks.
|
|
315
|
+
let inFence = false;
|
|
316
|
+
for (const line of lines) {
|
|
317
|
+
if (/^\s*```/.test(line)) {
|
|
318
|
+
inFence = !inFence;
|
|
319
|
+
kept.push(line);
|
|
320
|
+
continue;
|
|
321
|
+
}
|
|
322
|
+
if (
|
|
323
|
+
!inFence &&
|
|
324
|
+
MARKDOWN_NOISE_LINE_PATTERNS.some((pattern) => pattern.test(line.trim()))
|
|
325
|
+
)
|
|
326
|
+
continue;
|
|
327
|
+
kept.push(line);
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
// Second pass: drop runs of 3+ consecutive link-only lines when the run's
|
|
331
|
+
// links point mostly at relative URLs (`/path`, `#anchor`) — the signature
|
|
332
|
+
// of site navigation rather than cited external sources.
|
|
333
|
+
const out = [];
|
|
334
|
+
let run = [];
|
|
335
|
+
inFence = false;
|
|
336
|
+
const flushRun = () => {
|
|
337
|
+
if (!run.length) return;
|
|
338
|
+
const runText = run.join("\n");
|
|
339
|
+
const hrefs = [...runText.matchAll(/!?\[[^\]]*\]\(([^)\s]*)/g)].map(
|
|
340
|
+
(m) => m[1]
|
|
341
|
+
);
|
|
342
|
+
const relative = hrefs.filter(
|
|
343
|
+
(href) => href.startsWith("/") || href.startsWith("#")
|
|
344
|
+
);
|
|
345
|
+
const isNavRun =
|
|
346
|
+
run.length >= 3 && hrefs.length > 0 && relative.length / hrefs.length >= 0.5;
|
|
347
|
+
if (!isNavRun) out.push(...run);
|
|
348
|
+
run = [];
|
|
349
|
+
};
|
|
350
|
+
for (const line of kept) {
|
|
351
|
+
if (/^\s*```/.test(line)) inFence = !inFence;
|
|
352
|
+
if (!inFence && MARKDOWN_LINK_ONLY_LINE.test(line)) {
|
|
353
|
+
run.push(line);
|
|
354
|
+
continue;
|
|
355
|
+
}
|
|
356
|
+
flushRun();
|
|
357
|
+
out.push(line);
|
|
358
|
+
}
|
|
359
|
+
flushRun();
|
|
360
|
+
|
|
361
|
+
return out.join("\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
362
|
+
}
|
|
363
|
+
|
|
211
364
|
/**
|
|
212
365
|
* Escape the HTML-significant characters in a raw string so it can be
|
|
213
366
|
* safely embedded inside generated HTML (used for code spans/blocks).
|
|
@@ -231,6 +384,19 @@ function escapeHTMLChars(str) {
|
|
|
231
384
|
*/
|
|
232
385
|
function applyInlineMarkdown(text) {
|
|
233
386
|
return text
|
|
387
|
+
// Linked images: [](href) -- JINA emits these for
|
|
388
|
+
// thumbnail links; must run before both the image and link rules or the
|
|
389
|
+
// outer link's label swallows the inner image syntax.
|
|
390
|
+
.replace(
|
|
391
|
+
/\[!\[([^\]]*)\]\(([^)\s]+)(?:\s+"[^"]*")?\)\]\(([^)\s]+)\)/g,
|
|
392
|
+
(_m, alt, src, href) =>
|
|
393
|
+
`<a href="${href}"><img src="${src}" alt="${alt}" /></a>`
|
|
394
|
+
)
|
|
395
|
+
// Autolinks: <https://example.com>
|
|
396
|
+
.replace(
|
|
397
|
+
/<(https?:\/\/[^>\s]+)>/g,
|
|
398
|
+
(_m, href) => `<a href="${href}">${href}</a>`
|
|
399
|
+
)
|
|
234
400
|
// Images:  -- must run before links
|
|
235
401
|
.replace(
|
|
236
402
|
/!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/g,
|
|
@@ -260,11 +426,12 @@ function applyInlineMarkdown(text) {
|
|
|
260
426
|
* intended for post-processing content returned as Markdown (e.g. from the
|
|
261
427
|
* JINA reader fallback in the scraper).
|
|
262
428
|
*
|
|
263
|
-
* Supported block elements: ATX headers (`#`..`######`),
|
|
264
|
-
* (
|
|
265
|
-
* (
|
|
266
|
-
*
|
|
267
|
-
*
|
|
429
|
+
* Supported block elements: ATX headers (`#`..`######`), setext headers
|
|
430
|
+
* (`===`/`---` underlines), fenced code blocks (```lang), blockquotes (`>`),
|
|
431
|
+
* unordered lists (`-`, `*`, `+`), ordered lists (`1.`, `1)`), pipe tables,
|
|
432
|
+
* horizontal rules (`---`, `***`, `___`) and paragraphs.
|
|
433
|
+
* Supported inline elements: bold, italic, strikethrough, inline code, images,
|
|
434
|
+
* links, linked images (`[](href)`) and autolinks (`<https://…>`).
|
|
268
435
|
*
|
|
269
436
|
* @param {string} markdown - The Markdown content to convert.
|
|
270
437
|
* @returns {string} The resulting formatted HTML string.
|
|
@@ -299,6 +466,45 @@ export function convertMarkdownToFormattedHTML(markdown) {
|
|
|
299
466
|
return `\u0000IC${inlineCodes.length - 1}\u0000`;
|
|
300
467
|
});
|
|
301
468
|
|
|
469
|
+
// 3. Setext headers: a text line underlined with === (h1) or --- (h2).
|
|
470
|
+
// Converted to ATX form so the line loop below handles them; JINA uses
|
|
471
|
+
// long === underlines for page titles.
|
|
472
|
+
text = text
|
|
473
|
+
.replace(/^(?![\s#>])([^\n]{1,120})\n={3,}[ \t]*$/gm, "# $1")
|
|
474
|
+
.replace(/^(?![\s#>|-])([^\n]{0,119}[^\s|-])\n-{3,}[ \t]*$/gm, "## $1");
|
|
475
|
+
|
|
476
|
+
// 4. Pipe tables: header row, |---| separator row, then body rows. Swapped
|
|
477
|
+
// out for placeholders (like code blocks) so the line loop skips them.
|
|
478
|
+
const tables = [];
|
|
479
|
+
text = text.replace(
|
|
480
|
+
/(?<=^|\n)([^\n]*\|[^\n]*)\n\|?[ \t:]*-{2,}[ \t|:-]*\n((?:[^\n]*\|[^\n]*(?:\n|$))*)/g,
|
|
481
|
+
(_m, headerRow, bodyRows) => {
|
|
482
|
+
const splitRow = (row) =>
|
|
483
|
+
row
|
|
484
|
+
.trim()
|
|
485
|
+
.replace(/^\||\|$/g, "")
|
|
486
|
+
.split("|")
|
|
487
|
+
.map((cell) => applyInlineMarkdown(cell.trim()));
|
|
488
|
+
const header = splitRow(headerRow)
|
|
489
|
+
.map((cell) => `<th>${cell}</th>`)
|
|
490
|
+
.join("");
|
|
491
|
+
const body = bodyRows
|
|
492
|
+
.split("\n")
|
|
493
|
+
.filter((row) => row.trim())
|
|
494
|
+
.map(
|
|
495
|
+
(row) =>
|
|
496
|
+
`<tr>${splitRow(row)
|
|
497
|
+
.map((cell) => `<td>${cell}</td>`)
|
|
498
|
+
.join("")}</tr>`
|
|
499
|
+
)
|
|
500
|
+
.join("");
|
|
501
|
+
tables.push(
|
|
502
|
+
`<table><thead><tr>${header}</tr></thead><tbody>${body}</tbody></table>`
|
|
503
|
+
);
|
|
504
|
+
return `\u0000TB${tables.length - 1}\u0000\n`;
|
|
505
|
+
}
|
|
506
|
+
);
|
|
507
|
+
|
|
302
508
|
const lines = text.split("\n");
|
|
303
509
|
const out = [];
|
|
304
510
|
let inUl = false;
|
|
@@ -340,6 +546,16 @@ export function convertMarkdownToFormattedHTML(markdown) {
|
|
|
340
546
|
continue;
|
|
341
547
|
}
|
|
342
548
|
|
|
549
|
+
// Standalone table placeholder line
|
|
550
|
+
const tb = line.match(/^\u0000TB(\d+)\u0000$/);
|
|
551
|
+
if (tb) {
|
|
552
|
+
flushParagraph();
|
|
553
|
+
closeLists();
|
|
554
|
+
closeBlockquote();
|
|
555
|
+
out.push(tables[Number(tb[1])]);
|
|
556
|
+
continue;
|
|
557
|
+
}
|
|
558
|
+
|
|
343
559
|
// Blank line closes open blocks
|
|
344
560
|
if (/^\s*$/.test(line)) {
|
|
345
561
|
flushParagraph();
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
* Supports YouTube transcripts, PDFs, DOCX, and web articles.
|
|
4
4
|
*/
|
|
5
5
|
import { extractContentAndCite } from "../html-to-content/html-to-content";
|
|
6
|
+
import { detectMarkdown } from "../html-to-content/html-utils";
|
|
6
7
|
import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
|
|
7
8
|
import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
|
|
8
9
|
import { scrapeURL } from "./url-to-html";
|
|
@@ -195,9 +196,12 @@ export async function extractContent(
|
|
|
195
196
|
}
|
|
196
197
|
} else if (
|
|
197
198
|
typeof urlOrDoc === "string" &&
|
|
198
|
-
|
|
199
|
+
!urlOrDoc.startsWith("http") &&
|
|
200
|
+
(/<\/[^>]+>/.test(urlOrDoc.trim()) || detectMarkdown(urlOrDoc))
|
|
199
201
|
) {
|
|
200
|
-
|
|
202
|
+
// Raw HTML string, or raw Markdown (detected via regexp checks and
|
|
203
|
+
// converted to HTML inside extractContentAndCite).
|
|
204
|
+
console.log("[extractContent] input is raw HTML/Markdown string");
|
|
201
205
|
// If urlOrDoc is an HTML string, treat as HTML content
|
|
202
206
|
options.url = options.url || "";
|
|
203
207
|
|