extract-webpage 1.2.269 → 1.2.271

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -55,6 +55,39 @@ export declare function convertURLToAbsoluteURL(base: any, relative: any): any;
55
55
  * // <ul><li>List item 1</li><li>List item 2</li></ul>
56
56
  */
57
57
  export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): any;
58
+ /**
59
+ * Detect whether a string is Markdown (rather than HTML or plain text) using
60
+ * regexp checks. Content that is dominated by HTML tags is never treated as
61
+ * Markdown, so real scraped pages pass through untouched; text needs at least
62
+ * two distinct Markdown syntax signals (or several links/images in Markdown
63
+ * form) to qualify. Used to catch scraper responses (e.g. the JINA reader or
64
+ * proxies wrapping it) that return Markdown in place of HTML, so it can be
65
+ * converted before main-content extraction — otherwise the article panel
66
+ * renders raw `[text](url)` syntax.
67
+ *
68
+ * @param {string} text - The content to test.
69
+ * @returns {boolean} True when the content should be parsed as Markdown.
70
+ * @category HTML Utilities
71
+ * @example
72
+ * detectMarkdown("# Title\n\nSome **bold** text.") // true
73
+ * detectMarkdown("<html><body><p>Hi</p></body></html>") // false
74
+ */
75
+ export declare function detectMarkdown(text: any): boolean;
76
+ /**
77
+ * Remove extra non-article content from a Markdown extraction using regexp
78
+ * checks: JINA reader metadata lines, cookie/consent and navigation phrases,
79
+ * and runs of consecutive link-only lines (menus, breadcrumbs, "related"
80
+ * link farms) whose targets are mostly relative site navigation. Standalone
81
+ * links inside prose are kept.
82
+ *
83
+ * @param {string} markdown - The Markdown content to clean.
84
+ * @returns {string} The cleaned Markdown.
85
+ * @category HTML Utilities
86
+ * @example
87
+ * removeMarkdownNavigation("Title: Page\n[Skip to content](#main)\nReal text")
88
+ * // => "Real text"
89
+ */
90
+ export declare function removeMarkdownNavigation(markdown: any): string;
58
91
  /**
59
92
  * Convert a Markdown document to formatted HTML using regular expressions to
60
93
  * detect Markdown syntax. Unlike {@link convertMarkdownToHTML} (which relies on
@@ -62,11 +95,12 @@ export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): a
62
95
  * intended for post-processing content returned as Markdown (e.g. from the
63
96
  * JINA reader fallback in the scraper).
64
97
  *
65
- * Supported block elements: ATX headers (`#`..`######`), fenced code blocks
66
- * (```lang), blockquotes (`>`), unordered lists (`-`, `*`, `+`), ordered lists
67
- * (`1.`, `1)`), horizontal rules (`---`, `***`, `___`) and paragraphs.
68
- * Supported inline elements: bold, italic, strikethrough, inline code, images
69
- * and links.
98
+ * Supported block elements: ATX headers (`#`..`######`), setext headers
99
+ * (`===`/`---` underlines), fenced code blocks (```lang), blockquotes (`>`),
100
+ * unordered lists (`-`, `*`, `+`), ordered lists (`1.`, `1)`), pipe tables,
101
+ * horizontal rules (`---`, `***`, `___`) and paragraphs.
102
+ * Supported inline elements: bold, italic, strikethrough, inline code, images,
103
+ * links, linked images (`[![alt](src)](href)`) and autolinks (`<https://…>`).
70
104
  *
71
105
  * @param {string} markdown - The Markdown content to convert.
72
106
  * @returns {string} The resulting formatted HTML string.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.269",
3
+ "version": "1.2.271",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -74,11 +74,11 @@
74
74
  "dependencies": {
75
75
  "@huggingface/transformers": "^3.8.1",
76
76
  "ai": "^5.0.0",
77
- "chat-agent-toolkit": "^1.2.272",
77
+ "chat-agent-toolkit": "^1.2.274",
78
78
  "chrono-node": "^2.9.0",
79
79
  "drizzle-orm": "^0.45.1",
80
- "extract-pdf": "^0.1.259",
81
- "extract-youtube": "^1.0.253",
80
+ "extract-pdf": "^0.1.261",
81
+ "extract-youtube": "^1.0.256",
82
82
  "html-entities": "^2.6.0",
83
83
  "js-yaml": "^4.1.1",
84
84
  "jsdom": "^28.1.0",
@@ -0,0 +1,220 @@
1
+ /**
2
+ * @fileoverview Tests for regexp-based Markdown detection, navigation/noise
3
+ * removal, and Markdown-to-HTML conversion used on JINA-style extractions.
4
+ */
5
+ import { describe, expect, it } from "vitest";
6
+ import {
7
+ convertMarkdownToFormattedHTML,
8
+ detectMarkdown,
9
+ removeMarkdownNavigation,
10
+ } from "../html-utils";
11
+
12
+ describe("detectMarkdown", () => {
13
+ it("detects typical markdown documents", () => {
14
+ expect(
15
+ detectMarkdown("# Title\n\nSome **bold** text.\n\n- item 1\n- item 2")
16
+ ).toBe(true);
17
+ });
18
+
19
+ it("detects JINA reader output", () => {
20
+ const jina = [
21
+ "Markdown Content:",
22
+ "Michael Jackson",
23
+ "===============",
24
+ "",
25
+ "[![Image 1](https://i.scdn.co/image/abc)](/album/1)",
26
+ "[Thriller](/album/1C2h7mLntPSeVYciMRTF4a)",
27
+ "[Bad 25th Anniversary](/album/24TAupSNVWSAHL0R7n71vm)",
28
+ ].join("\n");
29
+ expect(detectMarkdown(jina)).toBe(true);
30
+ });
31
+
32
+ it("detects link-heavy markdown with a single syntax signal", () => {
33
+ const links = [
34
+ "[Dangerous](/album/0oX4SealMgNXrvRDhqqOKg) some text",
35
+ "more [Invincible](/album/52E4RP7XDzalpIrOgSTgiQ) text",
36
+ "and [HIStory](/album/3OBhnTLrvkoEEETjFA3Qfk) too",
37
+ ].join("\n");
38
+ expect(detectMarkdown(links)).toBe(true);
39
+ });
40
+
41
+ it("does not flag HTML documents", () => {
42
+ expect(
43
+ detectMarkdown(
44
+ "<html><body><h1>Title</h1><p>Some **bold** text</p><ul><li>a</li></ul></body></html>"
45
+ )
46
+ ).toBe(false);
47
+ });
48
+
49
+ it("does not flag HTML fragments with several closing tags", () => {
50
+ expect(
51
+ detectMarkdown(
52
+ "<div><p>One - two</p><p>Three</p><p>- looks like a list</p></div>"
53
+ )
54
+ ).toBe(false);
55
+ });
56
+
57
+ it("does not flag plain text", () => {
58
+ expect(
59
+ detectMarkdown("Just a plain sentence.\nAnother plain sentence here.")
60
+ ).toBe(false);
61
+ });
62
+
63
+ it("handles empty and non-string input", () => {
64
+ expect(detectMarkdown("")).toBe(false);
65
+ expect(detectMarkdown(null as any)).toBe(false);
66
+ expect(detectMarkdown(undefined as any)).toBe(false);
67
+ });
68
+ });
69
+
70
+ describe("removeMarkdownNavigation", () => {
71
+ it("removes JINA reader metadata lines", () => {
72
+ const input = [
73
+ "Title: Some Page",
74
+ "URL Source: https://example.com",
75
+ "Published Time: 2024-01-01",
76
+ "Markdown Content:",
77
+ "",
78
+ "Real article text here.",
79
+ ].join("\n");
80
+ const output = removeMarkdownNavigation(input);
81
+ expect(output).toBe("Real article text here.");
82
+ });
83
+
84
+ it("removes navigation phrase lines", () => {
85
+ const input = [
86
+ "[Skip to content](#main)",
87
+ "[Sign in](/login)",
88
+ "Real paragraph of the article.",
89
+ "[Back to top](#top)",
90
+ ].join("\n");
91
+ const output = removeMarkdownNavigation(input);
92
+ expect(output).toBe("Real paragraph of the article.");
93
+ });
94
+
95
+ it("removes runs of 3+ link-only lines pointing at relative URLs", () => {
96
+ const input = [
97
+ "Intro paragraph.",
98
+ "",
99
+ "[![Image 1](https://i.scdn.co/a)](/album/1)",
100
+ "[Thriller](/album/1)",
101
+ "[Bad](/album/2)",
102
+ "[Dangerous](/album/3)",
103
+ "",
104
+ "Closing paragraph.",
105
+ ].join("\n");
106
+ const output = removeMarkdownNavigation(input);
107
+ expect(output).toContain("Intro paragraph.");
108
+ expect(output).toContain("Closing paragraph.");
109
+ expect(output).not.toContain("/album/");
110
+ });
111
+
112
+ it("keeps short link-only runs and external citation links", () => {
113
+ const input = [
114
+ "See sources:",
115
+ "[Source A](https://a.example.com)",
116
+ "[Source B](https://b.example.com)",
117
+ "[Source C](https://c.example.com)",
118
+ ].join("\n");
119
+ const output = removeMarkdownNavigation(input);
120
+ expect(output).toContain("Source A");
121
+ expect(output).toContain("Source C");
122
+ });
123
+
124
+ it("keeps links inside prose", () => {
125
+ const input = "This mentions [a page](/internal/path) inside a sentence.";
126
+ expect(removeMarkdownNavigation(input)).toBe(input);
127
+ });
128
+
129
+ it("leaves fenced code blocks untouched", () => {
130
+ const input = [
131
+ "```",
132
+ "Title: not metadata, just code",
133
+ "[a](/x)",
134
+ "[b](/y)",
135
+ "[c](/z)",
136
+ "```",
137
+ ].join("\n");
138
+ const output = removeMarkdownNavigation(input);
139
+ expect(output).toContain("Title: not metadata, just code");
140
+ expect(output).toContain("[b](/y)");
141
+ });
142
+
143
+ it("handles empty and non-string input", () => {
144
+ expect(removeMarkdownNavigation("")).toBe("");
145
+ expect(removeMarkdownNavigation(null as any)).toBe("");
146
+ });
147
+ });
148
+
149
+ describe("convertMarkdownToFormattedHTML", () => {
150
+ it("converts linked images [![alt](src)](href)", () => {
151
+ const html = convertMarkdownToFormattedHTML(
152
+ "[![Cover](https://i.scdn.co/image/abc)](https://example.com/album/1)"
153
+ );
154
+ expect(html).toContain(
155
+ '<a href="https://example.com/album/1"><img src="https://i.scdn.co/image/abc" alt="Cover" /></a>'
156
+ );
157
+ });
158
+
159
+ it("converts images with empty alt text", () => {
160
+ const html = convertMarkdownToFormattedHTML(
161
+ "![](https://i.scdn.co/image/abc)"
162
+ );
163
+ expect(html).toContain('<img src="https://i.scdn.co/image/abc" alt="" />');
164
+ });
165
+
166
+ it("converts setext headers", () => {
167
+ const html = convertMarkdownToFormattedHTML(
168
+ "Page Title\n===============\n\nSection\n-------\n\nBody text."
169
+ );
170
+ expect(html).toContain("<h1>Page Title</h1>");
171
+ expect(html).toContain("<h2>Section</h2>");
172
+ expect(html).toContain("<p>Body text.</p>");
173
+ });
174
+
175
+ it("still treats --- after a blank line as a horizontal rule", () => {
176
+ const html = convertMarkdownToFormattedHTML("Some text.\n\n---\n\nMore.");
177
+ expect(html).toContain("<hr>");
178
+ });
179
+
180
+ it("converts autolinks", () => {
181
+ const html = convertMarkdownToFormattedHTML("Visit <https://example.com> now.");
182
+ expect(html).toContain('<a href="https://example.com">https://example.com</a>');
183
+ });
184
+
185
+ it("converts pipe tables", () => {
186
+ const html = convertMarkdownToFormattedHTML(
187
+ "| Name | Year |\n| --- | --- |\n| Thriller | 1982 |\n| Bad | 1987 |"
188
+ );
189
+ expect(html).toContain("<table>");
190
+ expect(html).toContain("<th>Name</th>");
191
+ expect(html).toContain("<td>Thriller</td>");
192
+ expect(html).toContain("<td>1987</td>");
193
+ expect(html).toContain("</table>");
194
+ });
195
+
196
+ it("applies inline markdown inside table cells", () => {
197
+ const html = convertMarkdownToFormattedHTML(
198
+ "| Album | Link |\n| --- | --- |\n| **Bad** | [play](/album/2) |"
199
+ );
200
+ expect(html).toContain("<strong>Bad</strong>");
201
+ expect(html).toContain('<a href="/album/2">play</a>');
202
+ });
203
+
204
+ it("converts a JINA-style extraction end to end", () => {
205
+ const jina = [
206
+ "Michael Jackson",
207
+ "===============",
208
+ "",
209
+ "The **King of Pop** released many albums.",
210
+ "",
211
+ "- [Thriller](https://open.spotify.com/album/1)",
212
+ "- [Bad](https://open.spotify.com/album/2)",
213
+ ].join("\n");
214
+ const html = convertMarkdownToFormattedHTML(jina);
215
+ expect(html).toContain("<h1>Michael Jackson</h1>");
216
+ expect(html).toContain("<strong>King of Pop</strong>");
217
+ expect(html).toContain('<li><a href="https://open.spotify.com/album/1">Thriller</a></li>');
218
+ expect(html).not.toContain("](");
219
+ });
220
+ });
@@ -6,6 +6,11 @@
6
6
  import { parseHTML } from "linkedom";
7
7
  import { extractCite } from "../html-to-cite/extract-cite";
8
8
  import { convertHTMLToBasicHTML } from "./html-to-basic-html";
9
+ import {
10
+ convertMarkdownToFormattedHTML,
11
+ detectMarkdown,
12
+ removeMarkdownNavigation,
13
+ } from "./html-utils";
9
14
  import { extractHumanName } from "../html-to-cite/human-names-recognize";
10
15
  import { extractMainContentFromHTML } from "./extract-content/extract-content-readability";
11
16
  import { extractMainContentFromHTML2 } from "./extract-content/extract-content-mercury";
@@ -48,6 +53,14 @@ export function extractContentAndCite(documentOrHTML, options = {}) {
48
53
 
49
54
  if (!html) return { error: "No HTML found" };
50
55
 
56
+ // Some scrape paths (the JINA reader, or proxies wrapping it) return the
57
+ // article as Markdown instead of HTML. Without conversion the sidebar
58
+ // renders raw `[text](url)` syntax, so use regexp checks to detect
59
+ // Markdown, strip navigation/reader-metadata noise, and convert it to
60
+ // formatted HTML before main-content extraction runs.
61
+ if (detectMarkdown(html))
62
+ html = convertMarkdownToFormattedHTML(removeMarkdownNavigation(html));
63
+
51
64
  try {
52
65
  var content1 = extractMainContentFromHTML(html, options);
53
66
  } catch (e) {
@@ -140,23 +140,23 @@ export function convertURLToAbsoluteURL(base, relative) {
140
140
 
141
141
  import { marked } from "marked";
142
142
  import Prism from "prismjs";
143
- import "prismjs/components/prism-markup";
144
- import "prismjs/components/prism-css";
145
- import "prismjs/components/prism-javascript";
146
- import "prismjs/components/prism-typescript";
147
- import "prismjs/components/prism-jsx";
148
- import "prismjs/components/prism-tsx";
149
- import "prismjs/components/prism-python";
150
- import "prismjs/components/prism-bash";
151
- import "prismjs/components/prism-json";
152
- import "prismjs/components/prism-yaml";
153
- import "prismjs/components/prism-markdown";
154
- import "prismjs/components/prism-sql";
155
- import "prismjs/components/prism-rust";
156
- import "prismjs/components/prism-go";
157
- import "prismjs/components/prism-java";
158
- import "prismjs/components/prism-c";
159
- import "prismjs/components/prism-cpp";
143
+ import "prismjs/components/prism-markup.js";
144
+ import "prismjs/components/prism-css.js";
145
+ import "prismjs/components/prism-javascript.js";
146
+ import "prismjs/components/prism-typescript.js";
147
+ import "prismjs/components/prism-jsx.js";
148
+ import "prismjs/components/prism-tsx.js";
149
+ import "prismjs/components/prism-python.js";
150
+ import "prismjs/components/prism-bash.js";
151
+ import "prismjs/components/prism-json.js";
152
+ import "prismjs/components/prism-yaml.js";
153
+ import "prismjs/components/prism-markdown.js";
154
+ import "prismjs/components/prism-sql.js";
155
+ import "prismjs/components/prism-rust.js";
156
+ import "prismjs/components/prism-go.js";
157
+ import "prismjs/components/prism-java.js";
158
+ import "prismjs/components/prism-c.js";
159
+ import "prismjs/components/prism-cpp.js";
160
160
 
161
161
  // Configure marked once at module load with Prism.js syntax highlighting.
162
162
  // marked v17 removed the `highlight` option from setOptions, so highlighting
@@ -208,6 +208,159 @@ export function convertMarkdownToHTML(content, toHtml = true) {
208
208
  return content?.length ? marked.parse(content) : "";
209
209
  }
210
210
 
211
+ /**
212
+ * Regexp patterns that identify Markdown syntax. Each entry is a signal that
213
+ * the text is Markdown rather than plain text or HTML — {@link detectMarkdown}
214
+ * counts how many distinct signals match.
215
+ * @private
216
+ */
217
+ const MARKDOWN_SYNTAX_PATTERNS = [
218
+ /^#{1,6}\s+\S/m, // ATX header: # Title
219
+ /^[^\n]{1,120}\n(?:={3,}|-{3,})[ \t]*$/m, // setext header underline
220
+ /!\[[^\]]*\]\([^)]+\)/, // image: ![alt](src)
221
+ /(?<!!)\[[^\]]+\]\([^)]+\)/, // link: [text](href)
222
+ /^\s{0,3}[-*+]\s+\S/m, // unordered list item
223
+ /^\s{0,3}\d+[.)]\s+\S/m, // ordered list item
224
+ /^```/m, // fenced code block
225
+ /^\s{0,3}>\s+\S/m, // blockquote
226
+ /\*\*[^*\n]+\*\*|__[^_\n]+__/, // bold
227
+ /^\|?[^\n|]*\|[^\n]*\n\|?[\s:]*-{2,}[\s|:-]*$/m, // pipe table header + separator
228
+ /^Markdown Content:$/m, // JINA reader preamble
229
+ ];
230
+
231
+ /**
232
+ * Detect whether a string is Markdown (rather than HTML or plain text) using
233
+ * regexp checks. Content that is dominated by HTML tags is never treated as
234
+ * Markdown, so real scraped pages pass through untouched; text needs at least
235
+ * two distinct Markdown syntax signals (or several links/images in Markdown
236
+ * form) to qualify. Used to catch scraper responses (e.g. the JINA reader or
237
+ * proxies wrapping it) that return Markdown in place of HTML, so it can be
238
+ * converted before main-content extraction — otherwise the article panel
239
+ * renders raw `[text](url)` syntax.
240
+ *
241
+ * @param {string} text - The content to test.
242
+ * @returns {boolean} True when the content should be parsed as Markdown.
243
+ * @category HTML Utilities
244
+ * @example
245
+ * detectMarkdown("# Title\n\nSome **bold** text.") // true
246
+ * detectMarkdown("<html><body><p>Hi</p></body></html>") // false
247
+ */
248
+ export function detectMarkdown(text) {
249
+ if (!text || typeof text !== "string") return false;
250
+
251
+ const sample = text.slice(0, 20000);
252
+
253
+ // HTML dominance check: a document skeleton or a meaningful density of
254
+ // closing tags means this is HTML, not Markdown.
255
+ if (/<!doctype\s+html|<html[\s>]|<body[\s>]/i.test(sample)) return false;
256
+ const closingTags = sample.match(
257
+ /<\/(?:p|div|a|span|h[1-6]|li|ul|ol|table|section|article|nav|em|strong|b|i)>/gi
258
+ );
259
+ if (closingTags && closingTags.length >= 3) return false;
260
+
261
+ let signals = 0;
262
+ for (const pattern of MARKDOWN_SYNTAX_PATTERNS)
263
+ if (pattern.test(sample)) signals++;
264
+
265
+ if (signals >= 2) return true;
266
+
267
+ // A single signal still counts when Markdown links/images repeat — the
268
+ // signature of JINA reader output for link-heavy pages.
269
+ const mdLinks = sample.match(/!?\[[^\]]*\]\([^)]+\)/g);
270
+ return signals >= 1 && !!mdLinks && mdLinks.length >= 3;
271
+ }
272
+
273
+ /**
274
+ * Line-level regexps for boilerplate that JINA-style Markdown extractions
275
+ * carry along with the article: reader-preamble metadata and common
276
+ * navigation/chrome phrases rendered as standalone links.
277
+ * @private
278
+ */
279
+ const MARKDOWN_NOISE_LINE_PATTERNS = [
280
+ /^(?:Title|URL Source|Published Time|Markdown Content|Warning|Links\/Buttons):.*$/i, // JINA metadata
281
+ /^\[?\s*(?:skip to (?:main )?content|main menu|jump to (?:content|navigation)|menu|navigation|sign (?:in|up)|log ?in|register|subscribe(?: now)?|share(?: this)?|tweet|print|download app|open app|back to top|show all|see all|see more|view all|load more|read more|previous|next|home)\s*\]?\s*(?:\([^)]*\))?\s*$/i, // nav phrases, bare or as a single link
282
+ /^(?:\W*\s*)?(?:accept(?: all)?(?: cookies)?|we use cookies.*|cookie (?:policy|settings|preferences)|privacy policy|terms of (?:use|service))\s*(?:\([^)]*\))?\]?\s*$/i, // cookie/legal chrome
283
+ ];
284
+
285
+ /**
286
+ * Matches a line that carries no prose of its own — only Markdown links,
287
+ * images, bullets, pipes and punctuation. Runs of these are navigation menus,
288
+ * breadcrumbs, and tag/related-content lists.
289
+ * @private
290
+ */
291
+ const MARKDOWN_LINK_ONLY_LINE =
292
+ /^\s*(?:[-*+>|]\s*)?(?:(?:\[!\[[^\]]*\]\([^)]*\)\]\([^)]*\)|!?\[[^\]]*\]\([^)]*\))\s*(?:[|•·,/>-]\s*)?)+[.\s]*$/;
293
+
294
+ /**
295
+ * Remove extra non-article content from a Markdown extraction using regexp
296
+ * checks: JINA reader metadata lines, cookie/consent and navigation phrases,
297
+ * and runs of consecutive link-only lines (menus, breadcrumbs, "related"
298
+ * link farms) whose targets are mostly relative site navigation. Standalone
299
+ * links inside prose are kept.
300
+ *
301
+ * @param {string} markdown - The Markdown content to clean.
302
+ * @returns {string} The cleaned Markdown.
303
+ * @category HTML Utilities
304
+ * @example
305
+ * removeMarkdownNavigation("Title: Page\n[Skip to content](#main)\nReal text")
306
+ * // => "Real text"
307
+ */
308
+ export function removeMarkdownNavigation(markdown) {
309
+ if (!markdown || typeof markdown !== "string") return "";
310
+
311
+ const lines = markdown.replace(/\r\n?/g, "\n").split("\n");
312
+ const kept = [];
313
+
314
+ // First pass: drop noise lines outside fenced code blocks.
315
+ let inFence = false;
316
+ for (const line of lines) {
317
+ if (/^\s*```/.test(line)) {
318
+ inFence = !inFence;
319
+ kept.push(line);
320
+ continue;
321
+ }
322
+ if (
323
+ !inFence &&
324
+ MARKDOWN_NOISE_LINE_PATTERNS.some((pattern) => pattern.test(line.trim()))
325
+ )
326
+ continue;
327
+ kept.push(line);
328
+ }
329
+
330
+ // Second pass: drop runs of 3+ consecutive link-only lines when the run's
331
+ // links point mostly at relative URLs (`/path`, `#anchor`) — the signature
332
+ // of site navigation rather than cited external sources.
333
+ const out = [];
334
+ let run = [];
335
+ inFence = false;
336
+ const flushRun = () => {
337
+ if (!run.length) return;
338
+ const runText = run.join("\n");
339
+ const hrefs = [...runText.matchAll(/!?\[[^\]]*\]\(([^)\s]*)/g)].map(
340
+ (m) => m[1]
341
+ );
342
+ const relative = hrefs.filter(
343
+ (href) => href.startsWith("/") || href.startsWith("#")
344
+ );
345
+ const isNavRun =
346
+ run.length >= 3 && hrefs.length > 0 && relative.length / hrefs.length >= 0.5;
347
+ if (!isNavRun) out.push(...run);
348
+ run = [];
349
+ };
350
+ for (const line of kept) {
351
+ if (/^\s*```/.test(line)) inFence = !inFence;
352
+ if (!inFence && MARKDOWN_LINK_ONLY_LINE.test(line)) {
353
+ run.push(line);
354
+ continue;
355
+ }
356
+ flushRun();
357
+ out.push(line);
358
+ }
359
+ flushRun();
360
+
361
+ return out.join("\n").replace(/\n{3,}/g, "\n\n").trim();
362
+ }
363
+
211
364
  /**
212
365
  * Escape the HTML-significant characters in a raw string so it can be
213
366
  * safely embedded inside generated HTML (used for code spans/blocks).
@@ -231,6 +384,19 @@ function escapeHTMLChars(str) {
231
384
  */
232
385
  function applyInlineMarkdown(text) {
233
386
  return text
387
+ // Linked images: [![alt](src "title")](href) -- JINA emits these for
388
+ // thumbnail links; must run before both the image and link rules or the
389
+ // outer link's label swallows the inner image syntax.
390
+ .replace(
391
+ /\[!\[([^\]]*)\]\(([^)\s]+)(?:\s+"[^"]*")?\)\]\(([^)\s]+)\)/g,
392
+ (_m, alt, src, href) =>
393
+ `<a href="${href}"><img src="${src}" alt="${alt}" /></a>`
394
+ )
395
+ // Autolinks: <https://example.com>
396
+ .replace(
397
+ /<(https?:\/\/[^>\s]+)>/g,
398
+ (_m, href) => `<a href="${href}">${href}</a>`
399
+ )
234
400
  // Images: ![alt](src "title") -- must run before links
235
401
  .replace(
236
402
  /!\[([^\]]*)\]\(([^)\s]+)(?:\s+"([^"]*)")?\)/g,
@@ -260,11 +426,12 @@ function applyInlineMarkdown(text) {
260
426
  * intended for post-processing content returned as Markdown (e.g. from the
261
427
  * JINA reader fallback in the scraper).
262
428
  *
263
- * Supported block elements: ATX headers (`#`..`######`), fenced code blocks
264
- * (```lang), blockquotes (`>`), unordered lists (`-`, `*`, `+`), ordered lists
265
- * (`1.`, `1)`), horizontal rules (`---`, `***`, `___`) and paragraphs.
266
- * Supported inline elements: bold, italic, strikethrough, inline code, images
267
- * and links.
429
+ * Supported block elements: ATX headers (`#`..`######`), setext headers
430
+ * (`===`/`---` underlines), fenced code blocks (```lang), blockquotes (`>`),
431
+ * unordered lists (`-`, `*`, `+`), ordered lists (`1.`, `1)`), pipe tables,
432
+ * horizontal rules (`---`, `***`, `___`) and paragraphs.
433
+ * Supported inline elements: bold, italic, strikethrough, inline code, images,
434
+ * links, linked images (`[![alt](src)](href)`) and autolinks (`<https://…>`).
268
435
  *
269
436
  * @param {string} markdown - The Markdown content to convert.
270
437
  * @returns {string} The resulting formatted HTML string.
@@ -299,6 +466,45 @@ export function convertMarkdownToFormattedHTML(markdown) {
299
466
  return `\u0000IC${inlineCodes.length - 1}\u0000`;
300
467
  });
301
468
 
469
+ // 3. Setext headers: a text line underlined with === (h1) or --- (h2).
470
+ // Converted to ATX form so the line loop below handles them; JINA uses
471
+ // long === underlines for page titles.
472
+ text = text
473
+ .replace(/^(?![\s#>])([^\n]{1,120})\n={3,}[ \t]*$/gm, "# $1")
474
+ .replace(/^(?![\s#>|-])([^\n]{0,119}[^\s|-])\n-{3,}[ \t]*$/gm, "## $1");
475
+
476
+ // 4. Pipe tables: header row, |---| separator row, then body rows. Swapped
477
+ // out for placeholders (like code blocks) so the line loop skips them.
478
+ const tables = [];
479
+ text = text.replace(
480
+ /(?<=^|\n)([^\n]*\|[^\n]*)\n\|?[ \t:]*-{2,}[ \t|:-]*\n((?:[^\n]*\|[^\n]*(?:\n|$))*)/g,
481
+ (_m, headerRow, bodyRows) => {
482
+ const splitRow = (row) =>
483
+ row
484
+ .trim()
485
+ .replace(/^\||\|$/g, "")
486
+ .split("|")
487
+ .map((cell) => applyInlineMarkdown(cell.trim()));
488
+ const header = splitRow(headerRow)
489
+ .map((cell) => `<th>${cell}</th>`)
490
+ .join("");
491
+ const body = bodyRows
492
+ .split("\n")
493
+ .filter((row) => row.trim())
494
+ .map(
495
+ (row) =>
496
+ `<tr>${splitRow(row)
497
+ .map((cell) => `<td>${cell}</td>`)
498
+ .join("")}</tr>`
499
+ )
500
+ .join("");
501
+ tables.push(
502
+ `<table><thead><tr>${header}</tr></thead><tbody>${body}</tbody></table>`
503
+ );
504
+ return `\u0000TB${tables.length - 1}\u0000\n`;
505
+ }
506
+ );
507
+
302
508
  const lines = text.split("\n");
303
509
  const out = [];
304
510
  let inUl = false;
@@ -340,6 +546,16 @@ export function convertMarkdownToFormattedHTML(markdown) {
340
546
  continue;
341
547
  }
342
548
 
549
+ // Standalone table placeholder line
550
+ const tb = line.match(/^\u0000TB(\d+)\u0000$/);
551
+ if (tb) {
552
+ flushParagraph();
553
+ closeLists();
554
+ closeBlockquote();
555
+ out.push(tables[Number(tb[1])]);
556
+ continue;
557
+ }
558
+
343
559
  // Blank line closes open blocks
344
560
  if (/^\s*$/.test(line)) {
345
561
  flushParagraph();
@@ -3,6 +3,7 @@
3
3
  * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
4
  */
5
5
  import { extractContentAndCite } from "../html-to-content/html-to-content";
6
+ import { detectMarkdown } from "../html-to-content/html-utils";
6
7
  import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
7
8
  import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
8
9
  import { scrapeURL } from "./url-to-html";
@@ -195,9 +196,12 @@ export async function extractContent(
195
196
  }
196
197
  } else if (
197
198
  typeof urlOrDoc === "string" &&
198
- /<\/[^>]+>/.test(urlOrDoc.trim())
199
+ !urlOrDoc.startsWith("http") &&
200
+ (/<\/[^>]+>/.test(urlOrDoc.trim()) || detectMarkdown(urlOrDoc))
199
201
  ) {
200
- console.log("[extractContent] input is raw HTML string");
202
+ // Raw HTML string, or raw Markdown (detected via regexp checks and
203
+ // converted to HTML inside extractContentAndCite).
204
+ console.log("[extractContent] input is raw HTML/Markdown string");
201
205
  // If urlOrDoc is an HTML string, treat as HTML content
202
206
  options.url = options.url || "";
203
207