extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,398 @@
|
|
|
1
|
+
// @ts-nocheck
|
|
2
|
+
/**
|
|
3
|
+
* @module research/extractor/html-to-content/html-utils
|
|
4
|
+
* @description Research library module.
|
|
5
|
+
*/
|
|
6
|
+
/**
|
|
7
|
+
* Converts URL-safe escaped HTML codes like &"'`’ & to standard HTML or in reverse.
|
|
8
|
+
* @param {string} str - The string to process.
|
|
9
|
+
* @param {boolean} toStandardHTML default=true - If true, converts url-safe codes
|
|
10
|
+
* to standard HTML. If false, converts standard HTML to url-safe codes.
|
|
11
|
+
* @return {string} The processed string.
|
|
12
|
+
* @category HTML Utilities
|
|
13
|
+
* @example
|
|
14
|
+
* var normalHTML = convertURLSafeHTMLToHTML('<p>This & that © 2023 '+
|
|
15
|
+
* '"Quotes"'Apostrophes' €100 ☺</p>', true)
|
|
16
|
+
* console.log(normalHTML) // "<p>This & that \u00a9 2023 "Quotes" 'Apostrophes' \u20ac100 \u263a</p>"
|
|
17
|
+
*/
|
|
18
|
+
export function convertURLSafeHTMLToHTML(str, toStandardHTML = true) {
|
|
19
|
+
const entityMap = {
|
|
20
|
+
"&": "&",
|
|
21
|
+
"<": "<",
|
|
22
|
+
">": ">",
|
|
23
|
+
'"': """,
|
|
24
|
+
" ": " ",
|
|
25
|
+
"'": "'",
|
|
26
|
+
"`": "`",
|
|
27
|
+
"\u00a2": "¢",
|
|
28
|
+
"\u00a3": "£",
|
|
29
|
+
"\u00a5": "¥",
|
|
30
|
+
"\u20ac": "€",
|
|
31
|
+
"\u00a9": "©",
|
|
32
|
+
"\u00ae": "®",
|
|
33
|
+
"\u2122": "™",
|
|
34
|
+
};
|
|
35
|
+
|
|
36
|
+
// Add numeric character references for Latin-1 Supplement characters
|
|
37
|
+
for (let i = 160; i <= 255; i++) {
|
|
38
|
+
entityMap[String.fromCharCode(i)] = `&#${i};`;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
if (toStandardHTML) {
|
|
42
|
+
// Create a reverse mapping for unescaping
|
|
43
|
+
const reverseEntityMap = Object.fromEntries(
|
|
44
|
+
Object.entries(entityMap).map(([k, v]) => [v, k])
|
|
45
|
+
);
|
|
46
|
+
|
|
47
|
+
// Add alternative representations
|
|
48
|
+
reverseEntityMap["'"] = "'";
|
|
49
|
+
reverseEntityMap["«"] = "\u00ab";
|
|
50
|
+
reverseEntityMap["»"] = "\u00bb";
|
|
51
|
+
|
|
52
|
+
// Regex to match all types of HTML entities
|
|
53
|
+
const entityRegex = new RegExp(
|
|
54
|
+
Object.keys(reverseEntityMap).join("|") + "|&#[0-9]+;|&#x[0-9a-fA-F]+;",
|
|
55
|
+
"g"
|
|
56
|
+
);
|
|
57
|
+
|
|
58
|
+
str = str.replace(entityRegex, (entity) => {
|
|
59
|
+
if (entity.startsWith("&#x")) {
|
|
60
|
+
// Convert hexadecimal numeric character reference
|
|
61
|
+
return String.fromCharCode(parseInt(entity.slice(3, -1), 16));
|
|
62
|
+
} else if (entity.startsWith("&#")) {
|
|
63
|
+
// Convert decimal numeric character reference
|
|
64
|
+
return String.fromCharCode(parseInt(entity.slice(2, -1), 10));
|
|
65
|
+
}
|
|
66
|
+
// Convert named entity
|
|
67
|
+
return reverseEntityMap[entity] || entity;
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
str = str.replace(/[\u0300-\u036f]/g, ""); //special chars
|
|
71
|
+
|
|
72
|
+
return str;
|
|
73
|
+
} else {
|
|
74
|
+
// Regex to match all characters that need to be escaped
|
|
75
|
+
const charRegex = new RegExp(`[${Object.keys(entityMap).join("")}]`, "g");
|
|
76
|
+
return str.replace(charRegex, (char) => entityMap[char]);
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Convert relative URL to absolute URL using base URL.
|
|
82
|
+
* @param {string} base base url of the domain
|
|
83
|
+
* @param {string} relative partial urls like ../images/image.jpg #hash
|
|
84
|
+
* @returns {string} absolute URL
|
|
85
|
+
* @example
|
|
86
|
+
* var absoluteURL = convertURLToAbsoluteURL('https://example.com', 'images/image.jpg')
|
|
87
|
+
* console.log(absoluteURL) // Returns: "https://example.com/images/image.jpg"
|
|
88
|
+
* var absoluteURL = convertURLToAbsoluteURL('https://example.com', '//images/image.jpg')
|
|
89
|
+
* console.log(absoluteURL) // Returns: "https:images/image.jpg"
|
|
90
|
+
* @category HTML Utilities
|
|
91
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
92
|
+
*/
|
|
93
|
+
export function convertURLToAbsoluteURL(base, relative) {
|
|
94
|
+
// remove the %20 codes like data:image/svg+xml,%3Csvg%20x
|
|
95
|
+
relative = decodeURI(relative);
|
|
96
|
+
base = decodeURI(base);
|
|
97
|
+
|
|
98
|
+
if (
|
|
99
|
+
relative.includes("data:") ||
|
|
100
|
+
relative.startsWith("#") ||
|
|
101
|
+
relative.startsWith("http")
|
|
102
|
+
)
|
|
103
|
+
return relative;
|
|
104
|
+
|
|
105
|
+
// Remove hash from base URL
|
|
106
|
+
base = base.replace(/#.*$/, "");
|
|
107
|
+
|
|
108
|
+
// If relative URL starts with '//', add scheme from base
|
|
109
|
+
if (relative.startsWith("//")) return base.split("://")[0] + ":" + relative;
|
|
110
|
+
|
|
111
|
+
// If relative URL starts with '/', replace everything after the host in base
|
|
112
|
+
if (relative[0] === "/") {
|
|
113
|
+
const matchdomain = base.match(/^(https?:\/\/[^\/]+)/i);
|
|
114
|
+
const domain = matchdomain ? matchdomain[1] : null;
|
|
115
|
+
|
|
116
|
+
return domain + relative;
|
|
117
|
+
}
|
|
118
|
+
// Handle relative URLs
|
|
119
|
+
|
|
120
|
+
if (relative.startsWith("../")) {
|
|
121
|
+
base = base.replace(/\/[^\/]+$/, "");
|
|
122
|
+
|
|
123
|
+
while (relative.substring(0, 3) === "../") {
|
|
124
|
+
relative = relative.substring(3);
|
|
125
|
+
base = base.replace(/\/[^\/]+$/, "");
|
|
126
|
+
}
|
|
127
|
+
relative = relative.replace(/^\.\//, "");
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// Combine base and relative
|
|
131
|
+
//
|
|
132
|
+
if (relative.startsWith("/")) {
|
|
133
|
+
base = base.replace(/\/[^\/]+$/, "");
|
|
134
|
+
|
|
135
|
+
return base.replace(/\/+$/, "") + relative;
|
|
136
|
+
} else {
|
|
137
|
+
return base.split("/").slice(0, -1).join("/") + "/" + relative;
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
import { marked } from "marked";
|
|
142
|
+
import hljs from "highlight.js";
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Converts Markdown text to HTML. It handles the following Markdown elements:
|
|
146
|
+
* - Headers (h1 to h6)
|
|
147
|
+
* - Bold text
|
|
148
|
+
* - Italic text
|
|
149
|
+
* - Unordered lists
|
|
150
|
+
* - Ordered lists
|
|
151
|
+
* - Paragraphs
|
|
152
|
+
* - Images
|
|
153
|
+
* - Links
|
|
154
|
+
* - Code blocks
|
|
155
|
+
* @param {string} content - The Markdown or HTML content to be converted.
|
|
156
|
+
* @param {boolean} toHtml - default=true - If true, converts Markdown to HTML.
|
|
157
|
+
* If false, converts HTML to Markdown.
|
|
158
|
+
* @returns {string} The resulting HTML string.
|
|
159
|
+
* @category HTML Utilities
|
|
160
|
+
* @example
|
|
161
|
+
* const markdown = "# Header\n\nThis is **bold** and *italic* text.\n\n* List item 1\n* List item 2";
|
|
162
|
+
* const html = convertMarkdownToHTML(markdown);
|
|
163
|
+
* console.log(html);
|
|
164
|
+
* // Output:
|
|
165
|
+
* // <h1>Header</h1>
|
|
166
|
+
* // <p>This is <strong>bold</strong> and <em>italic</em> text.</p>
|
|
167
|
+
* // <ul><li>List item 1</li><li>List item 2</li></ul>
|
|
168
|
+
*/
|
|
169
|
+
export function convertMarkdownToHTML(content, toHtml = true) {
|
|
170
|
+
if (!toHtml) return convertHTMLToMarkdown(content);
|
|
171
|
+
|
|
172
|
+
// const md = new MarkdownIt({
|
|
173
|
+
// highlight: function (str, lang) {
|
|
174
|
+
// // If a language is provided and it's recognized by hljs
|
|
175
|
+
// if (lang && hljs.getLanguage(lang)) {
|
|
176
|
+
// try {
|
|
177
|
+
// return (
|
|
178
|
+
// '<pre><code class="hljs">' +
|
|
179
|
+
// hljs.highlight(str, { language: lang, ignoreIllegals: true }).value
|
|
180
|
+
// + '</code></pre>'
|
|
181
|
+
// );
|
|
182
|
+
// } catch (__) {}
|
|
183
|
+
// }
|
|
184
|
+
|
|
185
|
+
// // Default fallback for unsupported or no language
|
|
186
|
+
// return (
|
|
187
|
+
// '<pre><code class="hljs">' + md.utils.escapeHtml(str) + '</code></pre>'
|
|
188
|
+
// );
|
|
189
|
+
// },
|
|
190
|
+
// }).use(function (md) {
|
|
191
|
+
// // Override the default fence rule for handling code blocks
|
|
192
|
+
// const fence = md.renderer.rules.fence || function (tokens, idx, options, env, slf) {
|
|
193
|
+
// const token = tokens[idx];
|
|
194
|
+
// const code = token.content
|
|
195
|
+
// .trim() // Trim leading/trailing whitespace
|
|
196
|
+
// .replace(/^[ \t]*/gm, '') // Remove leading whitespace while preserving relative indentation
|
|
197
|
+
// .replace(/&/g, '&') // Encode HTML special characters
|
|
198
|
+
// .replace(/</g, '<')
|
|
199
|
+
// .replace(/>/g, '>')
|
|
200
|
+
// .replace(/"/g, '"')
|
|
201
|
+
// .replace(/'/g, ''');
|
|
202
|
+
|
|
203
|
+
// // Wrap code in a blockquote and ignore the language name
|
|
204
|
+
// return `<blockquote class="custom-code-block"><pre><code>${code}</code></pre></blockquote>`;
|
|
205
|
+
// };
|
|
206
|
+
|
|
207
|
+
// md.renderer.rules.fence = fence;
|
|
208
|
+
// });
|
|
209
|
+
|
|
210
|
+
// // Render markdown content
|
|
211
|
+
// return md.render(content);
|
|
212
|
+
|
|
213
|
+
marked.setOptions({
|
|
214
|
+
highlight: function (code, lang) {
|
|
215
|
+
const language = hljs.getLanguage(lang) ? lang : "plaintext";
|
|
216
|
+
return hljs.highlight(code, { language }).value;
|
|
217
|
+
},
|
|
218
|
+
langPrefix: "hljs language-",
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
return content?.length ? marked.parse(content) : "";
|
|
222
|
+
|
|
223
|
+
var html = contentconvertMarkdownToHTML
|
|
224
|
+
// Convert headers
|
|
225
|
+
.replace(/^(#{1,6})\s(.+)$/gm, (match, hashes, content) => {
|
|
226
|
+
const level = hashes.length;
|
|
227
|
+
return `<h${level}>${content.trim()}</h${level}>`;
|
|
228
|
+
})
|
|
229
|
+
|
|
230
|
+
// Convert bold text
|
|
231
|
+
.replace(/\*\*(.+?)\*\*/g, "<b>$1</b>")
|
|
232
|
+
|
|
233
|
+
// Convert italic text
|
|
234
|
+
.replace(/\*(.+?)\*/g, "<em>$1</em>")
|
|
235
|
+
|
|
236
|
+
// Convert unordered lists
|
|
237
|
+
.replace(/^\s*\*\s(.+)$/gm, "<li>$1</li>")
|
|
238
|
+
.replace(/(<li>.*<\/li>)/s, "<ul>$1</ul>")
|
|
239
|
+
|
|
240
|
+
// Convert ordered lists
|
|
241
|
+
.replace(/^\s*\d+\.\s(.+)$/gm, "<li>$1</li>")
|
|
242
|
+
.replace(/(<li>.*<\/li>)/s, "<ol>$1</ol>")
|
|
243
|
+
|
|
244
|
+
// Convert horizontal rules (---, ___, ***)
|
|
245
|
+
.replace(/^[-_*]{3,}\s*$/gm, "<hr>")
|
|
246
|
+
|
|
247
|
+
// Convert code blocks (```)
|
|
248
|
+
.replace(/```([^`]+)```/g, "<code>$1</code>")
|
|
249
|
+
|
|
250
|
+
.replace(/```(\w*)\n([\s\S]*?)```/g, function (match, lang, code) {
|
|
251
|
+
code = code
|
|
252
|
+
.trim()
|
|
253
|
+
// Remove leading whitespace from each line while preserving relative indentation
|
|
254
|
+
.replace(/^[ \t]*/gm, "")
|
|
255
|
+
// Encode HTML special characters
|
|
256
|
+
.replace(/&/g, "&")
|
|
257
|
+
.replace(/</g, "<")
|
|
258
|
+
.replace(/>/g, ">")
|
|
259
|
+
.replace(/"/g, """)
|
|
260
|
+
.replace(/'/g, "'");
|
|
261
|
+
|
|
262
|
+
return lang
|
|
263
|
+
? `<code class="language-${lang}">${code}</code>`
|
|
264
|
+
: `<code>${code}</code>`;
|
|
265
|
+
})
|
|
266
|
+
|
|
267
|
+
// Handle inline code blocks
|
|
268
|
+
.replace(
|
|
269
|
+
/(^|[^\\])(`+)([^\r]*?[^`])\2(?!`)/gm,
|
|
270
|
+
function (match, pre, backticks, code) {
|
|
271
|
+
code = code
|
|
272
|
+
.trim()
|
|
273
|
+
// Remove leading and trailing whitespace
|
|
274
|
+
.replace(/^[ \t]*/g, "")
|
|
275
|
+
.replace(/[ \t]*$/g, "")
|
|
276
|
+
// Encode HTML special characters
|
|
277
|
+
|
|
278
|
+
.replace(/&/g, "&")
|
|
279
|
+
.replace(/</g, "<")
|
|
280
|
+
.replace(/>/g, ">")
|
|
281
|
+
.replace(/"/g, """)
|
|
282
|
+
.replace(/'/g, "'");
|
|
283
|
+
|
|
284
|
+
return pre + "<code>" + code + "</code>";
|
|
285
|
+
}
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
// Convert inline code (`)
|
|
289
|
+
.replace(/`([^`]+)`/g, "<code>$1</code>")
|
|
290
|
+
|
|
291
|
+
// Convert paragraphs
|
|
292
|
+
.split("\n\n")
|
|
293
|
+
.map((para) => {
|
|
294
|
+
if (!para.startsWith("<")) {
|
|
295
|
+
return `<p>${para.trim()}</p>`;
|
|
296
|
+
}
|
|
297
|
+
return para;
|
|
298
|
+
})
|
|
299
|
+
.join("\n")
|
|
300
|
+
|
|
301
|
+
// Convert images
|
|
302
|
+
.replace(/\!\[(.*?)\]\((.*?)\)/g, '<img src="$2" alt="$1" />')
|
|
303
|
+
|
|
304
|
+
// Convert links
|
|
305
|
+
.replace(/\[(.*?)\]\((.*?)\)/g, '<a href="$2">$1</a>')
|
|
306
|
+
|
|
307
|
+
// Clean up extra newlines
|
|
308
|
+
.replace(/\n\s*\n/g, "\n")
|
|
309
|
+
.trim();
|
|
310
|
+
|
|
311
|
+
return html;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
export function convertHTMLToMarkdown(html) {
|
|
315
|
+
var markdown = html
|
|
316
|
+
// Convert headers
|
|
317
|
+
.replace(/<h([1-6])>(.*?)<\/h[1-6]>/g, (match, level, content) => {
|
|
318
|
+
return "#".repeat(parseInt(level)) + " " + content.trim() + "\n\n";
|
|
319
|
+
})
|
|
320
|
+
|
|
321
|
+
// Convert bold text
|
|
322
|
+
.replace(/<strong>(.*?)<\/strong>/g, "**$1**")
|
|
323
|
+
.replace(/<b>(.*?)<\/b>/g, "**$1**")
|
|
324
|
+
|
|
325
|
+
// Convert italic text
|
|
326
|
+
.replace(/<em>(.*?)<\/em>/g, "*$1*")
|
|
327
|
+
|
|
328
|
+
// Convert unordered lists
|
|
329
|
+
.replace(/<ul>(.*?)<\/ul>/gs, (match, content) => {
|
|
330
|
+
return content.replace(/<li>(.*?)<\/li>/g, "* $1\n") + "\n";
|
|
331
|
+
})
|
|
332
|
+
|
|
333
|
+
// Convert ordered lists
|
|
334
|
+
.replace(/<ol>(.*?)<\/ol>/gs, (match, content) => {
|
|
335
|
+
let index = 1;
|
|
336
|
+
return (
|
|
337
|
+
content.replace(/<li>(.*?)<\/li>/g, () => `${index++}. $1\n`) + "\n"
|
|
338
|
+
);
|
|
339
|
+
})
|
|
340
|
+
|
|
341
|
+
// Convert paragraphs
|
|
342
|
+
.replace(/<p>(.*?)<\/p>/g, "$1\n\n")
|
|
343
|
+
|
|
344
|
+
// Convert images
|
|
345
|
+
.replace(/<img src="(.*?)" alt="(.*?)".*?\/>/g, "")
|
|
346
|
+
|
|
347
|
+
// Convert links
|
|
348
|
+
.replace(/<a href="(.*?)">(.*?)<\/a>/g, "[$2]($1)")
|
|
349
|
+
|
|
350
|
+
// Remove any remaining HTML tags
|
|
351
|
+
.replace(/<[^>]*>/g, "")
|
|
352
|
+
|
|
353
|
+
// Trim extra whitespace
|
|
354
|
+
.trim();
|
|
355
|
+
|
|
356
|
+
return markdown;
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
/**
|
|
360
|
+
* Copy HTML to clipboard. When pasting into rich text field,
|
|
361
|
+
* pastes rich text. When pasting into plain text field, pastes:
|
|
362
|
+
* plain text, html, or markdown.
|
|
363
|
+
*
|
|
364
|
+
* @param {string} html - The HTML content to be copied.
|
|
365
|
+
* @param {object} options - The options object.
|
|
366
|
+
* @param {number} options.pastePlainFormat -
|
|
367
|
+
* default=0
|
|
368
|
+
* 0 - plain text
|
|
369
|
+
* 1 - markdown
|
|
370
|
+
* 2 - html
|
|
371
|
+
* @returns {Promise<void>} - A promise that resolves when
|
|
372
|
+
* the HTML is copied to the clipboard.
|
|
373
|
+
* @category HTML Utilities
|
|
374
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
375
|
+
*/
|
|
376
|
+
export async function copyHTMLToClipboard(html, options = {}) {
|
|
377
|
+
var { pastePlainFormat = 0 } = options;
|
|
378
|
+
|
|
379
|
+
if (typeof window == "undefined" || !navigator?.clipboard) return;
|
|
380
|
+
|
|
381
|
+
const htmlBlob = new Blob([html], { type: "text/html" });
|
|
382
|
+
|
|
383
|
+
var plainText =
|
|
384
|
+
pastePlainFormat == 0
|
|
385
|
+
? html.replace(/<[^>]*>?/g, "")
|
|
386
|
+
: pastePlainFormat == 1
|
|
387
|
+
? convertMarkdownToHTML(html, false)
|
|
388
|
+
: html;
|
|
389
|
+
|
|
390
|
+
const textBlob = new Blob([plainText], { type: "text/plain" });
|
|
391
|
+
|
|
392
|
+
const clipboardItem = new window.ClipboardItem({
|
|
393
|
+
"text/html": htmlBlob,
|
|
394
|
+
"text/plain": textBlob,
|
|
395
|
+
});
|
|
396
|
+
|
|
397
|
+
return await navigator.clipboard.write([clipboardItem]);
|
|
398
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Research Agent Library entry point.
|
|
3
|
+
* Exports various specialized agents, tools, and utilities for AI-driven research.
|
|
4
|
+
*
|
|
5
|
+
* @author vtempest <grokthiscontact@gmail.com>
|
|
6
|
+
* @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
|
|
7
|
+
* to get a dual-use commercial license to remove the GPL requirements.
|
|
8
|
+
*/
|
|
9
|
+
export * from "./search/search-web";
|
|
10
|
+
// Re-export MetaSearchAgent and search handlers (with search functions) for backward compatibility
|
|
11
|
+
export * from "./search";
|
|
12
|
+
export * from "./tokenize/word-to-root-stem";
|
|
13
|
+
export * from "./tokenize/suggest-complete-word";
|
|
14
|
+
export * from "./tokenize/text-to-topic-tokens";
|
|
15
|
+
export * from "./tokenize/text-to-sentences";
|
|
16
|
+
export * from "./tokenize/text-to-chunks";
|
|
17
|
+
export * from "./url-to-content/url-to-content";
|
|
18
|
+
export * from "./url-to-content/url-to-html";
|
|
19
|
+
export * from "./html-to-cite/url-to-domain";
|
|
20
|
+
export * from "./url-to-content/youtube-to-text";
|
|
21
|
+
// PDF export removed from main index to prevent pdfjs-serverless from being evaluated at build time
|
|
22
|
+
// Import directly from "./pdf-to-html/pdfToHtml" when needed
|
|
23
|
+
export * from "./url-to-content/docx-to-content";
|
|
24
|
+
export * from "./html-to-content/html-to-content";
|
|
25
|
+
export * from "./html-to-content/extract-content/extract-content-readability";
|
|
26
|
+
export * from "./html-to-content/extract-content/extract-content-mercury";
|
|
27
|
+
export * from "./html-to-content/html-to-basic-html";
|
|
28
|
+
export * from "./html-to-cite/extract-cite";
|
|
29
|
+
export * from "./html-to-content/html-utils";
|