extract-webpage 1.2.50 → 1.2.52
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +3 -3
- package/dist/extract-webpage.es.js.map +1 -1
- package/dist/index.d.ts +0 -2
- package/package.json +4 -11
- package/src/html-to-content/html-utils.ts +8 -44
- package/src/index.ts +0 -3
- package/dist/search/index.d.ts +0 -14
- package/dist/search/meta-search-agent-reexport.d.ts +0 -8
- package/dist/search/public-searxng.d.ts +0 -47
- package/dist/search/search-web.d.ts +0 -33
- package/dist/search/tavily.d.ts +0 -20
- package/dist/search/url-to-html.d.ts +0 -62
- package/src/search/__tests__/public-searxng.test.ts +0 -529
- package/src/search/index.ts +0 -45
- package/src/search/meta-search-agent-reexport.ts +0 -38
- package/src/search/public-searxng.ts +0 -470
- package/src/search/search-web.ts +0 -668
- package/src/search/tavily.ts +0 -106
- package/src/search/url-to-html.ts +0 -278
package/dist/index.d.ts
CHANGED
|
@@ -6,8 +6,6 @@
|
|
|
6
6
|
* @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
|
|
7
7
|
* to get a dual-use commercial license to remove the GPL requirements.
|
|
8
8
|
*/
|
|
9
|
-
export * from './search/search-web';
|
|
10
|
-
export * from './search';
|
|
11
9
|
export * from './tokenize/word-to-root-stem';
|
|
12
10
|
export * from './tokenize/suggest-complete-word';
|
|
13
11
|
export * from './tokenize/text-to-topic-tokens';
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-webpage",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.52",
|
|
4
4
|
"module": "./dist/extract-webpage.es.js",
|
|
5
5
|
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
6
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
@@ -18,12 +18,6 @@
|
|
|
18
18
|
"import": "./dist/extract-webpage.es.js",
|
|
19
19
|
"require": "./dist/extract-webpage.cjs.js"
|
|
20
20
|
},
|
|
21
|
-
"./search": {
|
|
22
|
-
"types": "./src/search/index.ts",
|
|
23
|
-
"react-server": "./src/search/index.ts",
|
|
24
|
-
"import": "./src/search/index.ts",
|
|
25
|
-
"require": "./src/search/index.ts"
|
|
26
|
-
},
|
|
27
21
|
"./*": {
|
|
28
22
|
"types": "./src/*.ts",
|
|
29
23
|
"react-server": "./src/*",
|
|
@@ -78,12 +72,12 @@
|
|
|
78
72
|
"dependencies": {
|
|
79
73
|
"@huggingface/transformers": "^3.8.1",
|
|
80
74
|
"ai": "^5.0.0",
|
|
81
|
-
"chat-agent-toolkit": "^1.2.
|
|
75
|
+
"chat-agent-toolkit": "^1.2.51",
|
|
82
76
|
"chrono-node": "^2.9.0",
|
|
83
77
|
"drizzle-orm": "^0.45.1",
|
|
84
|
-
"extract-pdf": "^0.1.
|
|
78
|
+
"extract-pdf": "^0.1.39",
|
|
85
79
|
"extract-youtube": "^1.0.36",
|
|
86
|
-
"
|
|
80
|
+
"prismjs": "^1.30.0",
|
|
87
81
|
"html-entities": "^2.6.0",
|
|
88
82
|
"js-yaml": "^4.1.1",
|
|
89
83
|
"jsdom": "^28.1.0",
|
|
@@ -92,7 +86,6 @@
|
|
|
92
86
|
"marked": "^17.0.4",
|
|
93
87
|
"node-fetch": "^3.3.2",
|
|
94
88
|
"qwksearch-api-client": "^0.0.12",
|
|
95
|
-
"tldts": "^7.0.25",
|
|
96
89
|
"zod": "^4.3.6"
|
|
97
90
|
},
|
|
98
91
|
"keywords": [
|
|
@@ -139,7 +139,10 @@ export function convertURLToAbsoluteURL(base, relative) {
|
|
|
139
139
|
}
|
|
140
140
|
|
|
141
141
|
import { marked } from "marked";
|
|
142
|
-
import
|
|
142
|
+
import Prism from "prismjs";
|
|
143
|
+
import loadLanguages from "prismjs/components/index.js";
|
|
144
|
+
|
|
145
|
+
loadLanguages();
|
|
143
146
|
|
|
144
147
|
/**
|
|
145
148
|
* Converts Markdown text to HTML. It handles the following Markdown elements:
|
|
@@ -169,53 +172,14 @@ import hljs from "highlight.js";
|
|
|
169
172
|
export function convertMarkdownToHTML(content, toHtml = true) {
|
|
170
173
|
if (!toHtml) return convertHTMLToMarkdown(content);
|
|
171
174
|
|
|
172
|
-
// const md = new MarkdownIt({
|
|
173
|
-
// highlight: function (str, lang) {
|
|
174
|
-
// // If a language is provided and it's recognized by hljs
|
|
175
|
-
// if (lang && hljs.getLanguage(lang)) {
|
|
176
|
-
// try {
|
|
177
|
-
// return (
|
|
178
|
-
// '<pre><code class="hljs">' +
|
|
179
|
-
// hljs.highlight(str, { language: lang, ignoreIllegals: true }).value
|
|
180
|
-
// + '</code></pre>'
|
|
181
|
-
// );
|
|
182
|
-
// } catch (__) {}
|
|
183
|
-
// }
|
|
184
|
-
|
|
185
|
-
// // Default fallback for unsupported or no language
|
|
186
|
-
// return (
|
|
187
|
-
// '<pre><code class="hljs">' + md.utils.escapeHtml(str) + '</code></pre>'
|
|
188
|
-
// );
|
|
189
|
-
// },
|
|
190
|
-
// }).use(function (md) {
|
|
191
|
-
// // Override the default fence rule for handling code blocks
|
|
192
|
-
// const fence = md.renderer.rules.fence || function (tokens, idx, options, env, slf) {
|
|
193
|
-
// const token = tokens[idx];
|
|
194
|
-
// const code = token.content
|
|
195
|
-
// .trim() // Trim leading/trailing whitespace
|
|
196
|
-
// .replace(/^[ \t]*/gm, '') // Remove leading whitespace while preserving relative indentation
|
|
197
|
-
// .replace(/&/g, '&') // Encode HTML special characters
|
|
198
|
-
// .replace(/</g, '<')
|
|
199
|
-
// .replace(/>/g, '>')
|
|
200
|
-
// .replace(/"/g, '"')
|
|
201
|
-
// .replace(/'/g, ''');
|
|
202
|
-
|
|
203
|
-
// // Wrap code in a blockquote and ignore the language name
|
|
204
|
-
// return `<blockquote class="custom-code-block"><pre><code>${code}</code></pre></blockquote>`;
|
|
205
|
-
// };
|
|
206
|
-
|
|
207
|
-
// md.renderer.rules.fence = fence;
|
|
208
|
-
// });
|
|
209
|
-
|
|
210
|
-
// // Render markdown content
|
|
211
|
-
// return md.render(content);
|
|
212
175
|
|
|
213
176
|
marked.setOptions({
|
|
214
177
|
highlight: function (code, lang) {
|
|
215
|
-
const language =
|
|
216
|
-
|
|
178
|
+
const language = Prism.languages[lang] ? lang : "plaintext";
|
|
179
|
+
if (language === "plaintext") return code;
|
|
180
|
+
return Prism.highlight(code, Prism.languages[language], language);
|
|
217
181
|
},
|
|
218
|
-
langPrefix: "
|
|
182
|
+
langPrefix: "language-",
|
|
219
183
|
});
|
|
220
184
|
|
|
221
185
|
return content?.length ? marked.parse(content) : "";
|
package/src/index.ts
CHANGED
|
@@ -6,9 +6,6 @@
|
|
|
6
6
|
* @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
|
|
7
7
|
* to get a dual-use commercial license to remove the GPL requirements.
|
|
8
8
|
*/
|
|
9
|
-
export * from "./search/search-web";
|
|
10
|
-
// Re-export MetaSearchAgent and search handlers (with search functions) for backward compatibility
|
|
11
|
-
export * from "./search";
|
|
12
9
|
export * from "./tokenize/word-to-root-stem";
|
|
13
10
|
export * from "./tokenize/suggest-complete-word";
|
|
14
11
|
export * from "./tokenize/text-to-topic-tokens";
|
package/dist/search/index.d.ts
DELETED
|
@@ -1,14 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @module extract-webpage/search
|
|
3
|
-
* @description Re-exports MetaSearchAgent from agent-toolkit with search functions
|
|
4
|
-
*/
|
|
5
|
-
export declare const searchHandlers: {
|
|
6
|
-
webSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
7
|
-
academicSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
8
|
-
writingAssistant: import('chat-agent-toolkit').MetaSearchAgent;
|
|
9
|
-
wolframAlphaSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
10
|
-
youtubeSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
11
|
-
redditSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
12
|
-
};
|
|
13
|
-
export { MetaSearchAgent, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, registerUploadFileLoader, splitTextIntoChunks, LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
|
|
14
|
-
export type { UploadFileLoader, MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
|
|
@@ -1,8 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @fileoverview Re-exports MetaSearchAgent from chat-agent-toolkit
|
|
3
|
-
* @deprecated Import from 'chat-agent-toolkit' instead
|
|
4
|
-
*/
|
|
5
|
-
export { MetaSearchAgent as default, searchHandlers, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, splitTextIntoChunks, } from 'chat-agent-toolkit';
|
|
6
|
-
export type { MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
|
|
7
|
-
export { LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
|
|
8
|
-
export { webSearchResponsePrompt, webSearchRetrieverPrompt, webSearchRetrieverFewShots, writingAssistantPrompt, } from 'chat-agent-toolkit';
|
|
@@ -1,47 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Search Web via SearXNG metasearch of all major search engines.
|
|
3
|
-
*/
|
|
4
|
-
export declare function searchWeb(query: string, options?: SearchOptions): Promise<SearxngSearchResult[] | SearchResponse>;
|
|
5
|
-
interface SearxngSearchOptions {
|
|
6
|
-
categories?: string[];
|
|
7
|
-
engines?: string[];
|
|
8
|
-
language?: string;
|
|
9
|
-
pageno?: number;
|
|
10
|
-
}
|
|
11
|
-
export declare const searchSearxng: (query: string, opts?: SearxngSearchOptions) => Promise<{
|
|
12
|
-
results: SearxngSearchResult[];
|
|
13
|
-
suggestions: string[];
|
|
14
|
-
}>;
|
|
15
|
-
interface SearchOptions {
|
|
16
|
-
category?: string | number;
|
|
17
|
-
recency?: string;
|
|
18
|
-
privateSearxng?: string | boolean | null;
|
|
19
|
-
maxRetries?: number;
|
|
20
|
-
page?: number;
|
|
21
|
-
safesearch?: boolean;
|
|
22
|
-
lang?: string;
|
|
23
|
-
proxy?: string | null;
|
|
24
|
-
useProxy?: boolean;
|
|
25
|
-
}
|
|
26
|
-
export interface SearxngSearchResult {
|
|
27
|
-
title: string;
|
|
28
|
-
url: string;
|
|
29
|
-
snippet?: string;
|
|
30
|
-
domain?: string;
|
|
31
|
-
favicon?: string;
|
|
32
|
-
score?: number;
|
|
33
|
-
source?: string;
|
|
34
|
-
date?: string;
|
|
35
|
-
img_src?: string;
|
|
36
|
-
thumbnail_src?: string;
|
|
37
|
-
thumbnail?: string;
|
|
38
|
-
content?: string;
|
|
39
|
-
author?: string;
|
|
40
|
-
iframe_src?: string;
|
|
41
|
-
}
|
|
42
|
-
export interface SearchResponse {
|
|
43
|
-
results: SearxngSearchResult[];
|
|
44
|
-
suggestions: string[];
|
|
45
|
-
infoboxes?: any[];
|
|
46
|
-
}
|
|
47
|
-
export {};
|
|
@@ -1,33 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Search Web via SearXNG metasearch of all major search engines.
|
|
3
|
-
* Options are 10 search categories, recency, and how many
|
|
4
|
-
* times to retry other domains if first time fails.
|
|
5
|
-
* SearXNG is a free internet metasearch engine which aggregates results from
|
|
6
|
-
* more than [180+ search sources](https://docs.searxng.org/user/configured_engines.html).
|
|
7
|
-
*
|
|
8
|
-
* [Searxng Overview](https://medium.com/@elmo92/search-in-peace-with-searxng-an-alternative-search-engine-that-keeps-your-searches-private-accd8cddd6fc)
|
|
9
|
-
* [Searxng Installation Guide](https://github.com/searxng/searxng-docker/tree/master)
|
|
10
|
-
*
|
|
11
|
-
* 
|
|
12
|
-
* @param {string} query - The search query string.
|
|
13
|
-
* @param {Object} [options]
|
|
14
|
-
* @param {string} options.category default=general - ["general", "news", "videos", "images",
|
|
15
|
-
* "science","it", "files", "social+media", "map", "music"]
|
|
16
|
-
* @param {string} options.recency default=all - ["all", "day", "week", "month", "year"]
|
|
17
|
-
* @param {string|boolean} options.privateSearxng default=null - Use your custom domain SearXNG
|
|
18
|
-
* @param {number} options.maxRetries default=3 - Maximum number of retry attempts if the initial search fails.
|
|
19
|
-
* @param {number} options.page default=1 - The page number to retrieve.
|
|
20
|
-
* @param {boolean} options.safesearch default=false - Whether to block adult content.
|
|
21
|
-
* @param {string} options.lang default="en-US" - The language to use for the search.
|
|
22
|
-
* @param {string} options.proxy default=false - Use corsproxy.io to access in frontend JS
|
|
23
|
-
* @returns {Promise<Array<{title: string, url: string, snippet: string, domain: string, favicon: string, path: string, engines: string[]}>>} An array of search result objects.
|
|
24
|
-
* @example const advancedResults = await searchWeb('Node.js', {
|
|
25
|
-
* category: 2,
|
|
26
|
-
* recency: 1,
|
|
27
|
-
* maxRetries: 5
|
|
28
|
-
* });
|
|
29
|
-
* @category Search
|
|
30
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
31
|
-
* [Heiser, M., Tauber, A., Flament, A., et al. (2014-)](https://github.com/searxng/searxng/graphs/contributors)
|
|
32
|
-
*/
|
|
33
|
-
export declare function searchWeb(query: any, options?: any): Promise<any>;
|
package/dist/search/tavily.d.ts
DELETED
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
interface TavilySearchOptions {
|
|
2
|
-
searchDepth?: "basic" | "advanced";
|
|
3
|
-
maxResults?: number;
|
|
4
|
-
includeDomains?: string[];
|
|
5
|
-
excludeDomains?: string[];
|
|
6
|
-
}
|
|
7
|
-
interface TavilySearchResult {
|
|
8
|
-
title: string;
|
|
9
|
-
url: string;
|
|
10
|
-
content: string;
|
|
11
|
-
score: number;
|
|
12
|
-
raw_content?: string;
|
|
13
|
-
}
|
|
14
|
-
export declare const searchTavily: (query: string, opts?: TavilySearchOptions) => Promise<{
|
|
15
|
-
results: TavilySearchResult[];
|
|
16
|
-
suggestions: string[];
|
|
17
|
-
}>;
|
|
18
|
-
export declare const getTavilyApiKey: () => any;
|
|
19
|
-
export declare const isTavilyConfigured: () => boolean;
|
|
20
|
-
export {};
|
|
@@ -1,62 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* ### Tardigrade the Web Crawler
|
|
3
|
-
*
|
|
4
|
-
* 1. **Use Fetch API, check for bot detection.
|
|
5
|
-
* Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
|
|
6
|
-
* Scraping internet pages is a [free speech right
|
|
7
|
-
* ](https://blog.apify.com/is-web-scraping-legal/).
|
|
8
|
-
* 2. Features: timeout, redirects, default UA, referer as google, and bot
|
|
9
|
-
* detection checking. <br />
|
|
10
|
-
* 3. If fetch method does not get needed HTML, use Docker proxy as backup.
|
|
11
|
-
*
|
|
12
|
-
* 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
|
|
13
|
-
* container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
|
|
14
|
-
* secondary in-page API requests after the initial page request, including user login and cookie storage.
|
|
15
|
-
* 5. Bypass Cloudflare bot check: A webpage proxy that request
|
|
16
|
-
* through Chromium (puppeteer) - can be used to bypass Cloudflare
|
|
17
|
-
* anti bot using cookie id javascript method.
|
|
18
|
-
* 6. Send your request to the server with the port 3000 and add your URL to the "url"
|
|
19
|
-
* query string like this: `http://localhost:3000/?url=https://example.org`
|
|
20
|
-
*
|
|
21
|
-
* 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
|
|
22
|
-
* and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
|
|
23
|
-
* [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
|
|
24
|
-
* [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
|
|
25
|
-
* [Proxy-Cheap](https://app.proxy-cheap.com/order)
|
|
26
|
-
* [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
|
|
27
|
-
*
|
|
28
|
-
* @param {string} url - any domain's URL
|
|
29
|
-
* @param {Object} [options]
|
|
30
|
-
* @param {number} options.timeout default=5 - abort request if not retrived, in seconds
|
|
31
|
-
* @param {number} options.maxRedirects default=3 - max redirects to follow
|
|
32
|
-
* @param {number} options.checkBotDetection default=true - check for bot detection messages
|
|
33
|
-
* @param {number} options.changeReferer default=true - set referer as google
|
|
34
|
-
* @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
|
|
35
|
-
* @param {string} options.proxy default=false - use proxy url
|
|
36
|
-
* @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
|
|
37
|
-
* @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
|
|
38
|
-
* @category Extract
|
|
39
|
-
* @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
|
|
40
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
41
|
-
* @license MIT
|
|
42
|
-
*/
|
|
43
|
-
export declare function scrapeURL(url: any, options?: any): Promise<any>;
|
|
44
|
-
/**
|
|
45
|
-
* As backup, scrape with JINA to get html
|
|
46
|
-
* @param {string} url
|
|
47
|
-
* @returns {Promise<string>}
|
|
48
|
-
*/
|
|
49
|
-
export declare function scrapeJINA(url: any, timeout?: number): Promise<string>;
|
|
50
|
-
/**
|
|
51
|
-
* Fetches and parses the robots.txt file for a given URL.
|
|
52
|
-
* @param {string} url - The base URL to fetch the robots.txt from.
|
|
53
|
-
* @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
|
|
54
|
-
*/
|
|
55
|
-
export declare function fetchScrapingRules(url: any): Promise<{
|
|
56
|
-
directives: {};
|
|
57
|
-
crawlDelay: {};
|
|
58
|
-
sitemaps: any[];
|
|
59
|
-
preferredHost: any;
|
|
60
|
-
} | {
|
|
61
|
-
error: string;
|
|
62
|
-
}>;
|