extract-webpage 1.2.49 → 1.2.51

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -6,8 +6,6 @@
6
6
  * @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
7
7
  * to get a dual-use commercial license to remove the GPL requirements.
8
8
  */
9
- export * from './search/search-web';
10
- export * from './search';
11
9
  export * from './tokenize/word-to-root-stem';
12
10
  export * from './tokenize/suggest-complete-word';
13
11
  export * from './tokenize/text-to-topic-tokens';
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.49",
3
+ "version": "1.2.51",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -18,12 +18,6 @@
18
18
  "import": "./dist/extract-webpage.es.js",
19
19
  "require": "./dist/extract-webpage.cjs.js"
20
20
  },
21
- "./search": {
22
- "types": "./src/search/index.ts",
23
- "react-server": "./src/search/index.ts",
24
- "import": "./src/search/index.ts",
25
- "require": "./src/search/index.ts"
26
- },
27
21
  "./*": {
28
22
  "types": "./src/*.ts",
29
23
  "react-server": "./src/*",
@@ -78,12 +72,12 @@
78
72
  "dependencies": {
79
73
  "@huggingface/transformers": "^3.8.1",
80
74
  "ai": "^5.0.0",
81
- "chat-agent-toolkit": "^1.2.48",
75
+ "chat-agent-toolkit": "^1.2.50",
82
76
  "chrono-node": "^2.9.0",
83
77
  "drizzle-orm": "^0.45.1",
84
- "extract-pdf": "^0.1.36",
78
+ "extract-pdf": "^0.1.38",
85
79
  "extract-youtube": "^1.0.36",
86
- "highlight.js": "^11.11.1",
80
+ "prismjs": "^1.30.0",
87
81
  "html-entities": "^2.6.0",
88
82
  "js-yaml": "^4.1.1",
89
83
  "jsdom": "^28.1.0",
@@ -92,7 +86,6 @@
92
86
  "marked": "^17.0.4",
93
87
  "node-fetch": "^3.3.2",
94
88
  "qwksearch-api-client": "^0.0.12",
95
- "tldts": "^7.0.25",
96
89
  "zod": "^4.3.6"
97
90
  },
98
91
  "keywords": [
@@ -139,7 +139,10 @@ export function convertURLToAbsoluteURL(base, relative) {
139
139
  }
140
140
 
141
141
  import { marked } from "marked";
142
- import hljs from "highlight.js";
142
+ import Prism from "prismjs";
143
+ import loadLanguages from "prismjs/components/index.js";
144
+
145
+ loadLanguages();
143
146
 
144
147
  /**
145
148
  * Converts Markdown text to HTML. It handles the following Markdown elements:
@@ -169,53 +172,14 @@ import hljs from "highlight.js";
169
172
  export function convertMarkdownToHTML(content, toHtml = true) {
170
173
  if (!toHtml) return convertHTMLToMarkdown(content);
171
174
 
172
- // const md = new MarkdownIt({
173
- // highlight: function (str, lang) {
174
- // // If a language is provided and it's recognized by hljs
175
- // if (lang && hljs.getLanguage(lang)) {
176
- // try {
177
- // return (
178
- // '<pre><code class="hljs">' +
179
- // hljs.highlight(str, { language: lang, ignoreIllegals: true }).value
180
- // + '</code></pre>'
181
- // );
182
- // } catch (__) {}
183
- // }
184
-
185
- // // Default fallback for unsupported or no language
186
- // return (
187
- // '<pre><code class="hljs">' + md.utils.escapeHtml(str) + '</code></pre>'
188
- // );
189
- // },
190
- // }).use(function (md) {
191
- // // Override the default fence rule for handling code blocks
192
- // const fence = md.renderer.rules.fence || function (tokens, idx, options, env, slf) {
193
- // const token = tokens[idx];
194
- // const code = token.content
195
- // .trim() // Trim leading/trailing whitespace
196
- // .replace(/^[ \t]*/gm, '') // Remove leading whitespace while preserving relative indentation
197
- // .replace(/&/g, '&amp;') // Encode HTML special characters
198
- // .replace(/</g, '&lt;')
199
- // .replace(/>/g, '&gt;')
200
- // .replace(/"/g, '&quot;')
201
- // .replace(/'/g, '&#39;');
202
-
203
- // // Wrap code in a blockquote and ignore the language name
204
- // return `<blockquote class="custom-code-block"><pre><code>${code}</code></pre></blockquote>`;
205
- // };
206
-
207
- // md.renderer.rules.fence = fence;
208
- // });
209
-
210
- // // Render markdown content
211
- // return md.render(content);
212
175
 
213
176
  marked.setOptions({
214
177
  highlight: function (code, lang) {
215
- const language = hljs.getLanguage(lang) ? lang : "plaintext";
216
- return hljs.highlight(code, { language }).value;
178
+ const language = Prism.languages[lang] ? lang : "plaintext";
179
+ if (language === "plaintext") return code;
180
+ return Prism.highlight(code, Prism.languages[language], language);
217
181
  },
218
- langPrefix: "hljs language-",
182
+ langPrefix: "language-",
219
183
  });
220
184
 
221
185
  return content?.length ? marked.parse(content) : "";
package/src/index.ts CHANGED
@@ -6,9 +6,6 @@
6
6
  * @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
7
7
  * to get a dual-use commercial license to remove the GPL requirements.
8
8
  */
9
- export * from "./search/search-web";
10
- // Re-export MetaSearchAgent and search handlers (with search functions) for backward compatibility
11
- export * from "./search";
12
9
  export * from "./tokenize/word-to-root-stem";
13
10
  export * from "./tokenize/suggest-complete-word";
14
11
  export * from "./tokenize/text-to-topic-tokens";
@@ -1,14 +0,0 @@
1
- /**
2
- * @module extract-webpage/search
3
- * @description Re-exports MetaSearchAgent from agent-toolkit with search functions
4
- */
5
- export declare const searchHandlers: {
6
- webSearch: import('chat-agent-toolkit').MetaSearchAgent;
7
- academicSearch: import('chat-agent-toolkit').MetaSearchAgent;
8
- writingAssistant: import('chat-agent-toolkit').MetaSearchAgent;
9
- wolframAlphaSearch: import('chat-agent-toolkit').MetaSearchAgent;
10
- youtubeSearch: import('chat-agent-toolkit').MetaSearchAgent;
11
- redditSearch: import('chat-agent-toolkit').MetaSearchAgent;
12
- };
13
- export { MetaSearchAgent, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, registerUploadFileLoader, splitTextIntoChunks, LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
14
- export type { UploadFileLoader, MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
@@ -1,8 +0,0 @@
1
- /**
2
- * @fileoverview Re-exports MetaSearchAgent from chat-agent-toolkit
3
- * @deprecated Import from 'chat-agent-toolkit' instead
4
- */
5
- export { MetaSearchAgent as default, searchHandlers, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, splitTextIntoChunks, } from 'chat-agent-toolkit';
6
- export type { MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
7
- export { LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
8
- export { webSearchResponsePrompt, webSearchRetrieverPrompt, webSearchRetrieverFewShots, writingAssistantPrompt, } from 'chat-agent-toolkit';
@@ -1,47 +0,0 @@
1
- /**
2
- * Search Web via SearXNG metasearch of all major search engines.
3
- */
4
- export declare function searchWeb(query: string, options?: SearchOptions): Promise<SearxngSearchResult[] | SearchResponse>;
5
- interface SearxngSearchOptions {
6
- categories?: string[];
7
- engines?: string[];
8
- language?: string;
9
- pageno?: number;
10
- }
11
- export declare const searchSearxng: (query: string, opts?: SearxngSearchOptions) => Promise<{
12
- results: SearxngSearchResult[];
13
- suggestions: string[];
14
- }>;
15
- interface SearchOptions {
16
- category?: string | number;
17
- recency?: string;
18
- privateSearxng?: string | boolean | null;
19
- maxRetries?: number;
20
- page?: number;
21
- safesearch?: boolean;
22
- lang?: string;
23
- proxy?: string | null;
24
- useProxy?: boolean;
25
- }
26
- export interface SearxngSearchResult {
27
- title: string;
28
- url: string;
29
- snippet?: string;
30
- domain?: string;
31
- favicon?: string;
32
- score?: number;
33
- source?: string;
34
- date?: string;
35
- img_src?: string;
36
- thumbnail_src?: string;
37
- thumbnail?: string;
38
- content?: string;
39
- author?: string;
40
- iframe_src?: string;
41
- }
42
- export interface SearchResponse {
43
- results: SearxngSearchResult[];
44
- suggestions: string[];
45
- infoboxes?: any[];
46
- }
47
- export {};
@@ -1,33 +0,0 @@
1
- /**
2
- * Search Web via SearXNG metasearch of all major search engines.
3
- * Options are 10 search categories, recency, and how many
4
- * times to retry other domains if first time fails.
5
- * SearXNG is a free internet metasearch engine which aggregates results from
6
- * more than [180+ search sources](https://docs.searxng.org/user/configured_engines.html).
7
- *
8
- * [Searxng Overview](https://medium.com/@elmo92/search-in-peace-with-searxng-an-alternative-search-engine-that-keeps-your-searches-private-accd8cddd6fc)
9
- * [Searxng Installation Guide](https://github.com/searxng/searxng-docker/tree/master)
10
- *
11
- * ![google_dead](https://i.imgur.com/6rRpaY1.png)
12
- * @param {string} query - The search query string.
13
- * @param {Object} [options]
14
- * @param {string} options.category default=general - ["general", "news", "videos", "images",
15
- * "science","it", "files", "social+media", "map", "music"]
16
- * @param {string} options.recency default=all - ["all", "day", "week", "month", "year"]
17
- * @param {string|boolean} options.privateSearxng default=null - Use your custom domain SearXNG
18
- * @param {number} options.maxRetries default=3 - Maximum number of retry attempts if the initial search fails.
19
- * @param {number} options.page default=1 - The page number to retrieve.
20
- * @param {boolean} options.safesearch default=false - Whether to block adult content.
21
- * @param {string} options.lang default="en-US" - The language to use for the search.
22
- * @param {string} options.proxy default=false - Use corsproxy.io to access in frontend JS
23
- * @returns {Promise<Array<{title: string, url: string, snippet: string, domain: string, favicon: string, path: string, engines: string[]}>>} An array of search result objects.
24
- * @example const advancedResults = await searchWeb('Node.js', {
25
- * category: 2,
26
- * recency: 1,
27
- * maxRetries: 5
28
- * });
29
- * @category Search
30
- * @author [vtempest (2025)](https://github.com/vtempest)
31
- * [Heiser, M., Tauber, A., Flament, A., et al. (2014-)](https://github.com/searxng/searxng/graphs/contributors)
32
- */
33
- export declare function searchWeb(query: any, options?: any): Promise<any>;
@@ -1,20 +0,0 @@
1
- interface TavilySearchOptions {
2
- searchDepth?: "basic" | "advanced";
3
- maxResults?: number;
4
- includeDomains?: string[];
5
- excludeDomains?: string[];
6
- }
7
- interface TavilySearchResult {
8
- title: string;
9
- url: string;
10
- content: string;
11
- score: number;
12
- raw_content?: string;
13
- }
14
- export declare const searchTavily: (query: string, opts?: TavilySearchOptions) => Promise<{
15
- results: TavilySearchResult[];
16
- suggestions: string[];
17
- }>;
18
- export declare const getTavilyApiKey: () => any;
19
- export declare const isTavilyConfigured: () => boolean;
20
- export {};
@@ -1,62 +0,0 @@
1
- /**
2
- * ### Tardigrade the Web Crawler
3
- *
4
- * 1. **Use Fetch API, check for bot detection.
5
- * Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
6
- * Scraping internet pages is a [free speech right
7
- * ](https://blog.apify.com/is-web-scraping-legal/).
8
- * 2. Features: timeout, redirects, default UA, referer as google, and bot
9
- * detection checking. <br />
10
- * 3. If fetch method does not get needed HTML, use Docker proxy as backup.
11
- *
12
- * 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
13
- * container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
14
- * secondary in-page API requests after the initial page request, including user login and cookie storage.
15
- * 5. Bypass Cloudflare bot check: A webpage proxy that request
16
- * through Chromium (puppeteer) - can be used to bypass Cloudflare
17
- * anti bot using cookie id javascript method.
18
- * 6. Send your request to the server with the port 3000 and add your URL to the "url"
19
- * query string like this: `http://localhost:3000/?url=https://example.org`
20
- *
21
- * 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
22
- * and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
23
- * [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
24
- * [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
25
- * [Proxy-Cheap](https://app.proxy-cheap.com/order)
26
- * [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
27
- *
28
- * @param {string} url - any domain's URL
29
- * @param {Object} [options]
30
- * @param {number} options.timeout default=5 - abort request if not retrived, in seconds
31
- * @param {number} options.maxRedirects default=3 - max redirects to follow
32
- * @param {number} options.checkBotDetection default=true - check for bot detection messages
33
- * @param {number} options.changeReferer default=true - set referer as google
34
- * @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
35
- * @param {string} options.proxy default=false - use proxy url
36
- * @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
37
- * @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
38
- * @category Extract
39
- * @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
40
- * @author [vtempest (2025)](https://github.com/vtempest)
41
- * @license MIT
42
- */
43
- export declare function scrapeURL(url: any, options?: any): Promise<any>;
44
- /**
45
- * As backup, scrape with JINA to get html
46
- * @param {string} url
47
- * @returns {Promise<string>}
48
- */
49
- export declare function scrapeJINA(url: any, timeout?: number): Promise<string>;
50
- /**
51
- * Fetches and parses the robots.txt file for a given URL.
52
- * @param {string} url - The base URL to fetch the robots.txt from.
53
- * @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
54
- */
55
- export declare function fetchScrapingRules(url: any): Promise<{
56
- directives: {};
57
- crawlDelay: {};
58
- sitemaps: any[];
59
- preferredHost: any;
60
- } | {
61
- error: string;
62
- }>;