extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Unit tests for URL scraping functionality
|
|
3
|
+
*/
|
|
4
|
+
import { scrapeURL, scrapeJINA } from "../url-to-html";
|
|
5
|
+
import grab from "../utils/grab";
|
|
6
|
+
|
|
7
|
+
// Mock grab-url
|
|
8
|
+
jest.mock("grab-url");
|
|
9
|
+
const mockGrab = grab as jest.MockedFunction<typeof grab>;
|
|
10
|
+
|
|
11
|
+
describe("scrapeURL", () => {
|
|
12
|
+
beforeEach(() => {
|
|
13
|
+
jest.clearAllMocks();
|
|
14
|
+
// Reset environment variables
|
|
15
|
+
delete process.env.SCRAPER_URL;
|
|
16
|
+
delete process.env.SCRAPER_API_KEY;
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
it("should successfully scrape a URL", async () => {
|
|
20
|
+
const mockHtml = "<html><body><h1>Test Page</h1></body></html>";
|
|
21
|
+
mockGrab.mockResolvedValueOnce(mockHtml);
|
|
22
|
+
|
|
23
|
+
const result = await scrapeURL("https://example.com");
|
|
24
|
+
|
|
25
|
+
expect(result).toBe(mockHtml);
|
|
26
|
+
expect(mockGrab).toHaveBeenCalledWith(
|
|
27
|
+
"https://example.com",
|
|
28
|
+
expect.objectContaining({
|
|
29
|
+
responseType: "text",
|
|
30
|
+
"User-Agent": expect.any(String),
|
|
31
|
+
signal: expect.any(AbortSignal),
|
|
32
|
+
})
|
|
33
|
+
);
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
it("should use custom timeout", async () => {
|
|
37
|
+
const mockHtml = "<html><body>Content</body></html>";
|
|
38
|
+
mockGrab.mockResolvedValueOnce(mockHtml);
|
|
39
|
+
|
|
40
|
+
await scrapeURL("https://example.com", { timeout: 30 });
|
|
41
|
+
|
|
42
|
+
expect(mockGrab).toHaveBeenCalledWith(
|
|
43
|
+
"https://example.com",
|
|
44
|
+
expect.objectContaining({
|
|
45
|
+
signal: expect.any(AbortSignal),
|
|
46
|
+
})
|
|
47
|
+
);
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
it("should handle 403 Forbidden errors", async () => {
|
|
51
|
+
const error = new Error("Forbidden") as Error & { status?: number };
|
|
52
|
+
error.status = 403;
|
|
53
|
+
mockGrab.mockRejectedValueOnce(error);
|
|
54
|
+
|
|
55
|
+
// Mock Cloudflare scraper
|
|
56
|
+
global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
|
|
57
|
+
|
|
58
|
+
// Mock JINA fallback
|
|
59
|
+
mockGrab.mockResolvedValueOnce("Title: Test\nMarkdown Content:\nFallback content");
|
|
60
|
+
|
|
61
|
+
const result = await scrapeURL("https://blocked-site.com");
|
|
62
|
+
|
|
63
|
+
expect(typeof result).toBe("string");
|
|
64
|
+
expect(result).toContain("Fallback content");
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
it("should return error object when all methods fail", async () => {
|
|
68
|
+
const error = new Error("Network error") as Error & { status?: number };
|
|
69
|
+
error.status = 500;
|
|
70
|
+
mockGrab.mockRejectedValue(error);
|
|
71
|
+
|
|
72
|
+
global.fetch = jest.fn().mockRejectedValue(new Error("Cloudflare failed"));
|
|
73
|
+
|
|
74
|
+
const result = await scrapeURL("https://failing-site.com");
|
|
75
|
+
|
|
76
|
+
expect(typeof result).toBe("object");
|
|
77
|
+
expect((result as any).error).toBeDefined();
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
it("should detect bot protection and retry", async () => {
|
|
81
|
+
const botDetectionHtml =
|
|
82
|
+
"<html><body>Please verify you are a human</body></html>";
|
|
83
|
+
const successHtml = "<html><body>Actual content</body></html>";
|
|
84
|
+
|
|
85
|
+
// First attempt returns bot detection
|
|
86
|
+
mockGrab.mockResolvedValueOnce(botDetectionHtml);
|
|
87
|
+
|
|
88
|
+
// Mock successful Cloudflare bypass
|
|
89
|
+
global.fetch = jest.fn().mockResolvedValueOnce({
|
|
90
|
+
ok: true,
|
|
91
|
+
json: async () => ({
|
|
92
|
+
html: successHtml,
|
|
93
|
+
loadTime: 1000,
|
|
94
|
+
challengeBypassed: true,
|
|
95
|
+
}),
|
|
96
|
+
} as any);
|
|
97
|
+
|
|
98
|
+
const result = await scrapeURL("https://protected-site.com", {
|
|
99
|
+
checkBotDetection: true,
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
expect(result).toBe(successHtml);
|
|
103
|
+
expect(global.fetch).toHaveBeenCalled();
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it("should handle bot detection with JINA fallback", async () => {
|
|
107
|
+
const botDetectionHtml = "<html><body>Access to this page has been denied</body></html>";
|
|
108
|
+
|
|
109
|
+
// First attempt returns bot detection
|
|
110
|
+
mockGrab.mockResolvedValueOnce(botDetectionHtml);
|
|
111
|
+
|
|
112
|
+
// Cloudflare fails
|
|
113
|
+
global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
|
|
114
|
+
|
|
115
|
+
// JINA succeeds
|
|
116
|
+
mockGrab.mockResolvedValueOnce(
|
|
117
|
+
"Title: Test Article\n===============\nMarkdown Content:\n# Article\nContent here"
|
|
118
|
+
);
|
|
119
|
+
|
|
120
|
+
const result = await scrapeURL("https://protected-site.com", {
|
|
121
|
+
checkBotDetection: true,
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
expect(typeof result).toBe("string");
|
|
125
|
+
expect(result).toContain("Article");
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
it("should return error when bot detection persists", async () => {
|
|
129
|
+
const botDetectionHtml = "<html><body>Cloudflare Ray ID found</body></html>";
|
|
130
|
+
|
|
131
|
+
// Initial request returns bot detection
|
|
132
|
+
mockGrab.mockResolvedValueOnce(botDetectionHtml);
|
|
133
|
+
|
|
134
|
+
// Cloudflare attempt also fails
|
|
135
|
+
global.fetch = jest.fn().mockResolvedValueOnce({
|
|
136
|
+
ok: true,
|
|
137
|
+
json: async () => ({
|
|
138
|
+
html: botDetectionHtml, // Still bot detection
|
|
139
|
+
}),
|
|
140
|
+
} as any);
|
|
141
|
+
|
|
142
|
+
const result = await scrapeURL("https://protected-site.com", {
|
|
143
|
+
checkBotDetection: true,
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
expect((result as any).error).toBe("Bot detected");
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
it("should prepend proxy to URL if provided", async () => {
|
|
150
|
+
const mockHtml = "<html><body>Content</body></html>";
|
|
151
|
+
mockGrab.mockResolvedValueOnce(mockHtml);
|
|
152
|
+
|
|
153
|
+
await scrapeURL("https://example.com", {
|
|
154
|
+
proxy: "https://proxy.example.com/",
|
|
155
|
+
});
|
|
156
|
+
|
|
157
|
+
expect(mockGrab).toHaveBeenCalledWith(
|
|
158
|
+
"https://proxy.example.com/https://example.com",
|
|
159
|
+
expect.any(Object)
|
|
160
|
+
);
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
it("should use different user agents", async () => {
|
|
164
|
+
const mockHtml = "<html><body>Content</body></html>";
|
|
165
|
+
mockGrab.mockResolvedValueOnce(mockHtml);
|
|
166
|
+
|
|
167
|
+
await scrapeURL("https://example.com", { userAgentIndex: 1 });
|
|
168
|
+
|
|
169
|
+
expect(mockGrab).toHaveBeenCalledWith(
|
|
170
|
+
"https://example.com",
|
|
171
|
+
expect.objectContaining({
|
|
172
|
+
"User-Agent": expect.stringContaining("Chrome/85"),
|
|
173
|
+
})
|
|
174
|
+
);
|
|
175
|
+
});
|
|
176
|
+
|
|
177
|
+
it("should add referer header when requested", async () => {
|
|
178
|
+
const mockHtml = "<html><body>Content</body></html>";
|
|
179
|
+
mockGrab.mockResolvedValueOnce(mockHtml);
|
|
180
|
+
|
|
181
|
+
await scrapeURL("https://example.com", { changeReferer: 1 });
|
|
182
|
+
|
|
183
|
+
expect(mockGrab).toHaveBeenCalledWith(
|
|
184
|
+
"https://example.com",
|
|
185
|
+
expect.objectContaining({
|
|
186
|
+
Referer: "https://www.google.com/",
|
|
187
|
+
})
|
|
188
|
+
);
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
it("should check robots.txt when requested", async () => {
|
|
192
|
+
mockGrab
|
|
193
|
+
.mockResolvedValueOnce("User-agent: *\nDisallow: /admin\n") // robots.txt
|
|
194
|
+
.mockResolvedValueOnce("<html><body>Content</body></html>"); // actual page
|
|
195
|
+
|
|
196
|
+
const result = await scrapeURL("https://example.com/page", {
|
|
197
|
+
checkRobotsAllowed: true,
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
expect(typeof result).toBe("string");
|
|
201
|
+
expect(result).toContain("Content");
|
|
202
|
+
});
|
|
203
|
+
|
|
204
|
+
it("should block scraping if robots.txt forbids", async () => {
|
|
205
|
+
mockGrab.mockResolvedValueOnce("User-agent: *\nDisallow: /\n"); // Block all
|
|
206
|
+
|
|
207
|
+
const result = await scrapeURL("https://example.com/page", {
|
|
208
|
+
checkRobotsAllowed: true,
|
|
209
|
+
});
|
|
210
|
+
|
|
211
|
+
expect((result as any).error).toBe("Robots.txt forbids to scrape there");
|
|
212
|
+
});
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
describe("scrapeJINA", () => {
|
|
216
|
+
beforeEach(() => {
|
|
217
|
+
jest.clearAllMocks();
|
|
218
|
+
});
|
|
219
|
+
|
|
220
|
+
it("should extract content from JINA API", async () => {
|
|
221
|
+
const mockJinaResponse = `Title: Test Article
|
|
222
|
+
URL Source: https://example.com
|
|
223
|
+
Published Time: 2024-01-01
|
|
224
|
+
Markdown Content:
|
|
225
|
+
===============
|
|
226
|
+
|
|
227
|
+
# Test Article
|
|
228
|
+
|
|
229
|
+
This is the article content in markdown format.
|
|
230
|
+
`;
|
|
231
|
+
|
|
232
|
+
mockGrab.mockResolvedValueOnce(mockJinaResponse);
|
|
233
|
+
|
|
234
|
+
const result = await scrapeJINA("https://example.com/article");
|
|
235
|
+
|
|
236
|
+
expect(mockGrab).toHaveBeenCalledWith(
|
|
237
|
+
"https://r.jina.ai/https://example.com/article",
|
|
238
|
+
{
|
|
239
|
+
responseType: "text",
|
|
240
|
+
timeout: 30,
|
|
241
|
+
}
|
|
242
|
+
);
|
|
243
|
+
expect(result).toContain("<title>Test Article</title>");
|
|
244
|
+
expect(result).toContain("article content");
|
|
245
|
+
});
|
|
246
|
+
|
|
247
|
+
it("should handle JINA with separator", async () => {
|
|
248
|
+
const mockJinaResponse = `Some header info
|
|
249
|
+
===============
|
|
250
|
+
Markdown Content:
|
|
251
|
+
|
|
252
|
+
Content after separator
|
|
253
|
+
`;
|
|
254
|
+
|
|
255
|
+
mockGrab.mockResolvedValueOnce(mockJinaResponse);
|
|
256
|
+
|
|
257
|
+
const result = await scrapeJINA("https://example.com");
|
|
258
|
+
|
|
259
|
+
expect(result).toContain("Content after separator");
|
|
260
|
+
expect(result).not.toContain("Some header info");
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
it("should throw error when JINA fetch fails", async () => {
|
|
264
|
+
mockGrab.mockRejectedValueOnce(new Error("JINA timeout"));
|
|
265
|
+
|
|
266
|
+
await expect(scrapeJINA("https://example.com")).rejects.toThrow(
|
|
267
|
+
"JINA scraping failed"
|
|
268
|
+
);
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
it("should throw error when JINA returns invalid data", async () => {
|
|
272
|
+
mockGrab.mockResolvedValueOnce(null as any);
|
|
273
|
+
|
|
274
|
+
await expect(scrapeJINA("https://example.com")).rejects.toThrow(
|
|
275
|
+
"JINA returned invalid or empty data"
|
|
276
|
+
);
|
|
277
|
+
});
|
|
278
|
+
|
|
279
|
+
it("should convert markdown to HTML", async () => {
|
|
280
|
+
const mockJinaResponse = `Title: Markdown Test
|
|
281
|
+
===============
|
|
282
|
+
Markdown Content:
|
|
283
|
+
|
|
284
|
+
# Heading 1
|
|
285
|
+
|
|
286
|
+
This is **bold** and *italic* text.
|
|
287
|
+
|
|
288
|
+
- List item 1
|
|
289
|
+
- List item 2
|
|
290
|
+
`;
|
|
291
|
+
|
|
292
|
+
mockGrab.mockResolvedValueOnce(mockJinaResponse);
|
|
293
|
+
|
|
294
|
+
const result = await scrapeJINA("https://example.com");
|
|
295
|
+
|
|
296
|
+
expect(result).toContain("<h1>");
|
|
297
|
+
expect(result).toContain("<strong>");
|
|
298
|
+
expect(result).toContain("<em>");
|
|
299
|
+
expect(result).toContain("<li>");
|
|
300
|
+
});
|
|
301
|
+
});
|