extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,301 @@
1
+ /**
2
+ * @fileoverview Unit tests for URL scraping functionality
3
+ */
4
+ import { scrapeURL, scrapeJINA } from "../url-to-html";
5
+ import grab from "../utils/grab";
6
+
7
+ // Mock grab-url
8
+ jest.mock("grab-url");
9
+ const mockGrab = grab as jest.MockedFunction<typeof grab>;
10
+
11
+ describe("scrapeURL", () => {
12
+ beforeEach(() => {
13
+ jest.clearAllMocks();
14
+ // Reset environment variables
15
+ delete process.env.SCRAPER_URL;
16
+ delete process.env.SCRAPER_API_KEY;
17
+ });
18
+
19
+ it("should successfully scrape a URL", async () => {
20
+ const mockHtml = "<html><body><h1>Test Page</h1></body></html>";
21
+ mockGrab.mockResolvedValueOnce(mockHtml);
22
+
23
+ const result = await scrapeURL("https://example.com");
24
+
25
+ expect(result).toBe(mockHtml);
26
+ expect(mockGrab).toHaveBeenCalledWith(
27
+ "https://example.com",
28
+ expect.objectContaining({
29
+ responseType: "text",
30
+ "User-Agent": expect.any(String),
31
+ signal: expect.any(AbortSignal),
32
+ })
33
+ );
34
+ });
35
+
36
+ it("should use custom timeout", async () => {
37
+ const mockHtml = "<html><body>Content</body></html>";
38
+ mockGrab.mockResolvedValueOnce(mockHtml);
39
+
40
+ await scrapeURL("https://example.com", { timeout: 30 });
41
+
42
+ expect(mockGrab).toHaveBeenCalledWith(
43
+ "https://example.com",
44
+ expect.objectContaining({
45
+ signal: expect.any(AbortSignal),
46
+ })
47
+ );
48
+ });
49
+
50
+ it("should handle 403 Forbidden errors", async () => {
51
+ const error = new Error("Forbidden") as Error & { status?: number };
52
+ error.status = 403;
53
+ mockGrab.mockRejectedValueOnce(error);
54
+
55
+ // Mock Cloudflare scraper
56
+ global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
57
+
58
+ // Mock JINA fallback
59
+ mockGrab.mockResolvedValueOnce("Title: Test\nMarkdown Content:\nFallback content");
60
+
61
+ const result = await scrapeURL("https://blocked-site.com");
62
+
63
+ expect(typeof result).toBe("string");
64
+ expect(result).toContain("Fallback content");
65
+ });
66
+
67
+ it("should return error object when all methods fail", async () => {
68
+ const error = new Error("Network error") as Error & { status?: number };
69
+ error.status = 500;
70
+ mockGrab.mockRejectedValue(error);
71
+
72
+ global.fetch = jest.fn().mockRejectedValue(new Error("Cloudflare failed"));
73
+
74
+ const result = await scrapeURL("https://failing-site.com");
75
+
76
+ expect(typeof result).toBe("object");
77
+ expect((result as any).error).toBeDefined();
78
+ });
79
+
80
+ it("should detect bot protection and retry", async () => {
81
+ const botDetectionHtml =
82
+ "<html><body>Please verify you are a human</body></html>";
83
+ const successHtml = "<html><body>Actual content</body></html>";
84
+
85
+ // First attempt returns bot detection
86
+ mockGrab.mockResolvedValueOnce(botDetectionHtml);
87
+
88
+ // Mock successful Cloudflare bypass
89
+ global.fetch = jest.fn().mockResolvedValueOnce({
90
+ ok: true,
91
+ json: async () => ({
92
+ html: successHtml,
93
+ loadTime: 1000,
94
+ challengeBypassed: true,
95
+ }),
96
+ } as any);
97
+
98
+ const result = await scrapeURL("https://protected-site.com", {
99
+ checkBotDetection: true,
100
+ });
101
+
102
+ expect(result).toBe(successHtml);
103
+ expect(global.fetch).toHaveBeenCalled();
104
+ });
105
+
106
+ it("should handle bot detection with JINA fallback", async () => {
107
+ const botDetectionHtml = "<html><body>Access to this page has been denied</body></html>";
108
+
109
+ // First attempt returns bot detection
110
+ mockGrab.mockResolvedValueOnce(botDetectionHtml);
111
+
112
+ // Cloudflare fails
113
+ global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
114
+
115
+ // JINA succeeds
116
+ mockGrab.mockResolvedValueOnce(
117
+ "Title: Test Article\n===============\nMarkdown Content:\n# Article\nContent here"
118
+ );
119
+
120
+ const result = await scrapeURL("https://protected-site.com", {
121
+ checkBotDetection: true,
122
+ });
123
+
124
+ expect(typeof result).toBe("string");
125
+ expect(result).toContain("Article");
126
+ });
127
+
128
+ it("should return error when bot detection persists", async () => {
129
+ const botDetectionHtml = "<html><body>Cloudflare Ray ID found</body></html>";
130
+
131
+ // Initial request returns bot detection
132
+ mockGrab.mockResolvedValueOnce(botDetectionHtml);
133
+
134
+ // Cloudflare attempt also fails
135
+ global.fetch = jest.fn().mockResolvedValueOnce({
136
+ ok: true,
137
+ json: async () => ({
138
+ html: botDetectionHtml, // Still bot detection
139
+ }),
140
+ } as any);
141
+
142
+ const result = await scrapeURL("https://protected-site.com", {
143
+ checkBotDetection: true,
144
+ });
145
+
146
+ expect((result as any).error).toBe("Bot detected");
147
+ });
148
+
149
+ it("should prepend proxy to URL if provided", async () => {
150
+ const mockHtml = "<html><body>Content</body></html>";
151
+ mockGrab.mockResolvedValueOnce(mockHtml);
152
+
153
+ await scrapeURL("https://example.com", {
154
+ proxy: "https://proxy.example.com/",
155
+ });
156
+
157
+ expect(mockGrab).toHaveBeenCalledWith(
158
+ "https://proxy.example.com/https://example.com",
159
+ expect.any(Object)
160
+ );
161
+ });
162
+
163
+ it("should use different user agents", async () => {
164
+ const mockHtml = "<html><body>Content</body></html>";
165
+ mockGrab.mockResolvedValueOnce(mockHtml);
166
+
167
+ await scrapeURL("https://example.com", { userAgentIndex: 1 });
168
+
169
+ expect(mockGrab).toHaveBeenCalledWith(
170
+ "https://example.com",
171
+ expect.objectContaining({
172
+ "User-Agent": expect.stringContaining("Chrome/85"),
173
+ })
174
+ );
175
+ });
176
+
177
+ it("should add referer header when requested", async () => {
178
+ const mockHtml = "<html><body>Content</body></html>";
179
+ mockGrab.mockResolvedValueOnce(mockHtml);
180
+
181
+ await scrapeURL("https://example.com", { changeReferer: 1 });
182
+
183
+ expect(mockGrab).toHaveBeenCalledWith(
184
+ "https://example.com",
185
+ expect.objectContaining({
186
+ Referer: "https://www.google.com/",
187
+ })
188
+ );
189
+ });
190
+
191
+ it("should check robots.txt when requested", async () => {
192
+ mockGrab
193
+ .mockResolvedValueOnce("User-agent: *\nDisallow: /admin\n") // robots.txt
194
+ .mockResolvedValueOnce("<html><body>Content</body></html>"); // actual page
195
+
196
+ const result = await scrapeURL("https://example.com/page", {
197
+ checkRobotsAllowed: true,
198
+ });
199
+
200
+ expect(typeof result).toBe("string");
201
+ expect(result).toContain("Content");
202
+ });
203
+
204
+ it("should block scraping if robots.txt forbids", async () => {
205
+ mockGrab.mockResolvedValueOnce("User-agent: *\nDisallow: /\n"); // Block all
206
+
207
+ const result = await scrapeURL("https://example.com/page", {
208
+ checkRobotsAllowed: true,
209
+ });
210
+
211
+ expect((result as any).error).toBe("Robots.txt forbids to scrape there");
212
+ });
213
+ });
214
+
215
+ describe("scrapeJINA", () => {
216
+ beforeEach(() => {
217
+ jest.clearAllMocks();
218
+ });
219
+
220
+ it("should extract content from JINA API", async () => {
221
+ const mockJinaResponse = `Title: Test Article
222
+ URL Source: https://example.com
223
+ Published Time: 2024-01-01
224
+ Markdown Content:
225
+ ===============
226
+
227
+ # Test Article
228
+
229
+ This is the article content in markdown format.
230
+ `;
231
+
232
+ mockGrab.mockResolvedValueOnce(mockJinaResponse);
233
+
234
+ const result = await scrapeJINA("https://example.com/article");
235
+
236
+ expect(mockGrab).toHaveBeenCalledWith(
237
+ "https://r.jina.ai/https://example.com/article",
238
+ {
239
+ responseType: "text",
240
+ timeout: 30,
241
+ }
242
+ );
243
+ expect(result).toContain("<title>Test Article</title>");
244
+ expect(result).toContain("article content");
245
+ });
246
+
247
+ it("should handle JINA with separator", async () => {
248
+ const mockJinaResponse = `Some header info
249
+ ===============
250
+ Markdown Content:
251
+
252
+ Content after separator
253
+ `;
254
+
255
+ mockGrab.mockResolvedValueOnce(mockJinaResponse);
256
+
257
+ const result = await scrapeJINA("https://example.com");
258
+
259
+ expect(result).toContain("Content after separator");
260
+ expect(result).not.toContain("Some header info");
261
+ });
262
+
263
+ it("should throw error when JINA fetch fails", async () => {
264
+ mockGrab.mockRejectedValueOnce(new Error("JINA timeout"));
265
+
266
+ await expect(scrapeJINA("https://example.com")).rejects.toThrow(
267
+ "JINA scraping failed"
268
+ );
269
+ });
270
+
271
+ it("should throw error when JINA returns invalid data", async () => {
272
+ mockGrab.mockResolvedValueOnce(null as any);
273
+
274
+ await expect(scrapeJINA("https://example.com")).rejects.toThrow(
275
+ "JINA returned invalid or empty data"
276
+ );
277
+ });
278
+
279
+ it("should convert markdown to HTML", async () => {
280
+ const mockJinaResponse = `Title: Markdown Test
281
+ ===============
282
+ Markdown Content:
283
+
284
+ # Heading 1
285
+
286
+ This is **bold** and *italic* text.
287
+
288
+ - List item 1
289
+ - List item 2
290
+ `;
291
+
292
+ mockGrab.mockResolvedValueOnce(mockJinaResponse);
293
+
294
+ const result = await scrapeJINA("https://example.com");
295
+
296
+ expect(result).toContain("<h1>");
297
+ expect(result).toContain("<strong>");
298
+ expect(result).toContain("<em>");
299
+ expect(result).toContain("<li>");
300
+ });
301
+ });