extract-webpage 1.2.187 → 1.2.188
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js +1 -1
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js +1 -1
- package/dist/extract-webpage.es.js.map +1 -1
- package/package.json +3 -3
- package/src/url-to-content/__tests__/url-to-content.test.ts +33 -36
- package/src/url-to-content/__tests__/url-to-html.test.ts +23 -29
- package/src/url-to-content/url-to-html.ts +1 -10
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-webpage",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.188",
|
|
4
4
|
"module": "./dist/extract-webpage.es.js",
|
|
5
5
|
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
6
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
@@ -74,10 +74,10 @@
|
|
|
74
74
|
"dependencies": {
|
|
75
75
|
"@huggingface/transformers": "^3.8.1",
|
|
76
76
|
"ai": "^5.0.0",
|
|
77
|
-
"chat-agent-toolkit": "^1.2.
|
|
77
|
+
"chat-agent-toolkit": "^1.2.188",
|
|
78
78
|
"chrono-node": "^2.9.0",
|
|
79
79
|
"drizzle-orm": "^0.45.1",
|
|
80
|
-
"extract-pdf": "^0.1.
|
|
80
|
+
"extract-pdf": "^0.1.175",
|
|
81
81
|
"extract-youtube": "^1.0.103",
|
|
82
82
|
"html-entities": "^2.6.0",
|
|
83
83
|
"js-yaml": "^4.1.1",
|
|
@@ -1,50 +1,45 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* @fileoverview Unit tests for URL content extraction
|
|
3
3
|
*/
|
|
4
|
-
import { beforeEach, describe, expect, it, vi, type MockedFunction } from "vitest";
|
|
5
4
|
import { extractContent } from "../url-to-content";
|
|
6
5
|
import { scrapeURL } from "../url-to-html";
|
|
7
6
|
import { extractContentAndCite } from "../../html-to-content/html-to-content";
|
|
8
|
-
import { convertYoutubeToText
|
|
9
|
-
import grab from "../../utils/grab";
|
|
7
|
+
import { convertYoutubeToText } from "../youtube-helpers";
|
|
10
8
|
import { convertPDFToHTML } from "extract-pdf";
|
|
11
9
|
|
|
12
10
|
// Mock dependencies
|
|
13
|
-
|
|
14
|
-
scrapeURL:
|
|
11
|
+
jest.mock("../url-to-html", () => ({
|
|
12
|
+
scrapeURL: jest.fn(),
|
|
15
13
|
}));
|
|
16
14
|
|
|
17
|
-
|
|
18
|
-
extractContentAndCite:
|
|
15
|
+
jest.mock("../../html-to-content/html-to-content", () => ({
|
|
16
|
+
extractContentAndCite: jest.fn(),
|
|
19
17
|
}));
|
|
20
18
|
|
|
21
|
-
|
|
22
|
-
getURLYoutubeVideo:
|
|
23
|
-
convertYoutubeToText:
|
|
19
|
+
jest.mock("../youtube-helpers", () => ({
|
|
20
|
+
getURLYoutubeVideo: jest.fn(),
|
|
21
|
+
convertYoutubeToText: jest.fn(),
|
|
24
22
|
}));
|
|
25
23
|
|
|
26
|
-
|
|
27
|
-
convertPDFToHTML:
|
|
24
|
+
jest.mock("extract-pdf", () => ({
|
|
25
|
+
convertPDFToHTML: jest.fn(),
|
|
28
26
|
}));
|
|
29
27
|
|
|
30
|
-
|
|
28
|
+
jest.mock("grab-url", () => jest.fn());
|
|
31
29
|
|
|
32
|
-
const mockScrapeURL = scrapeURL as MockedFunction<typeof scrapeURL>;
|
|
33
|
-
const mockExtractContentAndCite = extractContentAndCite as MockedFunction<
|
|
30
|
+
const mockScrapeURL = scrapeURL as jest.MockedFunction<typeof scrapeURL>;
|
|
31
|
+
const mockExtractContentAndCite = extractContentAndCite as jest.MockedFunction<
|
|
34
32
|
typeof extractContentAndCite
|
|
35
33
|
>;
|
|
36
34
|
const mockConvertYoutubeToText =
|
|
37
|
-
convertYoutubeToText as MockedFunction<typeof convertYoutubeToText>;
|
|
38
|
-
const mockConvertPDFToHTML = convertPDFToHTML as MockedFunction<
|
|
35
|
+
convertYoutubeToText as jest.MockedFunction<typeof convertYoutubeToText>;
|
|
36
|
+
const mockConvertPDFToHTML = convertPDFToHTML as jest.MockedFunction<
|
|
39
37
|
typeof convertPDFToHTML
|
|
40
38
|
>;
|
|
41
|
-
const mockGetURLYoutubeVideo =
|
|
42
|
-
getURLYoutubeVideo as MockedFunction<typeof getURLYoutubeVideo>;
|
|
43
|
-
const mockGrab = grab as MockedFunction<typeof grab>;
|
|
44
39
|
|
|
45
40
|
describe("extractContent", () => {
|
|
46
41
|
beforeEach(() => {
|
|
47
|
-
|
|
42
|
+
jest.clearAllMocks();
|
|
48
43
|
});
|
|
49
44
|
|
|
50
45
|
describe("URL extraction", () => {
|
|
@@ -67,10 +62,17 @@ describe("extractContent", () => {
|
|
|
67
62
|
expect(mockScrapeURL).toHaveBeenCalledWith("https://example.com/article", {
|
|
68
63
|
proxy: null,
|
|
69
64
|
});
|
|
70
|
-
expect(mockExtractContentAndCite).toHaveBeenCalledWith(
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
65
|
+
expect(mockExtractContentAndCite).toHaveBeenCalledWith(mockHtml, {
|
|
66
|
+
url: "https://example.com/article",
|
|
67
|
+
images: true,
|
|
68
|
+
links: true,
|
|
69
|
+
formatting: true,
|
|
70
|
+
absoluteURLs: true,
|
|
71
|
+
timeout: 10,
|
|
72
|
+
proxy: null,
|
|
73
|
+
citeFormatMonthFull: false,
|
|
74
|
+
citeFormatAuthorFull: true,
|
|
75
|
+
});
|
|
74
76
|
expect(result.title).toBe("Test Article");
|
|
75
77
|
expect(result.html).toBe("<p>Test content</p>");
|
|
76
78
|
});
|
|
@@ -82,8 +84,7 @@ describe("extractContent", () => {
|
|
|
82
84
|
|
|
83
85
|
const result = await extractContent("https://blocked-site.com/article");
|
|
84
86
|
|
|
85
|
-
|
|
86
|
-
expect(result.error).toBe("Failed to fetch HTML content");
|
|
87
|
+
expect(result.error).toBe("HTTP error: 403 Forbidden");
|
|
87
88
|
expect(mockExtractContentAndCite).not.toHaveBeenCalled();
|
|
88
89
|
});
|
|
89
90
|
|
|
@@ -119,10 +120,9 @@ describe("extractContent", () => {
|
|
|
119
120
|
it("should handle scrapeURL throwing an error", async () => {
|
|
120
121
|
mockScrapeURL.mockRejectedValueOnce(new Error("Network timeout"));
|
|
121
122
|
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
expect(result.error).toBe("Failed to scrape URL: Network timeout");
|
|
123
|
+
await expect(
|
|
124
|
+
extractContent("https://timeout.com/article")
|
|
125
|
+
).rejects.toThrow("Network timeout");
|
|
126
126
|
});
|
|
127
127
|
|
|
128
128
|
it("should handle extraction returning no HTML", async () => {
|
|
@@ -192,7 +192,8 @@ describe("extractContent", () => {
|
|
|
192
192
|
|
|
193
193
|
describe("YouTube extraction", () => {
|
|
194
194
|
it("should extract YouTube video transcript", async () => {
|
|
195
|
-
|
|
195
|
+
const youtubeHelpers = require("../youtube-helpers");
|
|
196
|
+
youtubeHelpers.getURLYoutubeVideo.mockReturnValueOnce("dQw4w9WgXcQ");
|
|
196
197
|
|
|
197
198
|
const mockTranscript = {
|
|
198
199
|
title: "Test Video",
|
|
@@ -268,10 +269,6 @@ describe("extractContent", () => {
|
|
|
268
269
|
};
|
|
269
270
|
|
|
270
271
|
mockConvertPDFToHTML.mockResolvedValueOnce(mockPdfContent as any);
|
|
271
|
-
// isUrlPDF sniffs the leading "%PDF-" magic bytes via grab().
|
|
272
|
-
mockGrab.mockResolvedValueOnce(
|
|
273
|
-
new Uint8Array([0x25, 0x50, 0x44, 0x46, 0x2d]).buffer as any
|
|
274
|
-
);
|
|
275
272
|
|
|
276
273
|
await extractContent(
|
|
277
274
|
"https://drive.google.com/file/d/ABC123/view"
|
|
@@ -1,17 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* @fileoverview Unit tests for URL scraping functionality
|
|
3
3
|
*/
|
|
4
|
-
import { beforeEach, describe, expect, it, vi, type MockedFunction } from "vitest";
|
|
5
4
|
import { scrapeURL, scrapeJINA } from "../url-to-html";
|
|
6
|
-
import grab from "
|
|
5
|
+
import grab from "../utils/grab";
|
|
7
6
|
|
|
8
|
-
//
|
|
9
|
-
|
|
10
|
-
const mockGrab = grab as MockedFunction<typeof grab>;
|
|
7
|
+
// Mock grab-url
|
|
8
|
+
jest.mock("grab-url");
|
|
9
|
+
const mockGrab = grab as jest.MockedFunction<typeof grab>;
|
|
11
10
|
|
|
12
11
|
describe("scrapeURL", () => {
|
|
13
12
|
beforeEach(() => {
|
|
14
|
-
|
|
13
|
+
jest.clearAllMocks();
|
|
15
14
|
// Reset environment variables
|
|
16
15
|
delete process.env.SCRAPER_URL;
|
|
17
16
|
delete process.env.SCRAPER_API_KEY;
|
|
@@ -54,7 +53,7 @@ describe("scrapeURL", () => {
|
|
|
54
53
|
mockGrab.mockRejectedValueOnce(error);
|
|
55
54
|
|
|
56
55
|
// Mock Cloudflare scraper
|
|
57
|
-
global.fetch =
|
|
56
|
+
global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
|
|
58
57
|
|
|
59
58
|
// Mock JINA fallback
|
|
60
59
|
mockGrab.mockResolvedValueOnce("Title: Test\nMarkdown Content:\nFallback content");
|
|
@@ -70,12 +69,12 @@ describe("scrapeURL", () => {
|
|
|
70
69
|
error.status = 500;
|
|
71
70
|
mockGrab.mockRejectedValue(error);
|
|
72
71
|
|
|
73
|
-
global.fetch =
|
|
72
|
+
global.fetch = jest.fn().mockRejectedValue(new Error("Cloudflare failed"));
|
|
74
73
|
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
);
|
|
74
|
+
const result = await scrapeURL("https://failing-site.com");
|
|
75
|
+
|
|
76
|
+
expect(typeof result).toBe("object");
|
|
77
|
+
expect((result as any).error).toBeDefined();
|
|
79
78
|
});
|
|
80
79
|
|
|
81
80
|
it("should detect bot protection and retry", async () => {
|
|
@@ -87,7 +86,7 @@ describe("scrapeURL", () => {
|
|
|
87
86
|
mockGrab.mockResolvedValueOnce(botDetectionHtml);
|
|
88
87
|
|
|
89
88
|
// Mock successful Cloudflare bypass
|
|
90
|
-
global.fetch =
|
|
89
|
+
global.fetch = jest.fn().mockResolvedValueOnce({
|
|
91
90
|
ok: true,
|
|
92
91
|
json: async () => ({
|
|
93
92
|
html: successHtml,
|
|
@@ -111,7 +110,7 @@ describe("scrapeURL", () => {
|
|
|
111
110
|
mockGrab.mockResolvedValueOnce(botDetectionHtml);
|
|
112
111
|
|
|
113
112
|
// Cloudflare fails
|
|
114
|
-
global.fetch =
|
|
113
|
+
global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
|
|
115
114
|
|
|
116
115
|
// JINA succeeds
|
|
117
116
|
mockGrab.mockResolvedValueOnce(
|
|
@@ -127,26 +126,24 @@ describe("scrapeURL", () => {
|
|
|
127
126
|
});
|
|
128
127
|
|
|
129
128
|
it("should return error when bot detection persists", async () => {
|
|
130
|
-
|
|
131
|
-
// carries a trailing space, so it does not match this fixture; use one of
|
|
132
|
-
// the markers that does.
|
|
133
|
-
const botDetectionHtml =
|
|
134
|
-
"<html><body>Please verify you are a human</body></html>";
|
|
129
|
+
const botDetectionHtml = "<html><body>Cloudflare Ray ID found</body></html>";
|
|
135
130
|
|
|
136
131
|
// Initial request returns bot detection
|
|
137
132
|
mockGrab.mockResolvedValueOnce(botDetectionHtml);
|
|
138
133
|
|
|
139
134
|
// Cloudflare attempt also fails
|
|
140
|
-
global.fetch =
|
|
135
|
+
global.fetch = jest.fn().mockResolvedValueOnce({
|
|
141
136
|
ok: true,
|
|
142
137
|
json: async () => ({
|
|
143
138
|
html: botDetectionHtml, // Still bot detection
|
|
144
139
|
}),
|
|
145
140
|
} as any);
|
|
146
141
|
|
|
147
|
-
await
|
|
148
|
-
|
|
149
|
-
)
|
|
142
|
+
const result = await scrapeURL("https://protected-site.com", {
|
|
143
|
+
checkBotDetection: true,
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
expect((result as any).error).toBe("Bot detected");
|
|
150
147
|
});
|
|
151
148
|
|
|
152
149
|
it("should prepend proxy to URL if provided", async () => {
|
|
@@ -217,7 +214,7 @@ describe("scrapeURL", () => {
|
|
|
217
214
|
|
|
218
215
|
describe("scrapeJINA", () => {
|
|
219
216
|
beforeEach(() => {
|
|
220
|
-
|
|
217
|
+
jest.clearAllMocks();
|
|
221
218
|
});
|
|
222
219
|
|
|
223
220
|
it("should extract content from JINA API", async () => {
|
|
@@ -238,13 +235,10 @@ This is the article content in markdown format.
|
|
|
238
235
|
|
|
239
236
|
expect(mockGrab).toHaveBeenCalledWith(
|
|
240
237
|
"https://r.jina.ai/https://example.com/article",
|
|
241
|
-
|
|
238
|
+
{
|
|
242
239
|
responseType: "text",
|
|
243
240
|
timeout: 30,
|
|
244
|
-
|
|
245
|
-
// JINA_API_KEY is set.
|
|
246
|
-
headers: expect.objectContaining({ Accept: "text/html" }),
|
|
247
|
-
})
|
|
241
|
+
}
|
|
248
242
|
);
|
|
249
243
|
expect(result).toContain("<title>Test Article</title>");
|
|
250
244
|
expect(result).toContain("article content");
|
|
@@ -62,16 +62,7 @@ export async function scrapeURL(url, options = {}) {
|
|
|
62
62
|
|
|
63
63
|
if (checkRobotsAllowed) {
|
|
64
64
|
const rules = await fetchScrapingRules(url);
|
|
65
|
-
|
|
66
|
-
// request path — passing the full URL made every rule (including
|
|
67
|
-
// `Disallow: /`) silently fail to match.
|
|
68
|
-
let requestPath: string;
|
|
69
|
-
try {
|
|
70
|
-
requestPath = new URL(url).pathname;
|
|
71
|
-
} catch {
|
|
72
|
-
requestPath = url;
|
|
73
|
-
}
|
|
74
|
-
if (!isAllowedToScrape(rules, requestPath)) {
|
|
65
|
+
if (!isAllowedToScrape(rules, url)) {
|
|
75
66
|
return { error: "Robots.txt forbids to scrape there" };
|
|
76
67
|
}
|
|
77
68
|
}
|