extract-webpage 1.2.187 → 1.2.188

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.187",
3
+ "version": "1.2.188",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -74,10 +74,10 @@
74
74
  "dependencies": {
75
75
  "@huggingface/transformers": "^3.8.1",
76
76
  "ai": "^5.0.0",
77
- "chat-agent-toolkit": "^1.2.187",
77
+ "chat-agent-toolkit": "^1.2.188",
78
78
  "chrono-node": "^2.9.0",
79
79
  "drizzle-orm": "^0.45.1",
80
- "extract-pdf": "^0.1.174",
80
+ "extract-pdf": "^0.1.175",
81
81
  "extract-youtube": "^1.0.103",
82
82
  "html-entities": "^2.6.0",
83
83
  "js-yaml": "^4.1.1",
@@ -1,50 +1,45 @@
1
1
  /**
2
2
  * @fileoverview Unit tests for URL content extraction
3
3
  */
4
- import { beforeEach, describe, expect, it, vi, type MockedFunction } from "vitest";
5
4
  import { extractContent } from "../url-to-content";
6
5
  import { scrapeURL } from "../url-to-html";
7
6
  import { extractContentAndCite } from "../../html-to-content/html-to-content";
8
- import { convertYoutubeToText, getURLYoutubeVideo } from "../youtube-helpers";
9
- import grab from "../../utils/grab";
7
+ import { convertYoutubeToText } from "../youtube-helpers";
10
8
  import { convertPDFToHTML } from "extract-pdf";
11
9
 
12
10
  // Mock dependencies
13
- vi.mock("../url-to-html", () => ({
14
- scrapeURL: vi.fn(),
11
+ jest.mock("../url-to-html", () => ({
12
+ scrapeURL: jest.fn(),
15
13
  }));
16
14
 
17
- vi.mock("../../html-to-content/html-to-content", () => ({
18
- extractContentAndCite: vi.fn(),
15
+ jest.mock("../../html-to-content/html-to-content", () => ({
16
+ extractContentAndCite: jest.fn(),
19
17
  }));
20
18
 
21
- vi.mock("../youtube-helpers", () => ({
22
- getURLYoutubeVideo: vi.fn(),
23
- convertYoutubeToText: vi.fn(),
19
+ jest.mock("../youtube-helpers", () => ({
20
+ getURLYoutubeVideo: jest.fn(),
21
+ convertYoutubeToText: jest.fn(),
24
22
  }));
25
23
 
26
- vi.mock("extract-pdf", () => ({
27
- convertPDFToHTML: vi.fn(),
24
+ jest.mock("extract-pdf", () => ({
25
+ convertPDFToHTML: jest.fn(),
28
26
  }));
29
27
 
30
- vi.mock("../../utils/grab");
28
+ jest.mock("grab-url", () => jest.fn());
31
29
 
32
- const mockScrapeURL = scrapeURL as MockedFunction<typeof scrapeURL>;
33
- const mockExtractContentAndCite = extractContentAndCite as MockedFunction<
30
+ const mockScrapeURL = scrapeURL as jest.MockedFunction<typeof scrapeURL>;
31
+ const mockExtractContentAndCite = extractContentAndCite as jest.MockedFunction<
34
32
  typeof extractContentAndCite
35
33
  >;
36
34
  const mockConvertYoutubeToText =
37
- convertYoutubeToText as MockedFunction<typeof convertYoutubeToText>;
38
- const mockConvertPDFToHTML = convertPDFToHTML as MockedFunction<
35
+ convertYoutubeToText as jest.MockedFunction<typeof convertYoutubeToText>;
36
+ const mockConvertPDFToHTML = convertPDFToHTML as jest.MockedFunction<
39
37
  typeof convertPDFToHTML
40
38
  >;
41
- const mockGetURLYoutubeVideo =
42
- getURLYoutubeVideo as MockedFunction<typeof getURLYoutubeVideo>;
43
- const mockGrab = grab as MockedFunction<typeof grab>;
44
39
 
45
40
  describe("extractContent", () => {
46
41
  beforeEach(() => {
47
- vi.clearAllMocks();
42
+ jest.clearAllMocks();
48
43
  });
49
44
 
50
45
  describe("URL extraction", () => {
@@ -67,10 +62,17 @@ describe("extractContent", () => {
67
62
  expect(mockScrapeURL).toHaveBeenCalledWith("https://example.com/article", {
68
63
  proxy: null,
69
64
  });
70
- expect(mockExtractContentAndCite).toHaveBeenCalledWith(
71
- mockHtml,
72
- expect.objectContaining({ url: "https://example.com/article" })
73
- );
65
+ expect(mockExtractContentAndCite).toHaveBeenCalledWith(mockHtml, {
66
+ url: "https://example.com/article",
67
+ images: true,
68
+ links: true,
69
+ formatting: true,
70
+ absoluteURLs: true,
71
+ timeout: 10,
72
+ proxy: null,
73
+ citeFormatMonthFull: false,
74
+ citeFormatAuthorFull: true,
75
+ });
74
76
  expect(result.title).toBe("Test Article");
75
77
  expect(result.html).toBe("<p>Test content</p>");
76
78
  });
@@ -82,8 +84,7 @@ describe("extractContent", () => {
82
84
 
83
85
  const result = await extractContent("https://blocked-site.com/article");
84
86
 
85
- // Any non-string scrapeURL result collapses to one generic message.
86
- expect(result.error).toBe("Failed to fetch HTML content");
87
+ expect(result.error).toBe("HTTP error: 403 Forbidden");
87
88
  expect(mockExtractContentAndCite).not.toHaveBeenCalled();
88
89
  });
89
90
 
@@ -119,10 +120,9 @@ describe("extractContent", () => {
119
120
  it("should handle scrapeURL throwing an error", async () => {
120
121
  mockScrapeURL.mockRejectedValueOnce(new Error("Network timeout"));
121
122
 
122
- const result = await extractContent("https://timeout.com/article");
123
-
124
- // extractContent catches scrape failures and reports them in-band.
125
- expect(result.error).toBe("Failed to scrape URL: Network timeout");
123
+ await expect(
124
+ extractContent("https://timeout.com/article")
125
+ ).rejects.toThrow("Network timeout");
126
126
  });
127
127
 
128
128
  it("should handle extraction returning no HTML", async () => {
@@ -192,7 +192,8 @@ describe("extractContent", () => {
192
192
 
193
193
  describe("YouTube extraction", () => {
194
194
  it("should extract YouTube video transcript", async () => {
195
- mockGetURLYoutubeVideo.mockReturnValueOnce("dQw4w9WgXcQ");
195
+ const youtubeHelpers = require("../youtube-helpers");
196
+ youtubeHelpers.getURLYoutubeVideo.mockReturnValueOnce("dQw4w9WgXcQ");
196
197
 
197
198
  const mockTranscript = {
198
199
  title: "Test Video",
@@ -268,10 +269,6 @@ describe("extractContent", () => {
268
269
  };
269
270
 
270
271
  mockConvertPDFToHTML.mockResolvedValueOnce(mockPdfContent as any);
271
- // isUrlPDF sniffs the leading "%PDF-" magic bytes via grab().
272
- mockGrab.mockResolvedValueOnce(
273
- new Uint8Array([0x25, 0x50, 0x44, 0x46, 0x2d]).buffer as any
274
- );
275
272
 
276
273
  await extractContent(
277
274
  "https://drive.google.com/file/d/ABC123/view"
@@ -1,17 +1,16 @@
1
1
  /**
2
2
  * @fileoverview Unit tests for URL scraping functionality
3
3
  */
4
- import { beforeEach, describe, expect, it, vi, type MockedFunction } from "vitest";
5
4
  import { scrapeURL, scrapeJINA } from "../url-to-html";
6
- import grab from "../../utils/grab";
5
+ import grab from "../utils/grab";
7
6
 
8
- // The module under test imports the local grab wrapper, not grab-url.
9
- vi.mock("../../utils/grab");
10
- const mockGrab = grab as MockedFunction<typeof grab>;
7
+ // Mock grab-url
8
+ jest.mock("grab-url");
9
+ const mockGrab = grab as jest.MockedFunction<typeof grab>;
11
10
 
12
11
  describe("scrapeURL", () => {
13
12
  beforeEach(() => {
14
- vi.clearAllMocks();
13
+ jest.clearAllMocks();
15
14
  // Reset environment variables
16
15
  delete process.env.SCRAPER_URL;
17
16
  delete process.env.SCRAPER_API_KEY;
@@ -54,7 +53,7 @@ describe("scrapeURL", () => {
54
53
  mockGrab.mockRejectedValueOnce(error);
55
54
 
56
55
  // Mock Cloudflare scraper
57
- global.fetch = vi.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
56
+ global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
58
57
 
59
58
  // Mock JINA fallback
60
59
  mockGrab.mockResolvedValueOnce("Title: Test\nMarkdown Content:\nFallback content");
@@ -70,12 +69,12 @@ describe("scrapeURL", () => {
70
69
  error.status = 500;
71
70
  mockGrab.mockRejectedValue(error);
72
71
 
73
- global.fetch = vi.fn().mockRejectedValue(new Error("Cloudflare failed"));
72
+ global.fetch = jest.fn().mockRejectedValue(new Error("Cloudflare failed"));
74
73
 
75
- // Exhausting every strategy throws rather than returning an error object.
76
- await expect(scrapeURL("https://failing-site.com")).rejects.toThrow(
77
- /All scraping methods failed/
78
- );
74
+ const result = await scrapeURL("https://failing-site.com");
75
+
76
+ expect(typeof result).toBe("object");
77
+ expect((result as any).error).toBeDefined();
79
78
  });
80
79
 
81
80
  it("should detect bot protection and retry", async () => {
@@ -87,7 +86,7 @@ describe("scrapeURL", () => {
87
86
  mockGrab.mockResolvedValueOnce(botDetectionHtml);
88
87
 
89
88
  // Mock successful Cloudflare bypass
90
- global.fetch = vi.fn().mockResolvedValueOnce({
89
+ global.fetch = jest.fn().mockResolvedValueOnce({
91
90
  ok: true,
92
91
  json: async () => ({
93
92
  html: successHtml,
@@ -111,7 +110,7 @@ describe("scrapeURL", () => {
111
110
  mockGrab.mockResolvedValueOnce(botDetectionHtml);
112
111
 
113
112
  // Cloudflare fails
114
- global.fetch = vi.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
113
+ global.fetch = jest.fn().mockRejectedValueOnce(new Error("Cloudflare failed"));
115
114
 
116
115
  // JINA succeeds
117
116
  mockGrab.mockResolvedValueOnce(
@@ -127,26 +126,24 @@ describe("scrapeURL", () => {
127
126
  });
128
127
 
129
128
  it("should return error when bot detection persists", async () => {
130
- // Note: the "Cloudflare Ray ID found " marker in checkHTMLForBotDetection
131
- // carries a trailing space, so it does not match this fixture; use one of
132
- // the markers that does.
133
- const botDetectionHtml =
134
- "<html><body>Please verify you are a human</body></html>";
129
+ const botDetectionHtml = "<html><body>Cloudflare Ray ID found</body></html>";
135
130
 
136
131
  // Initial request returns bot detection
137
132
  mockGrab.mockResolvedValueOnce(botDetectionHtml);
138
133
 
139
134
  // Cloudflare attempt also fails
140
- global.fetch = vi.fn().mockResolvedValueOnce({
135
+ global.fetch = jest.fn().mockResolvedValueOnce({
141
136
  ok: true,
142
137
  json: async () => ({
143
138
  html: botDetectionHtml, // Still bot detection
144
139
  }),
145
140
  } as any);
146
141
 
147
- await expect(
148
- scrapeURL("https://protected-site.com", { checkBotDetection: true })
149
- ).rejects.toThrow(/Bot detected/);
142
+ const result = await scrapeURL("https://protected-site.com", {
143
+ checkBotDetection: true,
144
+ });
145
+
146
+ expect((result as any).error).toBe("Bot detected");
150
147
  });
151
148
 
152
149
  it("should prepend proxy to URL if provided", async () => {
@@ -217,7 +214,7 @@ describe("scrapeURL", () => {
217
214
 
218
215
  describe("scrapeJINA", () => {
219
216
  beforeEach(() => {
220
- vi.clearAllMocks();
217
+ jest.clearAllMocks();
221
218
  });
222
219
 
223
220
  it("should extract content from JINA API", async () => {
@@ -238,13 +235,10 @@ This is the article content in markdown format.
238
235
 
239
236
  expect(mockGrab).toHaveBeenCalledWith(
240
237
  "https://r.jina.ai/https://example.com/article",
241
- expect.objectContaining({
238
+ {
242
239
  responseType: "text",
243
240
  timeout: 30,
244
- // JINA is asked for HTML, and gets an Authorization header when
245
- // JINA_API_KEY is set.
246
- headers: expect.objectContaining({ Accept: "text/html" }),
247
- })
241
+ }
248
242
  );
249
243
  expect(result).toContain("<title>Test Article</title>");
250
244
  expect(result).toContain("article content");
@@ -62,16 +62,7 @@ export async function scrapeURL(url, options = {}) {
62
62
 
63
63
  if (checkRobotsAllowed) {
64
64
  const rules = await fetchScrapingRules(url);
65
- // isAllowedToScrape matches rule paths with startsWith, so it needs the
66
- // request path — passing the full URL made every rule (including
67
- // `Disallow: /`) silently fail to match.
68
- let requestPath: string;
69
- try {
70
- requestPath = new URL(url).pathname;
71
- } catch {
72
- requestPath = url;
73
- }
74
- if (!isAllowedToScrape(rules, requestPath)) {
65
+ if (!isAllowedToScrape(rules, url)) {
75
66
  return { error: "Robots.txt forbids to scrape there" };
76
67
  }
77
68
  }