@crawlee/puppeteer 3.0.0-alpha.2 → 3.0.0-alpha.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,149 @@
1
+ import { BrowserCrawler, BrowserCrawlerHandleRequest, BrowserCrawlerOptions, BrowserCrawlingContext, BrowserHook } from '@crawlee/browser';
2
+ import { PuppeteerPlugin } from '@crawlee/browser-pool';
3
+ import { HTTPResponse, LaunchOptions, Page } from 'puppeteer';
4
+ import { PuppeteerLaunchContext } from './puppeteer-launcher';
5
+ import { DirectNavigationOptions } from './utils/puppeteer_utils';
6
+ export declare type PuppeteerController = ReturnType<PuppeteerPlugin['_createController']>;
7
+ export declare type PuppeteerCrawlContext = BrowserCrawlingContext<Page, HTTPResponse, PuppeteerController>;
8
+ export declare type PuppeteerHook = BrowserHook<PuppeteerCrawlContext, PuppeteerGoToOptions>;
9
+ export declare type PuppeteerRequestHandlerParam = BrowserCrawlingContext<Page, HTTPResponse, PuppeteerController>;
10
+ export declare type PuppeteerRequestHandler = BrowserCrawlerHandleRequest<PuppeteerRequestHandlerParam>;
11
+ export declare type PuppeteerGoToOptions = Parameters<Page['goto']>[1];
12
+ export interface PuppeteerCrawlerOptions extends BrowserCrawlerOptions<PuppeteerCrawlContext, PuppeteerGoToOptions, {
13
+ browserPlugins: [PuppeteerPlugin];
14
+ }> {
15
+ /**
16
+ * Options used by {@link launchPuppeteer} to start new Puppeteer instances.
17
+ */
18
+ launchContext?: PuppeteerLaunchContext;
19
+ /**
20
+ * Async functions that are sequentially evaluated before the navigation. Good for setting additional cookies
21
+ * or browser properties before navigation. The function accepts two parameters, `crawlingContext` and `gotoOptions`,
22
+ * which are passed to the `page.goto()` function the crawler calls to navigate.
23
+ * Example:
24
+ * ```
25
+ * preNavigationHooks: [
26
+ * async (crawlingContext, gotoOptions) => {
27
+ * const { page } = crawlingContext;
28
+ * await page.evaluate((attr) => { window.foo = attr; }, 'bar');
29
+ * },
30
+ * ]
31
+ * ```
32
+ */
33
+ preNavigationHooks?: PuppeteerHook[];
34
+ /**
35
+ * Async functions that are sequentially evaluated after the navigation. Good for checking if the navigation was successful.
36
+ * The function accepts `crawlingContext` as the only parameter.
37
+ * Example:
38
+ * ```
39
+ * postNavigationHooks: [
40
+ * async (crawlingContext) => {
41
+ * const { page } = crawlingContext;
42
+ * if (hasCaptcha(page)) {
43
+ * await solveCaptcha (page);
44
+ * }
45
+ * },
46
+ * ]
47
+ * ```
48
+ */
49
+ postNavigationHooks?: PuppeteerHook[];
50
+ }
51
+ /**
52
+ * Provides a simple framework for parallel crawling of web pages
53
+ * using headless Chrome with [Puppeteer](https://github.com/puppeteer/puppeteer).
54
+ * The URLs to crawl are fed either from a static list of URLs
55
+ * or from a dynamic queue of URLs enabling recursive crawling of websites.
56
+ *
57
+ * Since `PuppeteerCrawler` uses headless Chrome to download web pages and extract data,
58
+ * it is useful for crawling of websites that require to execute JavaScript.
59
+ * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
60
+ * which downloads the pages using raw HTTP requests and is about 10x faster.
61
+ *
62
+ * The source URLs are represented using {@link Request} objects that are fed from
63
+ * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PuppeteerCrawlerOptions.requestList}
64
+ * or {@link PuppeteerCrawlerOptions.requestQueue} constructor options, respectively.
65
+ *
66
+ * If both {@link PuppeteerCrawlerOptions.requestList} and {@link PuppeteerCrawlerOptions.requestQueue} are used,
67
+ * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
68
+ * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
69
+ *
70
+ * The crawler finishes when there are no more {@link Request} objects to crawl.
71
+ *
72
+ * `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
73
+ * and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.handlePageFunction} option.
74
+ *
75
+ * New pages are only opened when there is enough free CPU and memory available,
76
+ * using the functionality provided by the {@link AutoscaledPool} class.
77
+ * All {@link AutoscaledPool} configuration options can be passed to the {@link PuppeteerCrawlerOptions.autoscaledPoolOptions}
78
+ * parameter of the `PuppeteerCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
79
+ * {@link AutoscaledPoolOptions} are available directly in the `PuppeteerCrawler` constructor.
80
+ *
81
+ * Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
82
+ *
83
+ * **Example usage:**
84
+ *
85
+ * ```javascript
86
+ * const crawler = new PuppeteerCrawler({
87
+ * requestList,
88
+ * handlePageFunction: async ({ page, request }) => {
89
+ * // This function is called to extract data from a single web page
90
+ * // 'page' is an instance of Puppeteer.Page with page.goto(request.url) already called
91
+ * // 'request' is an instance of Request class with information about the page to load
92
+ * await Actor.pushData({
93
+ * title: await page.title(),
94
+ * url: request.url,
95
+ * succeeded: true,
96
+ * })
97
+ * },
98
+ * handleFailedRequestFunction: async ({ request }) => {
99
+ * // This function is called when the crawling of a request failed too many times
100
+ * await Actor.pushData({
101
+ * url: request.url,
102
+ * succeeded: false,
103
+ * errors: request.errorMessages,
104
+ * })
105
+ * },
106
+ * });
107
+ *
108
+ * await crawler.run();
109
+ * ```
110
+ * @category Crawlers
111
+ */
112
+ export declare class PuppeteerCrawler extends BrowserCrawler<{
113
+ browserPlugins: [PuppeteerPlugin];
114
+ }, LaunchOptions, PuppeteerCrawlContext> {
115
+ protected static optionsShape: {
116
+ browserPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
117
+ handlePageFunction: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
118
+ gotoFunction: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
119
+ gotoTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
120
+ navigationTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
121
+ preNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
122
+ postNavigationHooks: import("ow").ArrayPredicate<unknown> & import("ow").BasePredicate<unknown[] | undefined>;
123
+ launchContext: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
124
+ sessionPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
125
+ persistCookiesPerSession: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
126
+ useSessionPool: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
127
+ proxyConfiguration: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
128
+ requestList: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
129
+ requestQueue: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
130
+ requestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
131
+ handleRequestFunction: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
132
+ requestHandlerTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
133
+ handleRequestTimeoutSecs: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
134
+ failedRequestHandler: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
135
+ handleFailedRequestFunction: import("ow").Predicate<Function> & import("ow").BasePredicate<Function | undefined>;
136
+ maxRequestRetries: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
137
+ maxRequestsPerCrawl: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
138
+ autoscaledPoolOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
139
+ minConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
140
+ maxConcurrency: import("ow").NumberPredicate & import("ow").BasePredicate<number | undefined>;
141
+ log: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
142
+ };
143
+ /**
144
+ * All `PuppeteerCrawler` parameters are passed via an options object.
145
+ */
146
+ constructor(options: PuppeteerCrawlerOptions);
147
+ protected _navigationHandler(crawlingContext: PuppeteerCrawlContext, gotoOptions: DirectNavigationOptions): Promise<any>;
148
+ }
149
+ //# sourceMappingURL=puppeteer-crawler.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"puppeteer-crawler.d.ts","sourceRoot":"","sources":["../../src/internals/puppeteer-crawler.ts"],"names":[],"mappings":"AAAA,OAAO,EACH,cAAc,EACd,2BAA2B,EAC3B,qBAAqB,EACrB,sBAAsB,EACtB,WAAW,EACd,MAAM,kBAAkB,CAAC;AAI1B,OAAO,EAAsB,eAAe,EAAE,MAAM,uBAAuB,CAAC;AAE5E,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,IAAI,EAAE,MAAM,WAAW,CAAC;AAC9D,OAAO,EAAE,sBAAsB,EAAqB,MAAM,sBAAsB,CAAC;AACjF,OAAO,EAAE,uBAAuB,EAAgB,MAAM,yBAAyB,CAAC;AAEhF,oBAAY,mBAAmB,GAAG,UAAU,CAAC,eAAe,CAAC,mBAAmB,CAAC,CAAC,CAAC;AAEnF,oBAAY,qBAAqB,GAAG,sBAAsB,CAAC,IAAI,EAAE,YAAY,EAAE,mBAAmB,CAAC,CAAA;AAEnG,oBAAY,aAAa,GAAG,WAAW,CAAC,qBAAqB,EAAE,oBAAoB,CAAC,CAAC;AAErF,oBAAY,4BAA4B,GAAG,sBAAsB,CAAC,IAAI,EAAE,YAAY,EAAE,mBAAmB,CAAC,CAAA;AAE1G,oBAAY,uBAAuB,GAAG,2BAA2B,CAAC,4BAA4B,CAAC,CAAC;AAEhG,oBAAY,oBAAoB,GAAG,UAAU,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;AAE/D,MAAM,WAAW,uBAAwB,SAAQ,qBAAqB,CAClE,qBAAqB,EACrB,oBAAoB,EACpB;IAAE,cAAc,EAAE,CAAC,eAAe,CAAC,CAAA;CAAE,CACxC;IACG;;OAEG;IACH,aAAa,CAAC,EAAE,sBAAsB,CAAC;IAEvC;;;;;;;;;;;;;OAaG;IACH,kBAAkB,CAAC,EAAE,aAAa,EAAE,CAAC;IAErC;;;;;;;;;;;;;;OAcG;IACH,mBAAmB,CAAC,EAAE,aAAa,EAAE,CAAC;CACzC;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4DG;AACH,qBAAa,gBAAiB,SAAQ,cAAc,CAAC;IAAE,cAAc,EAAE,CAAC,eAAe,CAAC,CAAA;CAAE,EAAE,aAAa,EAAE,qBAAqB,CAAC;IAC7H,iBAA0B,YAAY;;;;;;;;;;;;;;;;;;;;;;;;;;;MAGpC;IAEF;;OAEG;gBACS,OAAO,EAAE,uBAAuB;cAwBnB,kBAAkB,CAAC,eAAe,EAAE,qBAAqB,EAAE,WAAW,EAAE,uBAAuB;CAS3H"}
@@ -0,0 +1,106 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.PuppeteerCrawler = void 0;
4
+ const tslib_1 = require("tslib");
5
+ const browser_1 = require("@crawlee/browser");
6
+ const ow_1 = tslib_1.__importDefault(require("ow"));
7
+ const puppeteer_launcher_1 = require("./puppeteer-launcher");
8
+ const puppeteer_utils_1 = require("./utils/puppeteer_utils");
9
+ /**
10
+ * Provides a simple framework for parallel crawling of web pages
11
+ * using headless Chrome with [Puppeteer](https://github.com/puppeteer/puppeteer).
12
+ * The URLs to crawl are fed either from a static list of URLs
13
+ * or from a dynamic queue of URLs enabling recursive crawling of websites.
14
+ *
15
+ * Since `PuppeteerCrawler` uses headless Chrome to download web pages and extract data,
16
+ * it is useful for crawling of websites that require to execute JavaScript.
17
+ * If the target website doesn't need JavaScript, consider using {@link CheerioCrawler},
18
+ * which downloads the pages using raw HTTP requests and is about 10x faster.
19
+ *
20
+ * The source URLs are represented using {@link Request} objects that are fed from
21
+ * {@link RequestList} or {@link RequestQueue} instances provided by the {@link PuppeteerCrawlerOptions.requestList}
22
+ * or {@link PuppeteerCrawlerOptions.requestQueue} constructor options, respectively.
23
+ *
24
+ * If both {@link PuppeteerCrawlerOptions.requestList} and {@link PuppeteerCrawlerOptions.requestQueue} are used,
25
+ * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
26
+ * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
27
+ *
28
+ * The crawler finishes when there are no more {@link Request} objects to crawl.
29
+ *
30
+ * `PuppeteerCrawler` opens a new Chrome page (i.e. tab) for each {@link Request} object to crawl
31
+ * and then calls the function provided by user as the {@link PuppeteerCrawlerOptions.handlePageFunction} option.
32
+ *
33
+ * New pages are only opened when there is enough free CPU and memory available,
34
+ * using the functionality provided by the {@link AutoscaledPool} class.
35
+ * All {@link AutoscaledPool} configuration options can be passed to the {@link PuppeteerCrawlerOptions.autoscaledPoolOptions}
36
+ * parameter of the `PuppeteerCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
37
+ * {@link AutoscaledPoolOptions} are available directly in the `PuppeteerCrawler` constructor.
38
+ *
39
+ * Note that the pool of Puppeteer instances is internally managed by the [BrowserPool](https://github.com/apify/browser-pool) class.
40
+ *
41
+ * **Example usage:**
42
+ *
43
+ * ```javascript
44
+ * const crawler = new PuppeteerCrawler({
45
+ * requestList,
46
+ * handlePageFunction: async ({ page, request }) => {
47
+ * // This function is called to extract data from a single web page
48
+ * // 'page' is an instance of Puppeteer.Page with page.goto(request.url) already called
49
+ * // 'request' is an instance of Request class with information about the page to load
50
+ * await Actor.pushData({
51
+ * title: await page.title(),
52
+ * url: request.url,
53
+ * succeeded: true,
54
+ * })
55
+ * },
56
+ * handleFailedRequestFunction: async ({ request }) => {
57
+ * // This function is called when the crawling of a request failed too many times
58
+ * await Actor.pushData({
59
+ * url: request.url,
60
+ * succeeded: false,
61
+ * errors: request.errorMessages,
62
+ * })
63
+ * },
64
+ * });
65
+ *
66
+ * await crawler.run();
67
+ * ```
68
+ * @category Crawlers
69
+ */
70
+ class PuppeteerCrawler extends browser_1.BrowserCrawler {
71
+ /**
72
+ * All `PuppeteerCrawler` parameters are passed via an options object.
73
+ */
74
+ constructor(options) {
75
+ (0, ow_1.default)(options, 'PuppeteerCrawlerOptions', ow_1.default.object.exactShape(PuppeteerCrawler.optionsShape));
76
+ const { launchContext = {}, browserPoolOptions = {}, proxyConfiguration, ...browserCrawlerOptions } = options;
77
+ if (launchContext.proxyUrl) {
78
+ throw new Error('PuppeteerCrawlerOptions.launchContext.proxyUrl is not allowed in PuppeteerCrawler.'
79
+ + 'Use PuppeteerCrawlerOptions.proxyConfiguration');
80
+ }
81
+ const puppeteerLauncher = new puppeteer_launcher_1.PuppeteerLauncher(launchContext);
82
+ browserPoolOptions.browserPlugins = [
83
+ puppeteerLauncher.createBrowserPlugin(),
84
+ ];
85
+ super({ ...browserCrawlerOptions, launchContext, proxyConfiguration, browserPoolOptions });
86
+ }
87
+ async _navigationHandler(crawlingContext, gotoOptions) {
88
+ // TODO remove deprecated options in v3
89
+ if (this.gotoFunction) {
90
+ this.log.deprecated('PuppeteerCrawlerOptions.gotoFunction is deprecated. Use "preNavigationHooks" and "postNavigationHooks" instead.');
91
+ return this.gotoFunction(crawlingContext, gotoOptions);
92
+ }
93
+ return (0, puppeteer_utils_1.gotoExtended)(crawlingContext.page, crawlingContext.request, gotoOptions);
94
+ }
95
+ }
96
+ exports.PuppeteerCrawler = PuppeteerCrawler;
97
+ Object.defineProperty(PuppeteerCrawler, "optionsShape", {
98
+ enumerable: true,
99
+ configurable: true,
100
+ writable: true,
101
+ value: {
102
+ ...browser_1.BrowserCrawler.optionsShape,
103
+ browserPoolOptions: ow_1.default.optional.object,
104
+ }
105
+ });
106
+ //# sourceMappingURL=puppeteer-crawler.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"puppeteer-crawler.js","sourceRoot":"","sources":["../../src/internals/puppeteer-crawler.ts"],"names":[],"mappings":";;;;AAAA,8CAM0B;AAK1B,oDAAoB;AAEpB,6DAAiF;AACjF,6DAAgF;AA0DhF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4DG;AACH,MAAa,gBAAiB,SAAQ,wBAA2F;IAM7H;;OAEG;IACH,YAAY,OAAgC;QACxC,IAAA,YAAE,EAAC,OAAO,EAAE,yBAAyB,EAAE,YAAE,CAAC,MAAM,CAAC,UAAU,CAAC,gBAAgB,CAAC,YAAY,CAAC,CAAC,CAAC;QAE5F,MAAM,EACF,aAAa,GAAG,EAAE,EAClB,kBAAkB,GAAG,EAAwB,EAC7C,kBAAkB,EAClB,GAAG,qBAAqB,EAC3B,GAAG,OAAO,CAAC;QAEZ,IAAI,aAAa,CAAC,QAAQ,EAAE;YACxB,MAAM,IAAI,KAAK,CAAC,oFAAoF;kBAC9F,gDAAgD,CAAC,CAAC;SAC3D;QAED,MAAM,iBAAiB,GAAG,IAAI,sCAAiB,CAAC,aAAa,CAAC,CAAC;QAE/D,kBAAkB,CAAC,cAAc,GAAG;YAChC,iBAAiB,CAAC,mBAAmB,EAAE;SAC1C,CAAC;QAEF,KAAK,CAAC,EAAE,GAAG,qBAAqB,EAAE,aAAa,EAAE,kBAAkB,EAAE,kBAAkB,EAAE,CAAC,CAAC;IAC/F,CAAC;IAEkB,KAAK,CAAC,kBAAkB,CAAC,eAAsC,EAAE,WAAoC;QACpH,uCAAuC;QACvC,IAAI,IAAI,CAAC,YAAY,EAAE;YACnB,IAAI,CAAC,GAAG,CAAC,UAAU,CAAC,iHAAiH,CAAC,CAAC;YAEvI,OAAO,IAAI,CAAC,YAAY,CAAC,eAAe,EAAE,WAAyB,CAAC,CAAC;SACxE;QACD,OAAO,IAAA,8BAAY,EAAC,eAAe,CAAC,IAAI,EAAE,eAAe,CAAC,OAAO,EAAE,WAAW,CAAC,CAAC;IACpF,CAAC;;AAzCL,4CA0CC;AAzCG;;;;WAAyC;QACrC,GAAG,wBAAc,CAAC,YAAY;QAC9B,kBAAkB,EAAE,YAAE,CAAC,QAAQ,CAAC,MAAM;KACzC;GAAC"}
@@ -0,0 +1,115 @@
1
+ import { Browser } from 'puppeteer';
2
+ import { PuppeteerPlugin } from '@crawlee/browser-pool';
3
+ import { BrowserLaunchContext, BrowserLauncher } from '@crawlee/browser';
4
+ /**
5
+ * Apify extends the launch options of Puppeteer.
6
+ * You can use any of the Puppeteer compatible
7
+ * [`LaunchOptions`](https://pptr.dev/#?product=Puppeteer&show=api-puppeteerlaunchoptions)
8
+ * options by providing the `launchOptions` property.
9
+ *
10
+ * **Example:**
11
+ * ```js
12
+ * // launch a headless Chrome (not Chromium)
13
+ * const launchContext = {
14
+ * // Apify helpers
15
+ * useChrome: true,
16
+ * proxyUrl: 'http://user:password@some.proxy.com'
17
+ * // Native Puppeteer options
18
+ * launchOptions: {
19
+ * headless: true,
20
+ * args: ['--some-flag'],
21
+ * }
22
+ * }
23
+ * ```
24
+ */
25
+ export interface PuppeteerLaunchContext extends BrowserLaunchContext<PuppeteerPlugin['launchOptions'], unknown> {
26
+ /**
27
+ * `puppeteer.launch` [options](https://pptr.dev/#?product=Puppeteer&version=v13.5.1&show=api-puppeteerlaunchoptions)
28
+ */
29
+ launchOptions?: PuppeteerPlugin['launchOptions'];
30
+ /**
31
+ * URL to a HTTP proxy server. It must define the port number,
32
+ * and it may also contain proxy username and password.
33
+ *
34
+ * Example: `http://bob:pass123@proxy.example.com:1234`.
35
+ */
36
+ proxyUrl?: string;
37
+ /**
38
+ * If `true` and `executablePath` is not set,
39
+ * Puppeteer will launch full Google Chrome browser available on the machine
40
+ * rather than the bundled Chromium. The path to Chrome executable
41
+ * is taken from the `APIFY_CHROME_EXECUTABLE_PATH` environment variable if provided,
42
+ * or defaults to the typical Google Chrome executable location specific for the operating system.
43
+ * By default, this option is `false`.
44
+ * @default false
45
+ */
46
+ useChrome?: boolean;
47
+ /**
48
+ * Already required module (`Object`). This enables usage of various Puppeteer
49
+ * wrappers such as `puppeteer-extra`.
50
+ *
51
+ * Take caution, because it can cause all kinds of unexpected errors and weird behavior.
52
+ * Apify SDK is not tested with any other library besides `puppeteer` itself.
53
+ */
54
+ launcher?: unknown;
55
+ /**
56
+ * With this option selected, all pages will be opened in a new incognito browser context.
57
+ * This means they will not share cookies nor cache and their resources will not be throttled by one another.
58
+ * @default false
59
+ */
60
+ useIncognitoPages?: boolean;
61
+ }
62
+ /**
63
+ * `PuppeteerLauncher` is based on the `BrowserLauncher`. It launches `puppeteer` browser instance.
64
+ * @ignore
65
+ */
66
+ export declare class PuppeteerLauncher extends BrowserLauncher<PuppeteerPlugin, unknown> {
67
+ protected static optionsShape: {
68
+ launcher: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
69
+ proxyUrl: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
70
+ useChrome: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
71
+ useIncognitoPages: import("ow").BooleanPredicate & import("ow").BasePredicate<boolean | undefined>;
72
+ userDataDir: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
73
+ launchOptions: import("ow").ObjectPredicate<object> & import("ow").BasePredicate<object | undefined>;
74
+ userAgent: import("ow").StringPredicate & import("ow").BasePredicate<string | undefined>;
75
+ };
76
+ /**
77
+ * All `PuppeteerLauncher` parameters are passed via an launchContext object.
78
+ */
79
+ constructor(launchContext?: PuppeteerLaunchContext);
80
+ }
81
+ /**
82
+ * Launches headless Chrome using Puppeteer pre-configured to work within the Apify platform.
83
+ * The function has the same argument and the return value as `puppeteer.launch()`.
84
+ * See [Puppeteer documentation](https://github.com/puppeteer/puppeteer/blob/master/docs/api.md#puppeteerlaunchoptions) for more details.
85
+ *
86
+ * The `launchPuppeteer()` function alters the following Puppeteer options:
87
+ *
88
+ * - Passes the setting from the `APIFY_HEADLESS` environment variable to the `headless` option,
89
+ * unless it was already defined by the caller or `APIFY_XVFB` environment variable is set to `1`.
90
+ * Note that Apify Actor cloud platform automatically sets `APIFY_HEADLESS=1` to all running actors.
91
+ * - Takes the `proxyUrl` option, validates it and adds it to `args` as `--proxy-server=XXX`.
92
+ * The proxy URL must define a port number and have one of the following schemes: `http://`,
93
+ * `https://`, `socks4://` or `socks5://`.
94
+ * If the proxy is HTTP (i.e. has the `http://` scheme) and contains username or password,
95
+ * the `launchPuppeteer` functions sets up an anonymous proxy HTTP
96
+ * to make the proxy work with headless Chrome. For more information, read the
97
+ * [blog post about proxy-chain library](https://blog.apify.com/how-to-make-headless-chrome-and-puppeteer-use-a-proxy-server-with-authentication-249a21a79212).
98
+ *
99
+ * To use this function, you need to have the [puppeteer](https://www.npmjs.com/package/puppeteer)
100
+ * NPM package installed in your project.
101
+ * When running on the Apify cloud, you can achieve that simply
102
+ * by using the `apify/actor-node-chrome` base Docker image for your actor - see
103
+ * [Apify Actor documentation](https://docs.apify.com/actor/build#base-images)
104
+ * for details.
105
+ *
106
+ * For an example of usage, see the [Puppeteer proxy Example](/docs/examples/puppeteer-with-proxy).
107
+ *
108
+ * @param [launchContext]
109
+ * All `PuppeteerLauncher` parameters are passed via an launchContext object.
110
+ * If you want to pass custom `puppeteer.launch(options)` options you can use the `PuppeteerLaunchContext.launchOptions` property.
111
+ * @returns
112
+ * Promise that resolves to Puppeteer's `Browser` instance.
113
+ */
114
+ export declare function launchPuppeteer(launchContext?: PuppeteerLaunchContext): Promise<Browser>;
115
+ //# sourceMappingURL=puppeteer-launcher.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"puppeteer-launcher.d.ts","sourceRoot":"","sources":["../../src/internals/puppeteer-launcher.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AACpC,OAAO,EAAE,eAAe,EAAE,MAAM,uBAAuB,CAAC;AACxD,OAAO,EAAE,oBAAoB,EAAE,eAAe,EAAE,MAAM,kBAAkB,CAAC;AAEzE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,WAAW,sBAAuB,SAAQ,oBAAoB,CAAC,eAAe,CAAC,eAAe,CAAC,EAAE,OAAO,CAAC;IAC3G;;OAEG;IACH,aAAa,CAAC,EAAE,eAAe,CAAC,eAAe,CAAC,CAAC;IAEjD;;;;;OAKG;IACH,QAAQ,CAAC,EAAE,MAAM,CAAC;IAElB;;;;;;;;OAQG;IACH,SAAS,CAAC,EAAE,OAAO,CAAC;IAEpB;;;;;;OAMG;IACH,QAAQ,CAAC,EAAE,OAAO,CAAC;IAEnB;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,OAAO,CAAC;CAC/B;AAED;;;GAGG;AACH,qBAAa,iBAAkB,SAAQ,eAAe,CAAC,eAAe,EAAE,OAAO,CAAC;IAC5E,iBAA0B,YAAY;;;;;;;;MAGpC;IAEF;;OAEG;gBACS,aAAa,GAAE,sBAA2B;CAezD;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAgCG;AACH,wBAAsB,eAAe,CAAC,aAAa,CAAC,EAAE,sBAAsB,GAAG,OAAO,CAAC,OAAO,CAAC,CAI9F"}
@@ -0,0 +1,74 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.launchPuppeteer = exports.PuppeteerLauncher = void 0;
4
+ const tslib_1 = require("tslib");
5
+ const ow_1 = tslib_1.__importDefault(require("ow"));
6
+ const browser_pool_1 = require("@crawlee/browser-pool");
7
+ const browser_1 = require("@crawlee/browser");
8
+ /**
9
+ * `PuppeteerLauncher` is based on the `BrowserLauncher`. It launches `puppeteer` browser instance.
10
+ * @ignore
11
+ */
12
+ class PuppeteerLauncher extends browser_1.BrowserLauncher {
13
+ /**
14
+ * All `PuppeteerLauncher` parameters are passed via an launchContext object.
15
+ */
16
+ constructor(launchContext = {}) {
17
+ (0, ow_1.default)(launchContext, 'PuppeteerLauncher', ow_1.default.object.exactShape(PuppeteerLauncher.optionsShape));
18
+ const { launcher = browser_1.BrowserLauncher.requireLauncherOrThrow('puppeteer', 'apify/actor-node-puppeteer-chrome'), ...browserLauncherOptions } = launchContext;
19
+ super({
20
+ ...browserLauncherOptions,
21
+ launcher,
22
+ });
23
+ this.Plugin = browser_pool_1.PuppeteerPlugin;
24
+ }
25
+ }
26
+ exports.PuppeteerLauncher = PuppeteerLauncher;
27
+ Object.defineProperty(PuppeteerLauncher, "optionsShape", {
28
+ enumerable: true,
29
+ configurable: true,
30
+ writable: true,
31
+ value: {
32
+ ...browser_1.BrowserLauncher.optionsShape,
33
+ launcher: ow_1.default.optional.object,
34
+ }
35
+ });
36
+ /**
37
+ * Launches headless Chrome using Puppeteer pre-configured to work within the Apify platform.
38
+ * The function has the same argument and the return value as `puppeteer.launch()`.
39
+ * See [Puppeteer documentation](https://github.com/puppeteer/puppeteer/blob/master/docs/api.md#puppeteerlaunchoptions) for more details.
40
+ *
41
+ * The `launchPuppeteer()` function alters the following Puppeteer options:
42
+ *
43
+ * - Passes the setting from the `APIFY_HEADLESS` environment variable to the `headless` option,
44
+ * unless it was already defined by the caller or `APIFY_XVFB` environment variable is set to `1`.
45
+ * Note that Apify Actor cloud platform automatically sets `APIFY_HEADLESS=1` to all running actors.
46
+ * - Takes the `proxyUrl` option, validates it and adds it to `args` as `--proxy-server=XXX`.
47
+ * The proxy URL must define a port number and have one of the following schemes: `http://`,
48
+ * `https://`, `socks4://` or `socks5://`.
49
+ * If the proxy is HTTP (i.e. has the `http://` scheme) and contains username or password,
50
+ * the `launchPuppeteer` functions sets up an anonymous proxy HTTP
51
+ * to make the proxy work with headless Chrome. For more information, read the
52
+ * [blog post about proxy-chain library](https://blog.apify.com/how-to-make-headless-chrome-and-puppeteer-use-a-proxy-server-with-authentication-249a21a79212).
53
+ *
54
+ * To use this function, you need to have the [puppeteer](https://www.npmjs.com/package/puppeteer)
55
+ * NPM package installed in your project.
56
+ * When running on the Apify cloud, you can achieve that simply
57
+ * by using the `apify/actor-node-chrome` base Docker image for your actor - see
58
+ * [Apify Actor documentation](https://docs.apify.com/actor/build#base-images)
59
+ * for details.
60
+ *
61
+ * For an example of usage, see the [Puppeteer proxy Example](/docs/examples/puppeteer-with-proxy).
62
+ *
63
+ * @param [launchContext]
64
+ * All `PuppeteerLauncher` parameters are passed via an launchContext object.
65
+ * If you want to pass custom `puppeteer.launch(options)` options you can use the `PuppeteerLaunchContext.launchOptions` property.
66
+ * @returns
67
+ * Promise that resolves to Puppeteer's `Browser` instance.
68
+ */
69
+ async function launchPuppeteer(launchContext) {
70
+ const puppeteerLauncher = new PuppeteerLauncher(launchContext);
71
+ return puppeteerLauncher.launch();
72
+ }
73
+ exports.launchPuppeteer = launchPuppeteer;
74
+ //# sourceMappingURL=puppeteer-launcher.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"puppeteer-launcher.js","sourceRoot":"","sources":["../../src/internals/puppeteer-launcher.ts"],"names":[],"mappings":";;;;AAAA,oDAAoB;AAEpB,wDAAwD;AACxD,8CAAyE;AAiEzE;;;GAGG;AACH,MAAa,iBAAkB,SAAQ,yBAAyC;IAM5E;;OAEG;IACH,YAAY,gBAAwC,EAAE;QAClD,IAAA,YAAE,EAAC,aAAa,EAAE,mBAAmB,EAAE,YAAE,CAAC,MAAM,CAAC,UAAU,CAAC,iBAAiB,CAAC,YAAY,CAAC,CAAC,CAAC;QAE7F,MAAM,EACF,QAAQ,GAAG,yBAAe,CAAC,sBAAsB,CAAC,WAAW,EAAE,mCAAmC,CAAC,EACnG,GAAG,sBAAsB,EAC5B,GAAG,aAAa,CAAC;QAElB,KAAK,CAAC;YACF,GAAG,sBAAsB;YACzB,QAAQ;SACX,CAAC,CAAC;QAEH,IAAI,CAAC,MAAM,GAAG,8BAAe,CAAC;IAClC,CAAC;;AAvBL,8CAwBC;AAvBG;;;;WAAyC;QACrC,GAAG,yBAAe,CAAC,YAAY;QAC/B,QAAQ,EAAE,YAAE,CAAC,QAAQ,CAAC,MAAM;KAC/B;GAAC;AAsBN;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAgCG;AACI,KAAK,UAAU,eAAe,CAAC,aAAsC;IACxE,MAAM,iBAAiB,GAAG,IAAI,iBAAiB,CAAC,aAAa,CAAC,CAAC;IAE/D,OAAO,iBAAiB,CAAC,MAAM,EAAE,CAAC;AACtC,CAAC;AAJD,0CAIC"}
@@ -0,0 +1,60 @@
1
+ import { HTTPRequest as PuppeteerRequest, Page } from 'puppeteer';
2
+ export declare type InterceptHandler = (request: PuppeteerRequest) => unknown;
3
+ /**
4
+ * Adds request interception handler in similar to `page.on('request', handler);` but in addition to that
5
+ * supports multiple parallel handlers.
6
+ *
7
+ * All the handlers are executed sequentially in the order as they were added.
8
+ * Each of the handlers must call one of `request.continue()`, `request.abort()` and `request.respond()`.
9
+ * In addition to that any of the handlers may modify the request object (method, postData, headers)
10
+ * by passing its overrides to `request.continue()`.
11
+ * If multiple handlers modify same property then the last one wins. Headers are merged separately so you can
12
+ * override only a value of specific header.
13
+ *
14
+ * If one the handlers calls `request.abort()` or `request.respond()` then request is not propagated further
15
+ * to any of the remaining handlers.
16
+ *
17
+ *
18
+ * **Example usage:**
19
+ *
20
+ * ```javascript
21
+ * // Replace images with placeholder.
22
+ * await addInterceptRequestHandler(page, (request) => {
23
+ * if (request.resourceType() === 'image') {
24
+ * return request.respond({
25
+ * statusCode: 200,
26
+ * contentType: 'image/jpeg',
27
+ * body: placeholderImageBuffer,
28
+ * });
29
+ * }
30
+ * return request.continue();
31
+ * });
32
+ *
33
+ * // Abort all the scripts.
34
+ * await addInterceptRequestHandler(page, (request) => {
35
+ * if (request.resourceType() === 'script') return request.abort();
36
+ * return request.continue();
37
+ * });
38
+ *
39
+ * // Change requests to post.
40
+ * await addInterceptRequestHandler(page, (request) => {
41
+ * return request.continue({
42
+ * method: 'POST',
43
+ * });
44
+ * });
45
+ *
46
+ * await page.goto('http://example.com');
47
+ * ```
48
+ *
49
+ * @param page Puppeteer [`Page`](https://pptr.dev/#?product=Puppeteer&show=api-class-page) object.
50
+ * @param handler Request interception handler.
51
+ */
52
+ export declare function addInterceptRequestHandler(page: Page, handler: InterceptHandler): Promise<void>;
53
+ /**
54
+ * Removes request interception handler for given page.
55
+ *
56
+ * @param page Puppeteer [`Page`](https://pptr.dev/#?product=Puppeteer&show=api-class-page) object.
57
+ * @param handler Request interception handler.
58
+ */
59
+ export declare function removeInterceptRequestHandler(page: Page, handler: InterceptHandler): Promise<void>;
60
+ //# sourceMappingURL=puppeteer_request_interception.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"puppeteer_request_interception.d.ts","sourceRoot":"","sources":["../../../src/internals/utils/puppeteer_request_interception.ts"],"names":[],"mappings":"AAEA,OAAO,EAAe,WAAW,IAAI,gBAAgB,EAAE,IAAI,EAAE,MAAM,WAAW,CAAC;AAiC/E,oBAAY,gBAAgB,GAAG,CAAC,OAAO,EAAE,gBAAgB,KAAK,OAAO,CAAC;AAwEtE;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAgDG;AACH,wBAAsB,0BAA0B,CAAC,IAAI,EAAE,IAAI,EAAE,OAAO,EAAE,gBAAgB,GAAG,OAAO,CAAC,IAAI,CAAC,CAmCrG;AAED;;;;;GAKG;AACH,wBAAsB,6BAA6B,CAAC,IAAI,EAAE,IAAI,EAAE,OAAO,EAAE,gBAAgB,GAAG,OAAO,CAAC,IAAI,CAAC,CA+BxG"}