@crawlee/playwright 4.0.0-beta.165 → 4.0.0-beta.167

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -217,6 +217,7 @@ export declare class AdaptivePlaywrightCrawler<ContextExtension = Dictionary<nev
217
217
  * that are still running are not waited for, unlike {@link BasicCrawler.stop|`stop()`}.
218
218
  */
219
219
  teardown(): Promise<void>;
220
+ destroy(): Promise<void>;
220
221
  }
221
222
  export declare function createAdaptivePlaywrightRouter<Context extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
222
223
  export declare function createAdaptivePlaywrightRouter<Context extends AdaptivePlaywrightCrawlerContext = AdaptivePlaywrightCrawlerContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
@@ -77,6 +77,10 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
77
77
  * a discarded attempt's enqueues must never reach the queue.
78
78
  */
79
79
  #attemptWritePolicy;
80
+ /** Owns the browser pool this crawler's runs use, so its per-run resources are released with ours. */
81
+ #browserCrawler;
82
+ /** Nothing of its state is per-run, but it owns a session pool that outlives one. */
83
+ #staticCrawler;
80
84
  /**
81
85
  * In-flight rendering type detections, plus the pending results of an asynchronous `storeResult`.
82
86
  */
@@ -85,7 +89,6 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
85
89
  * Set once `teardown()` starts, so that requests still in the pool stop opening new detections.
86
90
  */
87
91
  #shutDown = false;
88
- #teardownHooks = [];
89
92
  constructor(options = {}) {
90
93
  const { requestHandler, renderingTypeDetectionRatio = 0.1, renderingTypePredictor, resultChecker, shouldPropagateError, resultComparator, statistics, requestHandlerTimeoutSecs = 60, errorHandler, failedRequestHandler, preNavigationHooks = [], postNavigationHooks = [], extendContext, contextPipelineBuilder, transactionalStorage, launchContext, headless, browserPool, remoteBrowser, ...rest } = options;
91
94
  // The user's value is replaced by `false` in the `super` call below — validate it separately,
@@ -180,7 +183,8 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
180
183
  browserPool,
181
184
  remoteBrowser,
182
185
  });
183
- this.#teardownHooks.push(browserCrawler.teardown.bind(browserCrawler));
186
+ this.#staticCrawler = staticCrawler;
187
+ this.#browserCrawler = browserCrawler;
184
188
  this.#staticContextPipeline = staticCrawler.contextPipeline.compose({
185
189
  action: this.adaptCheerioContext.bind(this),
186
190
  });
@@ -527,9 +531,12 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
527
531
  // Mirrors the owned-only `initialize()` in `init()` - without this, the predictor we built keeps its
528
532
  // PERSIST_STATE listener registered after the crawl and never gets a final write.
529
533
  await this.#renderingTypePredictor.ifOwned((predictor) => predictor.teardown());
530
- for (const hook of this.#teardownHooks) {
531
- await hook();
532
- }
534
+ await this.#browserCrawler.teardown();
535
+ }
536
+ async destroy() {
537
+ await super.destroy();
538
+ await this.#staticCrawler.destroy();
539
+ await this.#browserCrawler.destroy();
533
540
  }
534
541
  }
535
542
  export function createAdaptivePlaywrightRouter(routesOrSchemas) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@crawlee/playwright",
3
- "version": "4.0.0-beta.165",
3
+ "version": "4.0.0-beta.167",
4
4
  "description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
5
5
  "engines": {
6
6
  "node": ">=22.0.0"
@@ -49,13 +49,13 @@
49
49
  "dependencies": {
50
50
  "@apify/datastructures": "^3.0.1",
51
51
  "@apify/timeout": "^1.0.1",
52
- "@crawlee/basic": "4.0.0-beta.165",
53
- "@crawlee/browser": "4.0.0-beta.165",
54
- "@crawlee/browser-pool": "4.0.0-beta.165",
55
- "@crawlee/cheerio": "4.0.0-beta.165",
56
- "@crawlee/core": "4.0.0-beta.165",
57
- "@crawlee/types": "4.0.0-beta.165",
58
- "@crawlee/utils": "4.0.0-beta.165",
52
+ "@crawlee/basic": "4.0.0-beta.167",
53
+ "@crawlee/browser": "4.0.0-beta.167",
54
+ "@crawlee/browser-pool": "4.0.0-beta.167",
55
+ "@crawlee/cheerio": "4.0.0-beta.167",
56
+ "@crawlee/core": "4.0.0-beta.167",
57
+ "@crawlee/types": "4.0.0-beta.167",
58
+ "@crawlee/utils": "4.0.0-beta.167",
59
59
  "cheerio": "^1.0.0",
60
60
  "jquery": "^3.7.1",
61
61
  "ml-logistic-regression": "^2.0.0",
@@ -80,5 +80,5 @@
80
80
  }
81
81
  }
82
82
  },
83
- "gitHead": "e41d837b93d16b5e5c726d4dc00aa6f3f1282b51"
83
+ "gitHead": "8b26080bf07dc602aa0970e12fb66e7e9cba509b"
84
84
  }