mcp-scraper 0.94.3 → 0.95.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +52 -38
- package/README.md +4 -4
- package/dist/bin/api-server.js +1 -1
- package/dist/bin/mcp-scraper-cli.js +1 -1
- package/dist/bin/mcp-scraper-core.js +1 -1
- package/dist/bin/mcp-scraper-install.js +1 -1
- package/dist/bin/mcp-stdio-server.js +1 -1
- package/dist/bin/paa-harvest.js +1 -1
- package/dist/chunk-6FRHHGYH.js +6 -0
- package/dist/{chunk-VVVWIREO.js → chunk-7TX3QAUW.js} +7 -7
- package/dist/chunk-EDQ57S4P.js +1 -0
- package/dist/{chunk-ZTEDLK3Y.js → chunk-FCJXL4PU.js} +1 -1
- package/dist/{chunk-72MWPIMR.js → chunk-FN7U6KRI.js} +1 -1
- package/dist/{chunk-QDP3XTES.js → chunk-I3HNHCIJ.js} +2 -2
- package/dist/chunk-IDM6FAYS.js +1 -0
- package/dist/{chunk-U6QBJG3X.js → chunk-KUR2T5L2.js} +96 -96
- package/dist/{chunk-C5MWTMRD.js → chunk-L6LQYPXI.js} +4 -4
- package/dist/{chunk-C7QL25ZJ.js → chunk-M5QHXNFZ.js} +3 -3
- package/dist/chunk-NA76FFGD.js +15 -15
- package/dist/chunk-NZSDHDX3.js +1 -1
- package/dist/{chunk-PGCB7H6I.js → chunk-RJPQCPDX.js} +1 -1
- package/dist/chunk-SPI4XQFN.js +1 -0
- package/dist/{chunk-6FSYF5K5.js → chunk-WPNBO2MN.js} +1 -1
- package/dist/{chunk-WHUD4R2S.js → chunk-XG6GCEUE.js} +1 -1
- package/dist/chunk-XPEJB4BZ.js +1 -1
- package/dist/{chunk-BRNTVOCW.js → chunk-ZJTUAWZA.js} +1 -1
- package/dist/chunk-ZPK5LYEN.js +15 -15
- package/dist/chunk-ZUGTYD2I.js +4 -4
- package/dist/{extract-bundle-C4LA4NFP.js → extract-bundle-4CFBBSYQ.js} +1 -1
- package/dist/{gmail-service-U7NP5IMY.js → gmail-service-FZVJBSUT.js} +1 -1
- package/dist/index.cjs +49 -49
- package/dist/index.d.cts +14 -14
- package/dist/index.d.ts +14 -14
- package/dist/index.js +1 -1
- package/dist/{server-AUAOV6V7.js → server-AVTIYMCH.js} +43 -43
- package/dist/{site-extract-repository-DUBZXCRG.js → site-extract-repository-UVE7TWJO.js} +1 -1
- package/dist/{stripe-event-worker-2KOMIZXB.js → stripe-event-worker-2LSUVQOM.js} +1 -1
- package/dist/worker-VHDMSMEC.js +1 -0
- package/package.json +137 -17
- package/THIRD_PARTY_NOTICES.html +0 -203
- package/dist/chunk-DSALLDK7.js +0 -6
- package/dist/chunk-GEN6EGHI.js +0 -1
- package/dist/chunk-SH5KB4P7.js +0 -1
- package/dist/chunk-ZGITQEFC.js +0 -1
- package/dist/worker-HCLS7MZ7.js +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,11 +4,23 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.95.0] - 2026-09-25
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Filter Google SERP searches to the past week or past month through a managed browser at 35 Credits per delivered page. Two-page requests follow Google's native Page 2 control. Without a date filter, existing search providers remain primary and the managed browser is the final backup. Browser results contain organic listings only; an available page one is marked partial if page two fails.
|
|
12
|
+
|
|
13
|
+
## [0.94.4] - 2026-09-25
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
|
|
17
|
+
- Scroll through large Google Maps review lists while the previous small review batch is being saved, then wait for that save before sending another batch or returning a result. A shorter pause between scrolls gives the browser more chances to load reviews within Kernel's ten-minute session.
|
|
18
|
+
|
|
7
19
|
## [0.94.3] - 2026-09-25
|
|
8
20
|
|
|
9
21
|
### Fixed
|
|
10
22
|
|
|
11
|
-
- Give large Google Maps review requests more scrolling time within
|
|
23
|
+
- Give large Google Maps review requests more scrolling time within Kernel's ten-minute session. A hosted 1,000-review run on 0.94.2 saved 828 reviews before its shorter scrolling allowance expired; this patch uses the remaining session time while retaining saved partial results and count-based billing.
|
|
12
24
|
|
|
13
25
|
## [0.94.2] - 2026-09-25
|
|
14
26
|
|
|
@@ -21,7 +33,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
21
33
|
### Changed
|
|
22
34
|
|
|
23
35
|
- Accept up to 1,000 reviews in a Maps place request across API, MCP, dashboard, and workflows. Scale the review scroll allowance with the requested count and preserve each saved review batch through interrupted runs.
|
|
24
|
-
- Read the named business first during
|
|
36
|
+
- Read the named business first during Bright Data service and area acquisition, matching the full Bright Data place path. Retry a challenged Bright Data browser once with a fresh session and meter both sessions. Keep a visible partial run and refund its hold if Google challenges both.
|
|
25
37
|
|
|
26
38
|
### Fixed
|
|
27
39
|
|
|
@@ -34,23 +46,23 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
34
46
|
- Recoverable Maps place runs with owner-scoped status, paginated saved reviews and images, and explicit resume tools. MCP now reads the same run after a client timeout.
|
|
35
47
|
- Incremental, revision-fenced Maps place checkpoints and five-minute repair of expired attempts and unsettled customer holds. Image ZIP packaging can resume from saved image URLs without reopening a browser.
|
|
36
48
|
- Google Maps place calls accept an `include` declaration, including `all`, to request reviews, configured services and areas served, and images within the existing review and image limits.
|
|
37
|
-
- Maps uses
|
|
49
|
+
- Maps uses Bright Data Browser API for requested configured services and areas served, while Kernel remains the default for direct Google Maps discovery and place details.
|
|
38
50
|
|
|
39
51
|
### Changed
|
|
40
52
|
|
|
41
53
|
- Maps place requests reserve the maximum declared base and review charge before browser work, then settle completed attempts to the base plus unique saved reviews. Stopped attempts are refunded; saved results survive a billing repair.
|
|
42
54
|
- Ordinary Maps searches now use state-targeted mobile browser egress when `location` names a US state, then direct egress after two failed mobile attempts. Searches without a state use direct egress; the MCP tool asks for a state without exposing proxy settings.
|
|
43
55
|
|
|
44
|
-
- Maps search and place requests now use a provider broker and report both the result provider and, for split place requests, the acquisition provider.
|
|
45
|
-
- Align
|
|
46
|
-
-
|
|
47
|
-
-
|
|
48
|
-
-
|
|
49
|
-
-
|
|
56
|
+
- Maps search and place requests now use a provider broker and report both the result provider and, for split place requests, the acquisition provider. Kernel discovery reads the Google Maps feed; Bright Data reads the visible Businesses pack or clicks More businesses for larger service-enriched searches.
|
|
57
|
+
- Align Bright Data Maps acquisition with the observed Google Places DOM: recognize the `udm=local` finder after More businesses, open exact `pv-` result cards, read the `#local-place-viewer` profile and its complete Services and Areas Served dialogs, and verify place identity from its `ftid` Maps link. Named service requests use Kernel's place category to reach the organic Businesses pack before Bright Data enrichment.
|
|
58
|
+
- Bright Data-only named place requests now begin on the organic Google SERP and use the matching knowledge panel's category to find the business in the organic Businesses results. A cold exact-name Places finder is reserved for profiles without a category, and a Google challenge is reported as a CAPTCHA rather than an unexplained missing business.
|
|
59
|
+
- Bright Data-only place extraction stays in the opened Google Search Places viewer to read profile details, expanded hours, declared services and areas, photos, and review cards up to the requested limits; a live run collected 30 services, 10 areas, 50 reviews, and 10 photo URLs. Maps SERP navigation preserves provider errors and allows up to two minutes for slow Browser API navigation.
|
|
60
|
+
- Bright Data Maps search, service acquisition, and place details can start one new browser session after the provider confirms `no_ready_cookies` with zero navigations, separately from Maps extraction retries; that recovery waits five seconds after connecting before navigation.
|
|
61
|
+
- Bright Data place review and gallery collection now scale their time and scroll allowances with the declared review and image limits. A default full declaration has a five-minute planning budget; the largest allowed request has nine minutes.
|
|
50
62
|
|
|
51
63
|
### Fixed
|
|
52
64
|
|
|
53
|
-
- Attribute Maps provider cost to each browser actually used, including failed and retried
|
|
65
|
+
- Attribute Maps provider cost to each browser actually used, including failed and retried Bright Data search, place, services acquisition, and startup recovery sessions. Kernel rotations and broker fallbacks are linked as retries for CEO reporting, which now shows failed-attempt spend separately. Kernel costs remain estimates; Bright Data sessions with delayed usage remain pending for reconciliation.
|
|
54
66
|
- Include web-app embedding requests in the paid-action inventory and scan web source during future cost-coverage checks.
|
|
55
67
|
- Include configured S3-compatible storage and remote media processing in the paid-action inventory, and show live Stripe account fee totals separately from MCP operation costs in the CEO report.
|
|
56
68
|
|
|
@@ -66,13 +78,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
66
78
|
### Fixed
|
|
67
79
|
|
|
68
80
|
- Render the CEO report headline and detail from one cost-ledger snapshot, and label cost-linked surcharge multiples as pricing targets rather than guaranteed realized margin.
|
|
69
|
-
- Apply the owner-confirmed $1.50 per 1,000 successful
|
|
81
|
+
- Apply the owner-confirmed $1.50 per 1,000 successful Bright Data SERP requests to delivered provider receipts, with a guarded backfill for earlier successful requests. Failed attempts remain visibly unresolved until provider billing evidence is available.
|
|
70
82
|
|
|
71
83
|
## [0.93.2] - 2026-09-24
|
|
72
84
|
|
|
73
85
|
### Fixed
|
|
74
86
|
|
|
75
|
-
- Show Google Search and other provider costs in the CEO report with separate calculated, actual-or-measured, and unresolved receipt coverage. Failed queries now stop the report, missing
|
|
87
|
+
- Show Google Search and other provider costs in the CEO report with separate calculated, actual-or-measured, and unresolved receipt coverage. Failed queries now stop the report, missing Bright Data SERP rates leave an explicit delivered-but-unpriced receipt, and historical cost rows no longer imply a margin from subscription MRR.
|
|
76
88
|
|
|
77
89
|
## [0.93.1] - 2026-09-24
|
|
78
90
|
|
|
@@ -171,7 +183,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
171
183
|
|
|
172
184
|
- Retire the inactive Personal Assistant from server startup, HTTP and MCP routing, web navigation, Scheduler transitions, generated contracts, and Memory tool registration while preserving its source history, persisted data, secrets, and provider resources behind the verified archive branch.
|
|
173
185
|
- Make two-page SERP capture perform a distinct page-two request, preserve page provenance and query/location intent, and report local-pack evidence as present, absent, incomplete, or unknown instead of silently claiming completeness.
|
|
174
|
-
- Make Maps retries follow one immutable egress plan, retain location intent, close every browser/proxy attempt, and treat an already-gone
|
|
186
|
+
- Make Maps retries follow one immutable egress plan, retain location intent, close every browser/proxy attempt, and treat an already-gone Kernel session as successful cleanup.
|
|
175
187
|
|
|
176
188
|
### Fixed
|
|
177
189
|
|
|
@@ -185,13 +197,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
185
197
|
|
|
186
198
|
- Add the fixture-tested operation, attempt, provider-receipt, reconciliation, and billing-link foundation needed to trace failed and retried SERP, PAA, Maps, extraction, and transcription work without treating missing provider cost as zero.
|
|
187
199
|
- Add a generated MCP cost-coverage manifest and release gate that tracks every hosted input flag plus each named execution method's retry, timeout, billing, and cost-accounting owner.
|
|
188
|
-
- Trace direct
|
|
189
|
-
- Add a protected four-cell PAA benchmark runner for
|
|
200
|
+
- Trace direct Bright Data SERP requests and Bright Data/Kernel browser attempts into the shared operation timeline, including provider identities, retry or fallback causality, immediate estimates, hourly due-gated Browser API reconciliation, and customer billing links.
|
|
201
|
+
- Add a protected four-cell PAA benchmark runner for Kernel and Bright Data at 20 and 40 complete questions, with a dry-run contract gate, one paid attempt per cell, sanitized provider evidence, and no automatic replacement run.
|
|
190
202
|
- Add a once-daily due-gated purge of raw provider identifiers and verbose diagnostic errors after 30 days while preserving normalized cost facts and hashed correlation keys.
|
|
191
203
|
|
|
192
204
|
### Fixed
|
|
193
205
|
|
|
194
|
-
- Skip
|
|
206
|
+
- Skip Kernel proxy resolution when the active attempt uses Bright Data Browser, removing avoidable Kernel API traffic from Bright Data-first SERP and PAA work.
|
|
195
207
|
|
|
196
208
|
## [0.90.4] - 2026-09-22
|
|
197
209
|
|
|
@@ -203,21 +215,21 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
203
215
|
|
|
204
216
|
### Fixed
|
|
205
217
|
|
|
206
|
-
- Route ordinary
|
|
218
|
+
- Route ordinary Bright Data SERP searches through the zone's native parsed proxy, returning Google results within the existing bounded deadline while preserving the REST endpoint as a configuration fallback.
|
|
207
219
|
- Keep Browser API credentials and interactive PAA behavior independent from the native SERP transport, with one provider request and no hidden retry or browser fallback.
|
|
208
220
|
|
|
209
221
|
## [0.90.2] - 2026-09-21
|
|
210
222
|
|
|
211
223
|
### Fixed
|
|
212
224
|
|
|
213
|
-
- Unwrap
|
|
225
|
+
- Unwrap Bright Data's object-valued REST response body before mapping parsed SERP fields, so successful direct searches return their organic results instead of an empty collection.
|
|
214
226
|
|
|
215
227
|
## [0.90.1] - 2026-09-21
|
|
216
228
|
|
|
217
229
|
### Changed
|
|
218
230
|
|
|
219
|
-
- Route ordinary `search_serp` calls through one parsed
|
|
220
|
-
- Keep saved SERP identities on their existing
|
|
231
|
+
- Route ordinary `search_serp` calls through one parsed Bright Data SERP API request when the dedicated production zone is configured, avoiding browser startup and cleanup while leaving interactive PAA on Browser API.
|
|
232
|
+
- Keep saved SERP identities on their existing Kernel-backed path and retain the browser path as the configuration fallback for local and unconfigured environments.
|
|
221
233
|
|
|
222
234
|
### Fixed
|
|
223
235
|
|
|
@@ -260,7 +272,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
260
272
|
|
|
261
273
|
### Changed
|
|
262
274
|
|
|
263
|
-
- Cap every
|
|
275
|
+
- Cap every Kernel browser session at a ten-minute absolute lifetime and move expired-session reconciliation from the minute root cron to a dedicated hourly audit.
|
|
264
276
|
- Disable the inactive Personal Assistant reminder, reconciliation, and inbound cron schedules while preserving their routes and implementation.
|
|
265
277
|
- Disable the inactive Personal Memory heartbeat and weekly rollup registrations in the production scheduler while preserving their implementation.
|
|
266
278
|
- Pause twice-daily automatic memory optimization while preserving the workflow for deliberate use, preventing all-vault fan-out from consuming background execution capacity.
|
|
@@ -269,10 +281,10 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
269
281
|
|
|
270
282
|
### Fixed
|
|
271
283
|
|
|
272
|
-
- Treat
|
|
284
|
+
- Treat Kernel's not-found response during legacy session deletion as successful cleanup, preventing already-closed sessions from retrying forever.
|
|
273
285
|
- Show browser sessions awaiting provider cleanup separately in the CTO report instead of hiding them behind a non-null close timestamp.
|
|
274
286
|
- Cast the analytics pruning clock before PostgreSQL interval arithmetic so the root cron no longer fails every minute while pruning scheduled occurrences.
|
|
275
|
-
- Explicitly delete
|
|
287
|
+
- Explicitly delete Kernel screenshot sessions after capture, including when closing the browser connection fails.
|
|
276
288
|
|
|
277
289
|
## [0.89.7] - 2026-09-18
|
|
278
290
|
|
|
@@ -391,7 +403,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
391
403
|
|
|
392
404
|
### Fixed
|
|
393
405
|
|
|
394
|
-
- Give each `reddit_thread` retrieval a 300-second end-to-end deadline, with two 60-second
|
|
406
|
+
- Give each `reddit_thread` retrieval a 300-second end-to-end deadline, with two 60-second Kernel attempts and two 60-second managed-browser backup attempts, instead of exhausting the full retry ladder in about 50 seconds. The MCP client now waits long enough to receive the endpoint's structured terminal result.
|
|
395
407
|
|
|
396
408
|
## [0.88.2] - 2026-09-02
|
|
397
409
|
|
|
@@ -428,19 +440,19 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
428
440
|
|
|
429
441
|
### Fixed
|
|
430
442
|
|
|
431
|
-
- Kept
|
|
443
|
+
- Kept Bright Data telemetry lookup off the Reddit response critical path and reallocated the saved time to 17-second backup attempts, so all four provider attempts can finish before production ends the request.
|
|
432
444
|
|
|
433
445
|
## [0.86.4] - 2026-09-02
|
|
434
446
|
|
|
435
447
|
### Fixed
|
|
436
448
|
|
|
437
|
-
- Kept the complete two-primary, two-backup Reddit retry ladder inside the production request window by limiting
|
|
449
|
+
- Kept the complete two-primary, two-backup Reddit retry ladder inside the production request window by limiting Kernel attempts to 8 seconds, Bright Data attempts to 14 seconds, and browser cleanup to 1 second.
|
|
438
450
|
|
|
439
451
|
## [0.86.3] - 2026-09-02
|
|
440
452
|
|
|
441
453
|
### Fixed
|
|
442
454
|
|
|
443
|
-
- Applied 45-second
|
|
455
|
+
- Applied 45-second Kernel and 35-second Bright Data deadlines to the complete Reddit browser-attempt lifecycle, and made known-thread primary attempts find and click the target through DuckDuckGo before the residential landing.
|
|
444
456
|
|
|
445
457
|
## [0.86.2] - 2026-09-02
|
|
446
458
|
|
|
@@ -589,13 +601,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
589
601
|
|
|
590
602
|
### Added
|
|
591
603
|
|
|
592
|
-
- Added a
|
|
593
|
-
- Added a bounded managed-browser backup for Reddit thread hydration after the primary
|
|
604
|
+
- Added a Kernel-only Reddit workflow that searches DuckDuckGo with a `site:reddit.com` query, switches the same browser to a residential proxy before clicking the selected result, and reads modern Reddit posts plus bounded rendered-comment expansion through dedicated search, thread, and combined REST endpoints.
|
|
605
|
+
- Added a bounded managed-browser backup for Reddit thread hydration after the primary Kernel attempt fails or returns fewer than the semantic target, capped at 25 comments with measured bandwidth, duration, CAPTCHA, closure, and provider-cost telemetry.
|
|
594
606
|
|
|
595
607
|
### Changed
|
|
596
608
|
|
|
597
|
-
- Routed the production `reddit_thread` and `reddit_trending` MCP tools through modern Reddit on
|
|
598
|
-
- Cost probes now include Reddit
|
|
609
|
+
- Routed the production `reddit_thread` and `reddit_trending` MCP tools through modern Reddit on Kernel residential sessions, with DuckDuckGo site search for trend discovery; removed Google and old Reddit from their active execution path while preserving tool names, billing rates, bounded partial results, and refunds.
|
|
610
|
+
- Cost probes now include Reddit Kernel sessions and any managed-browser fallback bytes and cost in the same request receipt, and identify when the backup contributed to total cost.
|
|
599
611
|
|
|
600
612
|
### Fixed
|
|
601
613
|
|
|
@@ -640,7 +652,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
640
652
|
|
|
641
653
|
### Fixed
|
|
642
654
|
|
|
643
|
-
- Persisted per-control PAA dispatch and 0.7/1.0/1.4-second confirmation telemetry in durable checkpoints, exposed recent interaction and attempt correlation through MCP status, attached
|
|
655
|
+
- Persisted per-control PAA dispatch and 0.7/1.0/1.4-second confirmation telemetry in durable checkpoints, exposed recent interaction and attempt correlation through MCP status, attached Bright Data session IDs immediately after browser launch, and finalized dangling attempt rows during lease recovery without blocking customer settlement.
|
|
644
656
|
- Prevented inline style, script, and hidden DOM text inside Google answer containers from falsely confirming that PAA answer material loaded.
|
|
645
657
|
- Routed canonical `/assistant` page loads to the web app and the redacted private Assistant readiness endpoint to the main API function, preventing production 404s after the 0.79.1 launch.
|
|
646
658
|
|
|
@@ -664,7 +676,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
664
676
|
- Added Scheduling as the canonical Personal Assistant setup surface, with connection readiness for Gmail, Calendar, Zoom, browser profiles, Memory, SMS, and email; exact schedule confirmation; approval and spend review; run history; and explicit watch/takeover states.
|
|
665
677
|
- Added owner-scoped browser profiles that can hold multiple independently verified login bindings, while every browser schedule grant selects one exact profile, login, domain, and action set.
|
|
666
678
|
- Added immutable schedule revisions, readiness receipts, append-only activation records, additive legacy schedule projection, and single-owner occurrence transition receipts so migration cannot silently infer browser authority or double-dispatch work.
|
|
667
|
-
- Added
|
|
679
|
+
- Added Kernel and private-Mac browser runtime boundaries with collision-resistant tenant namespaces, per-owner concurrency ceilings, bounded sessions, explicit and timeout cleanup, owner-qualified account deletion, and provider deletion readback.
|
|
668
680
|
- Added an owner-controlled Personal Assistant that brings SMS/MMS, Gmail, Google Calendar, Zoom, browser work, reminders, and Memory context packets into one governed workflow with immutable plans, approval checkpoints, spend limits, and durable receipts.
|
|
669
681
|
- Added Twilio number discovery, owned-number attachment, purchase and registration previews, Messaging Service readiness, signed inbound and delivery webhooks, safe MMS ingestion, deterministic opt-out handling, single and reviewed bulk messaging, and reconciliation for unknown provider outcomes.
|
|
670
682
|
- Added immutable, revisioned Memory context packets with source and attachment provenance, Gmail full-message imports, MMS media metadata, lifecycle controls, and readback verification against the selected vault.
|
|
@@ -701,7 +713,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
701
713
|
- Made `maxQuestions` an explicit target count rather than a traversal-depth control, with separate discovery and material-completeness diagnostics.
|
|
702
714
|
- Preserved complete People Also Ask, AI Overview, and organic-result link provenance in JSON, structured MCP output, and CSV while classifying plain links and Google redirect links explicitly.
|
|
703
715
|
- Resolved opaque Google `/goto` targets through bounded concurrent manual-redirect requests with active-browser interception as a fallback, without following publisher destinations and without dropping unresolved material.
|
|
704
|
-
- Aligned the bounded PAA production-provider canary with the public `maxQuestions` contract and made
|
|
716
|
+
- Aligned the bounded PAA production-provider canary with the public `maxQuestions` contract and made Bright Data the default test provider.
|
|
705
717
|
|
|
706
718
|
### Fixed
|
|
707
719
|
|
|
@@ -909,7 +921,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
909
921
|
|
|
910
922
|
- Added portable `harvest_paa_start` and `harvest_paa_status` tools for durable long-running PAA research, with stable idempotency recovery, progress, attempt provenance, completeness, billing state, and bounded provider telemetry.
|
|
911
923
|
- Added progressive PAA checkpoints that preserve and merge the best unique rows across browser retries and stale-job recovery instead of losing already captured questions when a provider session or caller is interrupted.
|
|
912
|
-
- Added exact
|
|
924
|
+
- Added exact Bright Data browser-session identity, sanitized Session Logs enrichment, disconnect attribution, bandwidth usage telemetry, and retryable reconciliation without making provider telemetry a prerequisite for result delivery.
|
|
913
925
|
|
|
914
926
|
### Changed
|
|
915
927
|
|
|
@@ -1642,7 +1654,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
1642
1654
|
|
|
1643
1655
|
### Changed
|
|
1644
1656
|
|
|
1645
|
-
- PAA browser work now uses
|
|
1657
|
+
- PAA browser work now uses Kernel's co-located Playwright execution with stealth mode's default managed proxy and native browser metadata. Location is expressed only through Google UULE, CAPTCHA solver waiting is capped at 60 seconds, and a fresh session is allowed once only when no useful data was captured.
|
|
1646
1658
|
- PAA invocations stop browser work at 250 seconds inside the 280-second application budget, reserving 30 seconds for persistence, cleanup, and settlement. The legacy cron worker no longer claims Inngest-owned PAA jobs.
|
|
1647
1659
|
|
|
1648
1660
|
### Fixed
|
|
@@ -1878,7 +1890,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
1878
1890
|
|
|
1879
1891
|
### Changed
|
|
1880
1892
|
|
|
1881
|
-
- `maps_search` now applies a transport ladder across its retry attempts so it can recover from Google soft-blocks instead of only retrying the same way. The first attempt is unchanged (
|
|
1893
|
+
- `maps_search` now applies a transport ladder across its retry attempts so it can recover from Google soft-blocks instead of only retrying the same way. The first attempt is unchanged (Kernel's default stealth ISP proxy, direct navigation). Subsequent retries switch to direct egress and arrive at Google through a cross-site redirect (the combination that measurably clears blocks a cold navigation triggers); the final escalation attempt uses direct egress without the redirect and accepts any egress country. This only affects the `proxyMode: 'none'` default path and only its retries — a first-attempt success behaves exactly as before.
|
|
1882
1894
|
|
|
1883
1895
|
## [0.32.1] - 2026-07-22
|
|
1884
1896
|
|
|
@@ -2125,7 +2137,9 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
2125
2137
|
- Write actions remain unavailable until the account owner explicitly enables them.
|
|
2126
2138
|
- Provider-specific connection data is normalized into one agent-facing contract.
|
|
2127
2139
|
|
|
2128
|
-
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.
|
|
2140
|
+
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.95.0...HEAD
|
|
2141
|
+
[0.95.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.94.4...v0.95.0
|
|
2142
|
+
[0.94.4]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.94.3...v0.94.4
|
|
2129
2143
|
[0.94.3]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.94.2...v0.94.3
|
|
2130
2144
|
[0.94.2]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.94.1...v0.94.2
|
|
2131
2145
|
[0.94.1]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.94.0...v0.94.1
|
package/README.md
CHANGED
|
@@ -175,7 +175,7 @@ Build the branded one-click bundle:
|
|
|
175
175
|
npm run build:mcpb
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
-
The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.
|
|
178
|
+
The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.95.0`, SHA-256 `f97660a487fd1099ba0b6664c0e288fb3095f10adb7c24a2132323c2ed9c9894`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
|
|
179
179
|
|
|
180
180
|
The MCPB install exposes every tool — web-intelligence plus all `browser_*` tools — through the one `mcp-scraper` server.
|
|
181
181
|
|
|
@@ -345,7 +345,7 @@ Google Search Console exposes eight bounded reads and eight gated property and s
|
|
|
345
345
|
|
|
346
346
|
For accurate annotated videos, do not guess annotation times from a script. Start the replay, navigate until each target is visible and stable, call `browser_replay_mark` for each callout, then stop the replay and pass the returned annotations to `browser_replay_annotate` with the returned `source_width` and `source_height`.
|
|
347
347
|
|
|
348
|
-
For `search_serp`, callers provide a query and can reuse an `idempotencyKey` after an uncertain response. Light mode returns organic positions, URLs, titles, and descriptions.
|
|
348
|
+
For `search_serp`, callers provide a query and can reuse an `idempotencyKey` after an uncertain response. Light mode returns organic positions, URLs, titles, and descriptions. Unfiltered one-page full mode adds available same-page SERP features, including local results, discussions, videos, AI features, and on-page questions. Both modes default to one page; request `pages: 2` explicitly for a second page. Set `recency: "week"` or `recency: "month"` to apply Google's past-week or past-month filter. Filtered searches use Bright Data Browser API as the primary provider and return organic results only. A one-page filtered search costs 35 Credits. With `pages: 2`, the browser scrolls to Google's pagination control and clicks the native Page 2 link; two delivered pages cost 70 Credits. For unfiltered searches, Browser API is the last resort after the normal provider path fails; rich features are marked unsupported if it supplies the result. Full mode does not open result URLs, expand PAA questions, or fetch Maps business profiles. Unfiltered light mode costs 20 Credits per delivered page, or 35 Credits when a backup supplies it; full mode costs 35 Credits per delivered page. Two-page searches return organic results only. Location, language, device, and legacy optional-module fields remain accepted for compatibility but do not change ordinary searches. Other Google SERP and Maps tools retain their own regional inputs. MCP Scraper owns transport selection and bounded retries internally; implementation controls and receipts are not part of the public tool contract.
|
|
349
349
|
|
|
350
350
|
The `mcp-scraper` server (and the MCPB bundle, which runs it) exposes both sections through one MCP server.
|
|
351
351
|
|
|
@@ -366,8 +366,8 @@ The `mcp-scraper` NPX stdio server also exposes saved reports as MCP resources:
|
|
|
366
366
|
- `MCP_SCRAPER_OUTPUT_DIR` is optional and defaults to `~/Downloads/mcp-scraper`.
|
|
367
367
|
- `MCP_SCRAPER_SAVE_REPORTS=false` disables automatic Markdown report files.
|
|
368
368
|
- `MCP_SCRAPER_KEY_PATH` is optional. When no API key env var is set, the server also reads `~/.mcp-scraper-key` for compatibility with older installs.
|
|
369
|
-
- `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `mcp-scraper` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `
|
|
370
|
-
- `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `
|
|
369
|
+
- `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `mcp-scraper` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `KERNEL_BROWSER_PROFILE_NAME`, `KERNEL_PROFILE_NAME`.
|
|
370
|
+
- `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `KERNEL_BROWSER_PROFILE_SAVE_CHANGES`, `KERNEL_PROFILE_SAVE_CHANGES`.
|
|
371
371
|
|
|
372
372
|
Hosted operators can isolate authorization state in a dedicated Turso/libSQL database without changing the public MCP tool catalog or API-key authentication. The secured store uses atomic authorization-code exchange and refresh rotation, keyed secret lookup, bounded encrypted replay receipts, authority epochs, and fail-closed maintenance behavior. Production migration and rollback are controlled data moves, not ordinary mode flips; see [MCP OAuth operations](docs/operations/mcp-oauth-runbook.md). Existing client setup and reconnect behavior are unchanged in Phase 1.
|
|
373
373
|
|
package/dist/bin/api-server.js
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import{readFileSync as s}from"fs";function c(){try{for(let r of s(".env","utf8").split(`
|
|
3
|
-
`)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-
|
|
3
|
+
`)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-AVTIYMCH.js"),import("../worker-VHDMSMEC.js"),import("../db-B6RNXKEI.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-A7VMA24D.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-6ND5AFV5.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-XPEJB4BZ.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import{a as P}from"../chunk-
|
|
2
|
+
import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-6ND5AFV5.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-XPEJB4BZ.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import{a as P}from"../chunk-IDM6FAYS.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
|
|
3
3
|
`)){let l=a.match(/^\s{2,}([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);if(!l){if(a.trim().length&&!/^\s{2,}/.test(a))break;continue}p[l[1]]=l[2]}return{command:n,args:i,env:p}}function j(e){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(e.env))n.push("--env",`${c}=${i}`);return n.push("--",e.command,...e.args),n}function G(e={}){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(pe(e)))n.push("--env",`${c}=${i}`);return n.push("--","npx",...A(e)),n}function O(e){if(e==="claude-code")return"claude";if(e==="claude"||D.hosts.some(n=>n.id===e))return e;throw new Error('Unknown host "'+e+'". Use: codex, claude, claude-code, claude-desktop, cursor, windsurf, cline, or user-action-only')}function ue(e){return K(e==="claude"?"claude-code":e)}function W(e,n={}){let c=O(e),i=ue(c),p="Restart the MCP client so it starts a fresh npx process.",u='MCP_SCRAPER_API_KEY="$MCP_SCRAPER_API_KEY" npx -y -p mcp-scraper@latest mcp-scraper-cli agent install claude --apply',a=`X-Ray install protocol: ${v} (${b})`;return c==="codex"?["# Codex MCP config",a,i.exactConfig,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
|
|
4
4
|
`):c==="claude"?["# Claude Code command",a,i.exactConfig,"","# One-command Claude Code setup",u,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
|
|
5
5
|
`):c==="claude-desktop"?["# Claude Desktop config",a,i.exactConfig,"","Desktop Extension: https://mcpscraper.dev/downloads/mcp-scraper.mcpb",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,p].join(`
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as e}from"../chunk-
|
|
2
|
+
import{a as e}from"../chunk-WPNBO2MN.js";import"../chunk-KUR2T5L2.js";import"../chunk-NZSDHDX3.js";import"../chunk-W2BVJ7S2.js";import"../chunk-DOJMTFLB.js";import"../chunk-ZUGTYD2I.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-EDQ57S4P.js";import"../chunk-SPI4XQFN.js";import"../chunk-NA76FFGD.js";import"../chunk-6DTXIZY2.js";import"../chunk-L6LQYPXI.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-IDM6FAYS.js";import"../chunk-ZPK5LYEN.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as s}from"../chunk-
|
|
2
|
+
import{a as s}from"../chunk-L6LQYPXI.js";import{a as e}from"../chunk-IDM6FAYS.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
|
|
3
3
|
`)),process.exit(0));process.stdout.write(s({version:e,color:!r,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()}));
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as r}from"../chunk-
|
|
2
|
+
import{a as r}from"../chunk-WPNBO2MN.js";import"../chunk-KUR2T5L2.js";import"../chunk-NZSDHDX3.js";import"../chunk-W2BVJ7S2.js";import"../chunk-DOJMTFLB.js";import"../chunk-ZUGTYD2I.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-EDQ57S4P.js";import"../chunk-SPI4XQFN.js";import"../chunk-NA76FFGD.js";import"../chunk-6DTXIZY2.js";import"../chunk-L6LQYPXI.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-IDM6FAYS.js";import"../chunk-ZPK5LYEN.js";r();
|
package/dist/bin/paa-harvest.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{z as t}from"../chunk-
|
|
2
|
+
import{z as t}from"../chunk-7TX3QAUW.js";import"../chunk-M5QHXNFZ.js";import{b as r}from"../chunk-SPI4XQFN.js";import"../chunk-NA76FFGD.js";import"../chunk-6DTXIZY2.js";import"../chunk-XPEJB4BZ.js";import"../chunk-ZPK5LYEN.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--kernel-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,kernelApiKey:e.browserApiKey??e.kernelApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import{a as Dt}from"./chunk-XG6GCEUE.js";import{k as H,m as kt,n as Ot,z as ue}from"./chunk-7TX3QAUW.js";import{a as L,c as xe,d as ce,e as V,f as de,j as Me,k as Nt}from"./chunk-NZSDHDX3.js";import{b as It,c as Tt,d as $,f as z,g as B,h as U,i as Ct,k as F,n as xt,q as Mt}from"./chunk-ZUGTYD2I.js";import{a as Ne,b as De,g as qe}from"./chunk-EDQ57S4P.js";import{a as pt,b as mt}from"./chunk-SPI4XQFN.js";import{a as Rt,b as Pt,d as Et}from"./chunk-NA76FFGD.js";import{b as St,c as At}from"./chunk-6DTXIZY2.js";import{e as dt}from"./chunk-XPEJB4BZ.js";import{Nb as le,Rb as Te,Tb as Ce,Va as Oe,b as Ee,g as M,h as gt,hb as vt,ia as bt,j as ft,ja as _t,jb as Ie,la as wt,m as ht,o as yt,oa as ke}from"./chunk-ZPK5LYEN.js";import{request as Rr}from"https";import{HttpsProxyAgent as Pr}from"https-proxy-agent";var pe={source:"https://brightdata.com/users/zone/premium_domains",retrievedAt:"2026-09-23T20:14:30.483Z",includedStatuses:["premium","remove_candidate"],domains:["advanceautoparts.com","affitto.it","agoda.cn","albertsons.com","allpeople.com","autozone.com","bestbuy.com","bestwestern.com","billiger.de","bottlerover.com","carousell.com","carousell.com.hk","carousell.com.my","carousell.ph","carousell.sg","carsales.com.au","cdiscount.com","chewy.com","costco.com","cvs.com","despegar.com.mx","dickssportinggoods.com","dynos.es","emaxme.com","familytreenow.com","feuvert.fr","flooranddecor.com","foodlion.com","footlocker.co.uk","footlocker.com","giantfoodstores.com","gopuff.com","gplay.bg","hermes.com","hyatt.com","idealo.de","immobilienscout24.de","ingatlan.com","instacart.com","intersport.fr","joann.com","kroger.com","lazada.co.id","lazada.co.th","lazada.com.my","lazada.com.ph","lazada.sg","lazada.vn","lowes.ca","lowes.com","mcmaster.com","mediamarkt.de","mediamarkt.es","medline.com","mscdirect.com","napaonline.com","nofrills.ca","peoplefinders.com","platt.com","publicdatausa.com","realcanadiansuperstore.ca","realestate.com.au","restaurantguru.com","searchpeoplefree.com","shopee.cl","shopee.co.id","shopee.co.th","shopee.com.br","shopee.com.co","shopee.com.mx","shopee.com.my","shopee.ph","shopee.sg","shopee.tw","shopee.vn","similarweb.com","skyscanner.co.kr","skyscanner.net","stopandshop.com","target.com","temu.com","ticketmaster.com","totalwine.com","tractorsupply.com","walmart.com.mx","wayfair.com","weismarkets.com","wizzair.com","worten.pt"]};var oo=pe.source,qt=pe.retrievedAt,vr=new Set(pe.domains);function Z(e){let t=new URL(e).hostname.toLowerCase().replace(/\.$/,"");for(let o of vr)if(t===o||t.endsWith(`.${o}`))return"premium";return"standard"}function ee(e){for(let t of["providerProduct","billingTier","billableEvent","unitType","source","policyVersion","effectiveAt"])if(!e[t]?.trim())throw new Error(`Provider rate ${t} is required`);if(!Number.isSafeInteger(e.numeratorUsdNanos)||e.numeratorUsdNanos<0)throw new Error("Provider rate numerator must be a non-negative safe integer");if(!Number.isSafeInteger(e.denominatorUnits)||e.denominatorUnits<1)throw new Error("Provider rate denominator must be a positive safe integer");if(Number.isNaN(Date.parse(e.effectiveAt)))throw new Error("Provider rate effectiveAt must be a date");return e}function me(e,t){if(ee(e),!Number.isSafeInteger(t)||t<0)throw new Error("Billable units must be a non-negative safe integer");let o=BigInt(e.numeratorUsdNanos)*BigInt(t),r=BigInt(e.denominatorUnits),n=(o*2n+r)/(r*2n);if(n>BigInt(Number.MAX_SAFE_INTEGER))throw new Error("Provider amount exceeds safe integer range");return Number(n)}function Sr(){return ee({providerProduct:"brightdata_serp_api",billingTier:"payg",billableEvent:"successful_delivery",unitType:"request",numeratorUsdNanos:15e8,denominatorUnits:1e3,source:"owner_confirmed_2026-09-24;docs/decisions/2026-08-11-preserve-questions-only-paa-while-evaluating-serp-api.md",policyVersion:"brightdata-serp-payg-2026-09-24",effectiveAt:"2026-08-11T00:00:00Z"})}function Ht(e=process.env){let t=e.BRIGHTDATA_SERP_CONTRACT_USD_PER_1000?.trim(),o=e.BRIGHTDATA_SERP_RATE_SOURCE?.trim(),r=e.BRIGHTDATA_SERP_RATE_POLICY_VERSION?.trim(),n=e.BRIGHTDATA_SERP_RATE_EFFECTIVE_AT?.trim();if(!t&&!o&&!r&&!n)return Sr();if(!t||!o||!r||!n)throw new Error("Bright Data SERP contract rate requires amount, source, policy version, and effective date");let s=Number(t);if(!Number.isFinite(s)||s<0)throw new Error("Invalid Bright Data SERP contract amount");return ee({providerProduct:"brightdata_serp_api",billingTier:"payg",billableEvent:"successful_delivery",unitType:"request",numeratorUsdNanos:Math.round(s*1e9),denominatorUnits:1e3,source:o,policyVersion:r,effectiveAt:n})}function Ar(e){return ee({providerProduct:"brightdata_browser_api",billingTier:e,billableEvent:"transferred_byte",unitType:"byte",numeratorUsdNanos:(e==="premium"?11:8)*1e9,denominatorUnits:1e9,source:`owner_confirmed_payg_2026-09-23;brightdata_domain_snapshot_${qt}`,policyVersion:"brightdata-browser-domain-tier-2026-09-23",effectiveAt:"2026-09-23T00:00:00Z"})}function so(e){return Ar(Z(e))}function Bt(){return ee({providerProduct:"apify_google_serp_scraperlink",billingTier:"store_pay_per_event",billableEvent:"serp_page",unitType:"page",numeratorUsdNanos:5e8,denominatorUnits:1e3,source:"https://apify.com/scraperlink/google-search-results-serp-scraper",policyVersion:"scraperlink-listing-2026-09-24",effectiveAt:"2026-09-24T00:00:00Z"})}var Er="https://api.brightdata.com/request",kr=8e3,Or="brd.superproxy.io",Ir="44445",Tr=10*1024*1024;async function Cr(e,t){let o=new URL(`http://${t.host}:${t.port}`);o.username=t.username,o.password=t.password;let r=new Pr(o);try{return await new Promise((n,s)=>{let a=Rr(e,{agent:r,headers:{"x-unblock-data-format":"parsed"},rejectUnauthorized:!1,signal:t.signal},i=>{let u=[],l=0;i.on("data",c=>{let p=Buffer.isBuffer(c)?c:Buffer.from(c);if(l+=p.length,l>Tr){a.destroy(new Error("Bright Data SERP response exceeded the maximum size"));return}u.push(p)}),i.on("end",()=>n({ok:(i.statusCode??500)>=200&&(i.statusCode??500)<300,status:i.statusCode??500,rawText:Buffer.concat(u).toString("utf8"),providerRequestId:typeof i.headers["x-response-id"]=="string"?i.headers["x-response-id"]:typeof i.headers["x-request-id"]=="string"?i.headers["x-request-id"]:null}))});a.on("error",()=>s(new Error("Bright Data native SERP request failed"))),a.end()})}finally{r.destroy()}}function ye(e=process.env){let t=!!(e.BRIGHTDATA_SERP_PROXY_USERNAME?.trim()&&e.BRIGHTDATA_SERP_PROXY_PASSWORD?.trim()),o=!!((e.BRIGHTDATA_SERP_API_KEY||e.BRIGHTDATA_API_KEY)?.trim()&&e.BRIGHTDATA_SERP_ZONE?.trim());return t||o}function m(e){return e!=null&&typeof e=="object"&&!Array.isArray(e)?e:null}function N(e){return Array.isArray(e)?e:[]}function Be(e){return typeof e=="string"&&e.trim()?e.trim():null}function R(e){return typeof e=="number"&&Number.isFinite(e)?e:typeof e=="string"&&e.trim()&&Number.isFinite(Number(e))?Number(e):null}function g(e,t){for(let o of t){let r=Be(e[o]);if(r)return r}return null}function J(e){try{return new URL(e).hostname.replace(/^www\./,"")}catch{return""}}function Qt(e){return{rawUrl:e,resolvedUrl:e,linkType:"plain",resolutionStatus:"not_needed"}}function xr(e){let t=e;if(typeof t=="string")try{t=JSON.parse(t)}catch{throw new Error("Bright Data SERP API returned invalid JSON")}let o=m(t);if(!o)throw new Error("Bright Data SERP API returned an invalid payload");if(typeof o.body=="string")try{let n=JSON.parse(o.body);if(m(n))return m(n)}catch{}let r=m(o.body);return r?m(r.result)??m(r.data)??r:m(o.result)??m(o.data)??o}function j(e,t){for(let r of t){let n=N(e[r]);if(n.length)return n}let o=m(e.results);if(o)for(let r of t){let n=N(o[r]);if(n.length)return n}return[]}function K(e,t){let o=[e,m(e.results)].filter(r=>!!r);for(let r of o)for(let n of t)if(Object.prototype.hasOwnProperty.call(r,n))return{observed:!0,rawShapeValid:Array.isArray(r[n]),rawCount:N(r[n]).length};return{observed:!1,rawShapeValid:!1,rawCount:0}}function He(e,t){let o=[e,m(e.results)].filter(r=>!!r);for(let r of o)for(let n of t){if(!Object.prototype.hasOwnProperty.call(r,n))continue;let s=m(r[n]);return{observed:!0,rawShapeValid:s!==null,rawCount:s&&Object.keys(s).length>0?1:0}}return{observed:!1,rawShapeValid:!1,rawCount:0}}function Ut(e){return j(e,["organic","organic_results","organicResults"]).flatMap((t,o)=>{let r=m(t);if(!r)return[];let n=g(r,["link","url","href"]),s=g(r,["title","name"]);if(!n||!s)return[];let a=m(r.rating)??m(r.inline_rating);return[{position:Math.max(1,Math.trunc(R(r.global_rank)??R(r.rank)??R(r.position)??o+1)),title:s,url:n,...Qt(n),domain:J(n),cite:g(r,["display_link","displayLink","cite"]),snippet:g(r,["description","snippet","text"]),isRedditStyle:J(n)==="reddit.com",inlineRating:a?{value:String(a.value??a.rating??""),count:String(a.count??a.reviews??"")}:null}]}).sort((t,o)=>t.position-o.position)}function Ft(e){return j(e,["snack_pack","local","local_results","local_pack","localPack"]).flatMap((t,o)=>{let r=m(t);if(!r)return[];let n=g(r,["title","name"]);if(!n)return[];let s=N(r.extensions??r.metadata).map(Be).filter(a=>!!a);return[{position:Math.max(1,Math.trunc(R(r.global_rank)??R(r.rank)??R(r.position)??o+1)),name:n,cid:g(r,["cid"]),rating:g(r,["rating"])??(R(r.rating)!=null?String(R(r.rating)):null),reviewCount:g(r,["reviews_cnt","reviews","review_count","reviewCount"])??(R(r.reviews_cnt)!=null?String(R(r.reviews_cnt)):R(r.reviews)!=null?String(R(r.reviews)):null),metadata:s,websiteUrl:g(r,["website","website_url","link"]),directionsUrl:g(r,["directions","directions_url"])}]})}function Wt(e){let t=m(e);return N(t?.citations??t?.sources??t?.references).flatMap(o=>{let r=m(o);if(!r)return[];let n=g(r,["link","url","href"]);return n?[{text:g(r,["title","text","name"])??J(n),href:n,...Qt(n)}]:[]})}function he(e,t=0){if(t>4)return[];if(Array.isArray(e))return e.flatMap(s=>he(s,t+1));let o=m(e);if(!o)return[];let r=g(o,["snippet","text","answer"]),n=he(o.texts??o.list??o.items,t+1);return[...r?[r]:[],...n]}function Gt(e,t){if(!t)return{detected:!1,text:null,citations:[]};let o=m(e.ai_overview)??m(e.aiOverview)??m(e.overview);if(!o)return{detected:!1,text:null,citations:[]};let r=g(o,["text","answer","description"])??(he(o.texts).join(`
|
|
2
|
+
|
|
3
|
+
`).slice(0,5e4)||null),n=Wt(o);return{detected:!!(r||n.length),text:r,citations:n}}function Lt(e,t){if(!t)return{detected:!1,text:null,citations:[]};let o=m(e.ai_mode)??m(e.aiMode);if(!o)return{detected:!1,text:null,citations:[]};let r=g(o,["text","answer","description"])??(he(o.texts).join(`
|
|
4
|
+
|
|
5
|
+
`).slice(0,5e4)||null),n=Wt(o);return{detected:!!(r||n.length),text:r,citations:n}}function Vt(e,t){return t?[...j(e,["videos","video_results"]),...j(e,["short_videos","shorts"])].flatMap(o=>{let r=m(o);if(!r)return[];let n=g(r,["link","url"]),s=g(r,["title","name"]);return!n||!s?[]:[{type:"video",title:s,channel:g(r,["channel","source"])??"",platform:J(n),duration:g(r,["duration"])??"",url:n}]}):[]}function $t(e,t){if(!t)return[];let o=m(e.forums);return[...j(e,["discussions","forums","forum_results"]),...N(o?.items)].flatMap(r=>{let n=m(r);if(!n)return[];let s=g(n,["link","url"]),a=g(n,["title","name"]);return!s||!a?[]:[{title:a,source:g(n,["source"])??J(s),url:s}]})}function zt(e,t){return t?j(e,["what_people_are_saying","whatPeopleSaying"]).flatMap(o=>{let r=m(o);if(!r)return[];let n=g(r,["link","url"]),s=g(r,["title","name"]);if(!n||!s)return[];let a=g(r,["platform","source"])??J(n),i=a.toLowerCase();return[{type:i.includes("reddit")?"reddit":i.includes("facebook")?"facebook":i.includes("instagram")?"instagram":i.includes("tiktok")?"tiktok":i.includes("youtube")?"youtube":i.includes("news")?"news":"unknown",title:s,url:n,source:a,platform:a,popularComment:g(r,["comment","popular_comment"]),engagement:g(r,["engagement"])??"",date:g(r,["date"])??"",duration:g(r,["duration"]),authorNote:g(r,["author_note"])}]}):[]}var Mr={featuredSnippets:["featured_snippets","featured_snippet","featuredSnippet"],knowledgePanel:["knowledge","knowledge_graph","knowledgeGraph"],topStories:["top_stories","topStories"],news:["news","news_results"],images:["images","image_results"],shopping:["shopping","shopping_results"],relatedSearches:["related","related_searches","relatedSearches"],recipes:["recipes","recipe_results"],jobs:["jobs","job_results"],topAds:["top_ads"],bottomAds:["bottom_ads"],topProducts:["top_pla"],bottomProducts:["bottom_pla"],jackpotProducts:["jackpot_pla"],shortVideos:["short_videos","shorts"],localMap:["snack_pack_map"],flights:["flights"],hotels:["hotels_selection"],popularProducts:["popular_products"],weather:["weather"],currencyExchange:["currency_exchange"],latestPosts:["latest_posts"],perspectives:["perspectives"],navigation:["navigation"],chips:["chips"],spelling:["spelling"]};function ge(e,t){for(let o of[e,m(e.results)])if(o){for(let r of t)if(Object.prototype.hasOwnProperty.call(o,r))return{observed:!0,value:o[r]}}return{observed:!1,value:null}}function fe(e,t,o){return e?t==null||Array.isArray(t)&&t.length===0?"observed_absent":Array.isArray(t)?o>0?"observed_present":"incomplete":m(t)?o>0?"observed_present":Object.keys(m(t)).length===0?"observed_absent":"incomplete":"incomplete":"unknown"}function Nr(e,t){let o={},r=[["localPack",["snack_pack","local","local_results","local_pack","localPack"],t.localPack],["forums",["discussions","forums","forum_results"],t.forums],["videos",["videos","video_results","short_videos","shorts"],t.videos],["aiOverview",["ai_overview","aiOverview","overview"],t.aiOverview],["whatPeopleSaying",["what_people_are_saying","whatPeopleSaying"],t.whatPeopleSaying],["entityIds",["knowledge","knowledge_graph","knowledgeGraph"],t.entityIds]];for(let[u,l,c]of r){let p=ge(e,l);o[u]=fe(p.observed,p.value,c)}let n=ge(e,["ai_mode","aiMode"]);o.aiMode=fe(n.observed,n.value,t.aiMode);let s=ge(e,["people_also_ask","peopleAlsoAsk","related_questions","paa"]),a=N(s.value).slice(0,30).flatMap(u=>{let l=m(u);if(!l)return[];let c=g(l,["question","title"]);return c?[{question:c.slice(0,1e3),answer:g(l,["answer","snippet","text"])?.slice(0,1e4)??null,url:g(l,["link","url"])??null}]:[]});o.paaPreview=fe(s.observed,s.value,a.length);let i=[];for(let[u,l]of Object.entries(Mr)){let c=ge(e,l);if(!c.observed)continue;let p=m(c.value),w=(Array.isArray(c.value)?c.value:p&&(Array.isArray(p.items)||Array.isArray(p.posts))?[...N(p.items),...N(p.posts)]:[c.value]).slice(0,20).flatMap(S=>{let y=m(S);if(!y)return[];let T=g(y,["title","name","heading"]),k=g(y,["link","url","website"]),D=g(y,["description","snippet","text","summary","story"]);return!T&&!k&&!D?[]:[{title:T?.slice(0,1e3)??null,url:k,description:D?.slice(0,1e4)??null,position:R(y.global_rank)??R(y.rank)??R(y.position)??null}]});o[u]=fe(c.observed,c.value,w.length);let A=Object.fromEntries(Object.entries(p??{}).slice(0,30).flatMap(([S,y])=>/^[a-zA-Z0-9_]{1,80}$/.test(S)&&(y===null||["string","number","boolean"].includes(typeof y))?[[S,typeof y=="string"?y.slice(0,2e3):y]]:[]));i.push({name:u,attributes:A,items:w})}return{status:o,paaPreview:a,additional:i}}function Kt(e,t){let o=new Set,r=new Set(t.map(a=>a.cid).filter(a=>!!a)),n=new Set,s=(a,i=0)=>{if(i>6||a==null)return;if(Array.isArray(a)){for(let l of a)s(l,i+1);return}let u=m(a);if(u)for(let[l,c]of Object.entries(u)){let p=l.toLowerCase().replace(/[^a-z]/g,""),b=Be(c);b&&(p==="kgid"||p==="kgmid"||p==="mid")?o.add(b):b&&p==="cid"?r.add(b):b&&p==="gcid"?n.add(b):typeof c=="object"&&s(c,i+1)}};return s(e),{entities:[],kgIds:[...o],cids:[...r],gcids:[...n]}}function Jt(e,t){let o=e instanceof Error?e.message:String(e);return t.aborted||/timeout|timed out/i.test(o)?"timeout":/captcha|recaptcha|unusual traffic|google\.com\/sorry|blocked/i.test(o)?"captcha":/invalid (?:json|payload)/i.test(o)?"invalid_payload":"provider_error"}function jt(e){return e==="invalid_payload"?"invalid_payload":e==="captcha"?"serp_readiness":e==="timeout"||e==="provider_error"?"provider_transport":"pagination"}async function Ue(e,t=process.env,o=fetch,r=Cr,n){if(e.pages!==1)throw new Error("Bright Data SERP supports one page only; use Apify for two pages");let s=(t.BRIGHTDATA_SERP_API_KEY||t.BRIGHTDATA_API_KEY)?.trim(),a=t.BRIGHTDATA_SERP_ZONE?.trim(),i=t.BRIGHTDATA_SERP_PROXY_USERNAME?.trim(),u=t.BRIGHTDATA_SERP_PROXY_PASSWORD?.trim(),l=!!(i&&u);if(!l&&(!s||!a))throw new Error("Direct SERP API is not configured");let c=Date.now(),p=At(),b=null,w=e.location?Ot(e.location):null;if(e.location&&!w?.canonical)throw new L("query_integrity","location_unresolved",`SERP location could not be resolved without changing the query (candidates: ${w?.ambiguousCandidates??0})`);let A=w?.canonical?kt(w.canonical):null,S=AbortSignal.timeout(e.includeAllSerpFeatures?33e3:kr),y=e.signal?AbortSignal.any([e.signal,S]):S,T=1,k=[],D=[],We=0,Ye=async d=>{let O=Date.now(),_=l?"brightdata_serp_native_proxy":"brightdata_serp_rest",I=null,q=null,Q=!1,W=!1,G=null,Y=null;if(p?.operationRunId&&(d===1||b))try{I=await $({runId:p.operationRunId,attemptNumber:n?n.attemptNumber+d-1:d,method:_,provider:"brightdata",...d===2?{parentAttemptId:b,edgeKind:"next_phase"}:n?{parentAttemptId:n.parentAttemptId,edgeKind:"provider_fallback",retryKind:"provider_fallback"}:{},normalizedFlags:{page:d,device:e.device,hasLocation:!!e.location}}),d===1&&(b=I),q=(await B({attemptId:I,provider:"brightdata",providerProduct:"brightdata_serp_api",accountScope:p.operationAccountScope??"production-default",method:_,providerStatus:"started"})).id}catch(h){throw console.warn(JSON.stringify({event:"direct_serp_telemetry_start_failed",page:d,message:h instanceof Error?h.message:String(h)})),h}let P=new URLSearchParams({q:e.query,gl:e.gl,hl:e.hl,pws:"0",num:"10",...d===2?{start:"10"}:{}}),ie=dt(e.recency);ie&&P.set("tbs",ie),e.device==="mobile"&&P.set("brd_mobile","1"),(e.includeAiOverview||e.includeAllSerpFeatures)&&P.set("brd_ai_overview","2"),A&&P.set("uule",A),l&&P.set("brd_json","1"),We+=1;let x;try{if(l){let E=t.BRIGHTDATA_SERP_PROXY_HOST?.trim()||Or,X=t.BRIGHTDATA_SERP_PROXY_PORT?.trim()||Ir;x=await r(`https://www.google.com/search?${P}`,{username:i,password:u,host:E,port:X,signal:y})}else{let E=await o(Er,{method:"POST",headers:{Authorization:`Bearer ${s}`,"Content-Type":"application/json"},body:JSON.stringify({zone:a,url:`https://www.google.com/search?${P}`,format:"json",data_format:"parsed"}),signal:y});x={ok:E.ok,status:E.status,rawText:await E.text(),providerRequestId:E.headers.get("x-response-id")??E.headers.get("x-request-id")}}if(Y=`http_${x.status}`,q&&x.providerRequestId&&await U(q,x.providerRequestId),!x.ok)throw new Error(`Bright Data SERP API failed with HTTP ${x.status}`);Q=!0;let h;try{h=xr(JSON.parse(x.rawText))}catch(E){let X=E instanceof Error&&E.message!=="Unexpected end of JSON input"?E.message:"Bright Data SERP API returned invalid JSON";throw new L("invalid_payload","invalid_provider_payload",X)}let f=Ut(h);return k.push({page:d,attempted:!0,captured:f.length>0,resultCount:f.length,requestOffset:d===2?10:0,providerRequestIdObserved:!!x.providerRequestId,...f.length===0?{failureCode:"empty_page"}:{}}),W=!0,h}catch(h){G=h;let f=Jt(h,y);if(k.some(X=>X.page===d)||k.push({page:d,attempted:!0,captured:!1,resultCount:0,requestOffset:d===2?10:0,providerRequestIdObserved:!1,failureCode:f}),d===2)throw h;let E=h instanceof L?h.stage:jt(f);throw h instanceof L?h:new L(E,f,h instanceof Error?h.message:String(h))}finally{let h=null;if(q&&Q)try{let f=Ht(t);await F({receiptId:q,costState:"measured",units:1,unitType:"request",amountUsdNanos:me(f,1),rateSource:f.source,providerProduct:f.providerProduct,billingTier:f.billingTier,billableEvent:f.billableEvent,measurementSource:"provider_http_delivery",rateNumerator:f.numeratorUsdNanos,rateDenominator:f.denominatorUnits,ratePolicyVersion:f.policyVersion,rateEffectiveAt:f.effectiveAt,providerStatus:Y??"delivered",providerDurationMs:Date.now()-O})}catch(f){console.warn(JSON.stringify({event:"direct_serp_receipt_finalize_failed",page:d,message:f instanceof Error?f.message:String(f)})),h=f}if(I&&await z({attemptId:I,status:W?"succeeded":y.aborted?"timed_out":"failed",errorCode:G?Jt(G,y):null,durationMs:Date.now()-O}).catch(f=>console.warn(JSON.stringify({event:"direct_serp_attempt_finish_failed",page:d,message:f instanceof Error?f.message:String(f)}))),h)throw h}};try{D.push({page:1,payload:await Ye(1)})}catch(d){throw d}let Xe=null;if(T===2)try{D.push({page:2,payload:await Ye(2)})}catch(d){Xe=d}let dr=Date.now()-c,pr=Date.now(),v=D[0].payload,mr=D.flatMap(({page:d,payload:O})=>Ut(O).map(_=>({..._,position:d===2&&_.position<=10?_.position+10:_.position,sourcePage:d}))).filter((d,O,_)=>_.findIndex(I=>xe(I.url)===xe(d.url))===O),Ze=e.includeLocalPack||e.includeAllSerpFeatures,ve=Ze?Ft(v):[],Se=K(v,["snack_pack","local","local_results","local_pack","localPack"]),gr=V({requested:Ze,moduleObserved:Se.observed,rawCount:Se.rawCount,mappedCount:ve.length,rawShapeValid:Se.rawShapeValid}),et=e.includeAiOverview||e.includeAllSerpFeatures,tt=Gt(v,et),rt=e.includeForums||e.includeAllSerpFeatures,ot=$t(v,rt),nt=e.includeVideos||e.includeAllSerpFeatures,at=Vt(v,nt),st=e.includeWhatPeopleSaying||e.includeAllSerpFeatures,it=zt(v,st),Ae=m(v.forums),Re=Ae&&Array.isArray(Ae.items)?{observed:!0,rawCount:Ae.items.length,rawShapeValid:!0}:K(v,["discussions","forums","forum_results"]),Pe=K(v,["videos","video_results","short_videos","shorts"]),lt={localPack:gr,forums:V({requested:rt,moduleObserved:Re.observed,rawCount:Re.rawCount,mappedCount:ot.length,rawShapeValid:Re.rawShapeValid}),videos:V({requested:nt,moduleObserved:Pe.observed,rawCount:Pe.rawCount,mappedCount:at.length,rawShapeValid:Pe.rawShapeValid}),aiOverview:V({requested:et,moduleObserved:He(v,["ai_overview","aiOverview","overview"]).observed,rawCount:He(v,["ai_overview","aiOverview","overview"]).rawCount,mappedCount:tt.detected?1:0,rawShapeValid:He(v,["ai_overview","aiOverview","overview"]).rawShapeValid}),whatPeopleSaying:V({requested:st,moduleObserved:K(v,["what_people_are_saying","whatPeopleSaying"]).observed,rawCount:K(v,["what_people_are_saying","whatPeopleSaying"]).rawCount,mappedCount:it.length,rawShapeValid:K(v,["what_people_are_saying","whatPeopleSaying"]).rawShapeValid})},fr=ce(1,k),hr=k.find(d=>d.page===1)?.resultCount??0,ae=k.find(d=>d.page===2)?.resultCount??0,yr=de({query:e.query,finalQuery:e.query,canonicalLocation:w?.canonical??null,locationResolutionSource:w?.source??"not_requested",locationAmbiguousCandidates:w?.ambiguousCandidates??0,uule:A,outboundProviderMethod:l?"brightdata_serp_native_proxy":"brightdata_serp_rest"}),se=Xe?jt(k.find(d=>d.page===2)?.failureCode??"provider_error"):Object.values(lt).some(d=>d==="incomplete"||d==="unknown")?"optional_feature_parsing":null,br={pagination:{...fr,providerRequestCount:We},features:lt,queryIntegrity:yr,failureStage:se},ut=e.includeAllSerpFeatures?(()=>{let d=D.map(({page:O,payload:_})=>{let I=Ft(_),q=$t(_,!0),Q=Vt(_,!0),W=zt(_,!0),G=Gt(_,!0),Y=Lt(_,!0),P=Kt(_,I),ie=Nr(_,{localPack:I.length,forums:q.length,videos:Q.length,aiOverview:G.detected?1:0,aiMode:Y.detected?1:0,whatPeopleSaying:W.length,entityIds:P.entities.length+P.kgIds.length+P.cids.length+P.gcids.length});return{page:O,...ie,localPack:I,forums:q,videos:Q,whatPeopleSaying:W,aiOverview:G,aiMode:Y,entityIds:P}});return{status:d[0].status,paaPreview:d.flatMap(O=>O.paaPreview),additional:d.flatMap(O=>O.additional),pages:d}})():void 0,_r=new Date().toISOString(),ct=Date.now()-c;return{seed:e.query,location:e.location??null,extractedAt:_r,diagnostics:{phaseTimings:{providerConnectMs:dr,serpParseMs:Date.now()-pr,totalServerMs:ct},pagination:{requestedPages:T,capturedPages:T===2&&ae>0?2:1,page2Status:T===2?ae>0?"captured":"failed":"not_requested",page1OrganicCount:hr,page2OrganicCount:ae,...T===2&&ae===0?{failureCode:k.find(d=>d.page===2)?.failureCode==="captcha"?"captcha":k.find(d=>d.page===2)?.failureCode==="timeout"?"timeout":"empty_page"}:{}},completionStatus:"serp_only",problem:null,resultQuality:se?"partial":"complete",degradedResult:!!se,degradationReasons:[],retryRecommended:!!se,serpCompleteness:br},totalQuestions:0,surface:"web",aiOverview:tt,aiMode:Lt(v,e.includeAllSerpFeatures),...ut?{fullSerpFeatures:ut}:{},whatPeopleSaying:it,tree:[],flat:[],videos:at,forums:ot,organicResults:mr,localPack:ve,entityIds:Kt(v,ve),stats:{seed:e.query,totalQuestions:0,maxDepthReached:0,durationMs:ct,errorCount:0}}}var Dr="scraperlink~google-search-results-serp-scraper",te="https://api.apify.com/v2";var C=class extends Error{constructor(o,r=null,n=!1,s=null){super(o);this.runId=r;this.retrySafe=n;this.terminalStatus=s;this.name="ApifyGoogleRunError"}runId;retrySafe;terminalStatus};function re(e){return e!==null&&typeof e=="object"&&!Array.isArray(e)?e:null}function oe(e){return typeof e=="string"&&e.trim()?e.trim():null}function qr(e){try{return["http:","https:"].includes(new URL(e).protocol)}catch{return!1}}function Hr(e){if(typeof e!="string")return e;try{return JSON.parse(e)}catch{throw new Error("Apify searchResults is not valid JSON")}}function Yt(e,t){if(!Array.isArray(e)||e.length===0)throw new Error("Apify dataset contains no search results");let o=[];for(let[s,a]of e.entries()){let i=re(a);if(!i)throw new Error("Apify dataset contains an invalid row");let u=Hr(i.searchResults??i),l=re(u),c=Array.isArray(u)?u:Array.isArray(l?.results)?l.results:Array.isArray(l?.organicResults)?l.organicResults:Array.isArray(l?.organic)?l.organic:l&&(l.position!==void 0||l.title!==void 0)?[l]:null;if(!c)throw new Error("Apify dataset search result shape is unsupported");let p=Number(i.page_number??l?.page_number??s+1);if(!Number.isInteger(p)||p<1||p>t)throw new Error("Apify dataset has an unexpected page number");o.push(...c.map(b=>({row:b,page:p})))}let r=o.map(({row:s,page:a},i)=>{let u=re(s),l=u?.position,c=oe(u?.title),p=oe(u?.url??u?.link);if(!u||!c||!p||!qr(p)||l!==void 0&&(!Number.isInteger(l)||l<1))throw new Error(`Apify search result ${i+1} is missing a valid position, URL, or title`);let b=oe(u.description??u.snippet),w=typeof l=="number"?l:i%10+1;return{position:a>1&&w<=10?w+(a-1)*10:w,url:p,title:c,description:b}});if(r.length>t*10)throw new Error("Apify returned more results than requested");let n={page1:r.filter(s=>s.position<=10).length,page2:r.filter(s=>s.position>10).length};if(n.page1===0)throw new Error("Apify did not return the first result page");return{results:r,pageResultCounts:n}}async function be(e,t,o,r){let n=await o(e,{headers:{Authorization:`Bearer ${t}`},signal:r});if(!n.ok)throw new Error(`Apify API returned HTTP ${n.status}`);return n.json()}function Fe(e){let t=re(re(e)?.data),o=oe(t?.id),r=oe(t?.status);if(!o||!r)throw new Error("Apify returned an invalid run receipt");return{id:o,status:r}}async function Br(e,t=fetch){let o=performance.now(),r=AbortSignal.timeout(9e4),n=e.signal?AbortSignal.any([e.signal,r]):r,s=e.runId,a=!1,i=null;try{if(!/^[A-Za-z0-9]+$/.test(s))throw new Error("Invalid Apify run ID");let u=Fe(await be(`${te}/actor-runs/${s}?waitForFinish=30`,e.token,t,n));for(;["READY","RUNNING","TIMING-OUT","ABORTING"].includes(u.status);)u=Fe(await be(`${te}/actor-runs/${s}?waitForFinish=30`,e.token,t,n));if(a=!0,i=u.status,u.status!=="SUCCEEDED")throw new Error(`Apify run ended with status ${u.status}`);let l=e.pages??1,c=await be(`${te}/actor-runs/${s}/dataset/items?format=json`,e.token,t,n);return{query:e.query,...Yt(c,l),requestedPages:l,runId:s,elapsedMs:Math.round(performance.now()-o)}}catch(u){throw new C(u instanceof Error?u.message:String(u),s,a,i)}}async function Xt(e,t,o=fetch){let r=performance.now(),n=e.query.trim(),s=e.pages??1;if(!n||n.includes(`
|
|
6
|
+
`))throw new Error("Apify search requires one nonempty query");if(s!==1&&s!==2)throw new Error("Apify search supports one or two pages");let a=t.token.trim();if(!a)throw new Error("Apify API token is required");let i=t.maxChargeUsd??.01;if(!Number.isFinite(i)||i<=0)throw new Error("Invalid Apify charge cap");let u=AbortSignal.timeout(9e4),l=e.signal?AbortSignal.any([e.signal,u]):u,c=`${te}/actors/${Dr}/runs?maxTotalChargeUsd=${i}&restartOnError=false&waitForFinish=30`,p=null,b=!1,w=null;try{let A=await o(c,{method:"POST",headers:{Authorization:`Bearer ${a}`,"Content-Type":"application/json"},body:JSON.stringify({keyword:n,limit:s===2?"20":"10",page:1,proxy_location:"us",include_merged:!1}),signal:l});if(!A.ok)throw new Error(`Apify run start returned HTTP ${A.status}`);let S=Fe(await A.json());if(p=S.id,b=!["READY","RUNNING","TIMING-OUT","ABORTING"].includes(S.status),b&&(w=S.status),S.status!=="SUCCEEDED"&&S.status!=="READY"&&S.status!=="RUNNING")throw new Error(`Apify run ended with status ${S.status}`);if(S.status==="SUCCEEDED"){let T=await be(`${te}/actor-runs/${p}/dataset/items?format=json`,a,o,l);return{query:n,...Yt(T,s),requestedPages:s,runId:p,elapsedMs:Math.round(performance.now()-r)}}return{...await Br({runId:p,query:n,token:a,pages:s,signal:l},o),elapsedMs:Math.round(performance.now()-r)}}catch(A){throw A instanceof C?A:new C(A instanceof Error?A.message:String(A),p,b,w)}}var Zt="apify_google_serp",er="apify",Ur="apify_google_serp_scraperlink";async function tr(e,t,o,r){let n=Bt();await F({receiptId:e,costState:"estimated",units:t,unitType:n.unitType,amountUsdNanos:me(n,t),rateSource:n.source,providerProduct:n.providerProduct,billingTier:n.billingTier,billableEvent:n.billableEvent,measurementSource:"actor_run_succeeded",rateNumerator:n.numeratorUsdNanos,rateDenominator:n.denominatorUnits,ratePolicyVersion:n.policyVersion,rateEffectiveAt:n.effectiveAt,providerStatus:r,providerDurationMs:o})}async function rr(e){let t=performance.now(),o=await $({runId:e.operationRunId,attemptNumber:e.attemptNumber,method:Zt,provider:er,parentAttemptId:e.parentAttemptId,edgeKind:e.edgeKind,retryKind:e.retryKind,normalizedFlags:{page:1,pages:e.options.pages??1,proxyLocation:"us"}}),r=await B({attemptId:o,provider:er,providerProduct:Ur,billingTier:"store_pay_per_event",accountScope:e.accountScope,method:Zt,providerStatus:"before_actor_start"}),n=null,s=null;try{return n=await Xt(e.options,{token:e.token},e.fetchImpl),await U(r.id,n.runId),await tr(r.id,n.requestedPages,n.elapsedMs,"succeeded_estimated"),{...n,attemptId:o}}catch(a){throw s=a,n?Object.assign(new C("Apify cost tracking failed after run completion",n.runId),{attemptId:o}):(a instanceof C&&a.runId&&(await U(r.id,a.runId),a.terminalStatus==="SUCCEEDED"?await tr(r.id,e.options.pages??1,Math.round(performance.now()-t),"succeeded_dataset_unavailable"):await Ct({receiptId:r.id,units:e.options.pages??1,unitType:"page",providerStatus:"run_outcome_or_cost_pending"})),Object.assign(a instanceof Error?a:new Error(String(a)),{attemptId:o}))}finally{await z({attemptId:o,status:s?e.options.signal?.aborted?"timed_out":"failed":"succeeded",errorCode:s?"apify_search_failed":null,durationMs:Math.round(performance.now()-t)})}}function or(e,t=e.requestedPages){let o=e.results.map(a=>{let i=new URL(a.url).hostname.replace(/^www\./,"");return{position:a.position,title:a.title,url:a.url,rawUrl:a.url,resolvedUrl:a.url,linkType:"plain",resolutionStatus:"not_needed",domain:i,cite:null,snippet:a.description,isRedditStyle:i==="reddit.com",inlineRating:null}}),r=t===2&&e.pageResultCounts.page2>0?2:1,n=t===2&&r===1,s=ce(t,[{page:1,attempted:!0,captured:!0,resultCount:e.pageResultCounts.page1,requestOffset:0,providerRequestIdObserved:!0},...t===2?[{page:2,attempted:!0,captured:!n,resultCount:e.pageResultCounts.page2,requestOffset:10,providerRequestIdObserved:!0,...n?{failureCode:"empty_page"}:{}}]:[]]);return{seed:e.query,location:null,extractedAt:new Date().toISOString(),diagnostics:{phaseTimings:{providerConnectMs:e.elapsedMs,totalServerMs:e.elapsedMs},pagination:{requestedPages:t,capturedPages:r,page2Status:t===2?n?"failed":"captured":"not_requested",page1OrganicCount:e.pageResultCounts.page1,page2OrganicCount:e.pageResultCounts.page2},completionStatus:"serp_only",problem:null,resultQuality:n?"partial":"complete",degradedResult:n,degradationReasons:n?["page_two_missing"]:[],retryRecommended:!1,serpCompleteness:{pagination:s,features:{localPack:"not_requested",forums:"not_requested",videos:"not_requested",aiOverview:"not_requested",whatPeopleSaying:"not_requested"},queryIntegrity:de({query:e.query,finalQuery:e.query,locationResolutionSource:"not_requested",outboundProviderMethod:"default_search"}),failureStage:null}},totalQuestions:0,surface:"web",aiOverview:{detected:!1,text:null,citations:[],shareUrl:null},aiMode:{detected:!1,text:null,citations:[]},whatPeopleSaying:[],tree:[],flat:[],videos:[],forums:[],organicResults:o,localPack:[],entityIds:{entities:[],kgIds:[],cids:[],gcids:[]},stats:{seed:e.query,totalQuestions:0,maxDepthReached:0,durationMs:e.elapsedMs,errorCount:0}}}var Fr=12e3;function ar(e){return e.recency?H():e.pages===2?!!e.token||H():e.mode==="full"?ye()||H():!!e.token||ye()||H()}async function Ge(e){if(e.recency)return ne(e);if(e.mode==="full")return Gr(e);if(!e.token)return nr(e);let t=performance.now(),o;for(let r=1;r<=2;r++){let n=AbortSignal.any([e.signal,AbortSignal.timeout(Fr)]),s=e.pages===2&&r===2?1:e.pages;try{let a=await rr({options:{query:e.query,pages:s,signal:n},token:e.token,operationRunId:e.operationRunId,accountScope:e.accountScope,attemptNumber:r,...o?{parentAttemptId:o,edgeKind:"same_method_retry",retryKind:"same_method"}:{}});console.info(JSON.stringify({event:"serp_provider_succeeded",provider:"apify",operationRunId:e.operationRunId,attemptNumber:r,durationMs:a.elapsedMs,resultCount:a.results.length,requestedPages:e.pages,providerRequestedPages:a.requestedPages,deliveredPages:a.pageResultCounts.page2>0?2:1,runId:a.runId}));let i=or(a,e.pages);return i.stats.durationMs=Math.round(performance.now()-t),i.diagnostics.phaseTimings&&(i.diagnostics.phaseTimings.totalServerMs=i.stats.durationMs),{result:i,mode:"light",deliveryProvider:"apify",deliveredPages:a.pageResultCounts.page2>0?2:1}}catch(a){o=a&&typeof a=="object"&&"attemptId"in a?String(a.attemptId):void 0;let i=a instanceof C?a:null;if(console.warn(JSON.stringify({event:"serp_provider_failed",provider:"apify",operationRunId:e.operationRunId,attemptNumber:r,runId:i?.runId??null,requestedPages:e.pages,attemptPages:s,retrySafe:i?.retrySafe===!0,error:a instanceof Error?a.message:String(a)})),e.signal.aborted||!o||i?.retrySafe!==!0)throw a}}return nr(e,o)}async function nr(e,t){if(e.pages===2||!ye())return ne(e);let o=performance.now(),r={query:e.query,gl:"us",hl:"en",device:"desktop",pages:1,includeAllSerpFeatures:!1,includeLocalPack:!1,includeForums:!1,includeVideos:!1,includeAiOverview:!1,includeWhatPeopleSaying:!1,signal:e.signal};console.info(JSON.stringify({event:"serp_provider_fallback",provider:"brightdata",operationRunId:e.operationRunId,attemptNumber:3,requestedPages:e.pages,providerRequestedPages:1}));try{let n=await Ue(r,process.env,fetch,void 0,t?{attemptNumber:3,parentAttemptId:t}:void 0);if(n.organicResults.length===0)throw new Error("Search returned no organic results");return n.stats.durationMs=Math.round(performance.now()-o),n.diagnostics.phaseTimings&&(n.diagnostics.phaseTimings.totalServerMs=n.stats.durationMs),console.info(JSON.stringify({event:"serp_provider_succeeded",provider:"brightdata",operationRunId:e.operationRunId,attemptNumber:3,durationMs:n.stats.durationMs,resultCount:n.organicResults.length,requestedPages:e.pages,providerRequestedPages:1,deliveredPages:1,resultQuality:n.diagnostics.resultQuality})),{result:n,mode:"light",deliveryProvider:"brightdata",deliveredPages:1}}catch(n){if(console.warn(JSON.stringify({event:"serp_provider_failed",provider:"brightdata",operationRunId:e.operationRunId,attemptNumber:3,error:n instanceof Error?n.message:String(n)})),e.signal.aborted||!H())throw n;return ne(e)}}async function ne(e){let t=await ue({query:e.query,gl:"us",hl:"en",device:"desktop",pages:e.pages,recency:e.recency,serpOnly:!0,forceManagedSerp:!0,maxAttempts:2,proxyMode:"none",maxQuestions:1,depth:1,headless:!0,format:"json",outputDir:"/tmp/paa-output-api",softDeadlineMs:Date.now()+14e4,signal:e.signal,onAttemptEvent:e.onAttemptEvent});if(t.organicResults.length===0)throw new Error("Search returned no organic results");let o=t.diagnostics.pagination?.capturedPages===2?2:1;if(e.pages===2&&o===1&&(t.diagnostics.resultQuality="partial",t.diagnostics.degradedResult=!0,t.diagnostics.degradationReasons=["page_two_missing"],t.diagnostics.retryRecommended=!0),e.mode==="full"){let r=Object.fromEntries(["localPack","forums","videos","aiOverview","aiMode","whatPeopleSaying","entityIds","paaPreview","topStories"].map(n=>[n,"unsupported"]));t.fullSerpFeatures={status:r,paaPreview:[],additional:[],pages:[]}}return{result:t,mode:e.mode==="full"?"full":"light",deliveryProvider:"brightdata",deliveredPages:o}}async function Gr(e){if(e.pages===2){let t=await Ge({...e,mode:"light",pages:2}),o=Object.fromEntries(["localPack","forums","videos","aiOverview","aiMode","whatPeopleSaying","entityIds","paaPreview","topStories"].map(n=>[n,"unsupported"]));t.result.fullSerpFeatures={status:o,paaPreview:[],additional:[],pages:[]};let r=t.result.diagnostics;return r.serpCompleteness&&(r.serpCompleteness.features={localPack:"unsupported",forums:"unsupported",videos:"unsupported",aiOverview:"unsupported",whatPeopleSaying:"unsupported"}),{...t,mode:"full"}}try{return await Lr({...e,mode:"full"})}catch(t){if(e.signal.aborted||!H())throw t;return ne({...e,mode:"full"})}}async function Lr(e){let t=performance.now(),o=e.mode==="full",r={query:e.query,gl:"us",hl:"en",device:"desktop",pages:1,...e.recency?{recency:e.recency}:{},includeAllSerpFeatures:o,includeLocalPack:o,includeForums:o,includeVideos:o,includeAiOverview:o,includeWhatPeopleSaying:o,signal:e.signal},n=await Ue(r);if(n.organicResults.length===0)throw new Error("Search returned no organic results");return n.stats.durationMs=Math.round(performance.now()-t),n.diagnostics.phaseTimings&&(n.diagnostics.phaseTimings.totalServerMs=n.stats.durationMs),{result:n,mode:o?"full":"light",deliveryProvider:"brightdata",deliveredPages:n.diagnostics.pagination?.capturedPages===2?2:1}}async function Le(e,t,o=3){for(let r=1;r<=o;r++){try{let n=await fetch(e,{method:"POST",headers:{"content-type":"application/json"},body:JSON.stringify(t),signal:AbortSignal.timeout(1e4)});if(n.ok)return;console.warn(`[webhook] attempt ${r} \u2192 ${n.status} from ${e}`)}catch(n){console.warn(`[webhook] attempt ${r} failed:`,n instanceof Error?n.message:n)}r<o&&await new Promise(n=>setTimeout(n,1e3*r*2))}console.error(`[webhook] gave up after ${o} attempts for ${e}`)}function Vr(e){return e instanceof Error?e.message:String(e)}function $r(e,t){return e instanceof DOMException&&(e.name==="TimeoutError"||e.name==="AbortError")?!0:/timeout|timed out|Timeout \d+ms exceeded|deadline/i.test(t)}function zr(e){return/captcha|recaptcha|unusual traffic|google\.com\/sorry|blocked/i.test(e)}function Kr(e){return/ERR_TUNNEL_CONNECTION_FAILED|ERR_PROXY_CONNECTION_FAILED|ERR_SOCKS_CONNECTION_FAILED|tunnel connection failed|proxy connection failed|transport error: proxy/i.test(e)}function Jr(e){return/proxy unavailable|proxy_unavailable|connection_test_failed|did not return a proxy id|configured fallback/i.test(e)}function jr(e){return/(?:spending|billing|organization|account).{0,80}(?:cap|limit|blocked|disabled)|(?:quota|capacity).{0,40}(?:reached|exceeded)|https?:\/\/\S*dashboard/i.test(e)}function Qr(e){return/browser (?:has been )?closed|context (?:has been )?closed|target page, context or browser has been closed|session closed|page has been closed/i.test(e)}function Ve(e){let t=Vr(e);return e instanceof ht?{error_code:"request_aborted",error_type:"request_aborted",message:M("request_aborted"),retryable:!0,httpStatus:408,terminalStatus:"cancelled"}:e instanceof ft||zr(t)?{error_code:"captcha_exhausted",error_type:"captcha",message:M("captcha_exhausted"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:Dt(e,t)?{error_code:"vendor_unavailable",error_type:"service_unavailable",message:M("vendor_unavailable"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:e instanceof yt?{error_code:"location_mismatch",error_type:"location_mismatch",message:M("location_mismatch"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:Kr(t)?{error_code:"proxy_tunnel_failed",error_type:"connection",message:M("proxy_tunnel_failed"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:Jr(t)?{error_code:"proxy_unavailable",error_type:"connection",message:M("proxy_unavailable"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:Qr(t)?{error_code:"browser_session_interrupted",error_type:"extraction",message:M("browser_session_interrupted"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:$r(e,t)?{error_code:"harvest_timeout",error_type:"timeout",message:M("harvest_timeout"),retryable:!0,httpStatus:504,terminalStatus:"failed"}:jr(t)?{error_code:"service_unavailable",error_type:"service_unavailable",message:Ee,retryable:!0,httpStatus:503,terminalStatus:"failed"}:{error_code:"extraction_failed",error_type:"extraction",message:Ee,retryable:!1,httpStatus:500,terminalStatus:"failed"}}function $e(e){return JSON.stringify({error_code:e.error_code,error_type:e.error_type,message:e.message,retryable:e.retryable})}function _e(e,t={}){let o=gt({errorCode:e.error_code,errorType:e.error_type,retryable:e.retryable,retryAfterSeconds:t.retryAfterSeconds,chargeStatus:t.chargeStatus,details:t.details});return{error:o.message,...o}}function ze(e,t={}){let{error:o,...r}=_e(e,t);return r}function qo(e,t={}){let o=Ve(e);return{body:_e(o,t),status:o.httpStatus}}import{createHash as Wr}from"crypto";var Yr=1e9;function Xr(e){return Wr("sha256").update(JSON.stringify(e)).digest("hex")}function we(e){return`harvest:${e}`}async function Lo(e){let t=we(e.jobId);try{return(await It({id:t,userId:e.userId,ownerScope:`user:${e.userId}`,tool:e.tool,normalizedFlags:e.normalizedFlags,idempotencyKey:`job:${e.jobId}`,requestFingerprint:Xr(e.normalizedFlags),jobId:e.jobId,requestId:e.requestId??null,deploymentCommit:process.env.VERCEL_GIT_COMMIT_SHA?.trim()||process.env.GIT_COMMIT_SHA?.trim()||null})).id}catch(o){return console.warn(JSON.stringify({event:"operation_run_start_failed",job_id:e.jobId,tool:e.tool,message:o instanceof Error?o.message:String(o)})),null}}async function Je(e,t,o){if(e)try{await Tt({runId:e,status:t,failureClass:o})}catch(r){console.warn(JSON.stringify({event:"operation_run_finish_failed",operation_run_id:e,message:r instanceof Error?r.message:String(r)}))}}async function Vo(e,t,o,r){if(!(!e||!t))try{await xt({runId:e,billingTable:"billing_debits",billingKey:t,relation:o,amountMc:r})}catch(n){console.warn(JSON.stringify({event:"operation_billing_link_failed",operation_run_id:e,relation:o,message:n instanceof Error?n.message:String(n)}))}}function Ke(e){return e==="timeout"?"timed_out":e==="request_aborted"?"cancelled":e==="browser_session_interrupted"?"interrupted":e==="captcha"||e==="proxy_tunnel_failed"||e==="proxy_unavailable"||e==="location_mismatch"||e==="error"?"failed":"succeeded"}function Zr(e){return e==="bright_data_browser_api"?"brightdata":"kernel"}function sr(e,t,o=0){let r=new Map,n=null;return async s=>{if(e)try{if(s.type==="started"){let i=o+s.attemptNumber;if(i>1&&!n){let p=(await Mt(e))?.attempts.at(-1);n=p?.id==null?null:String(p.id)}let u=await $({runId:e,attemptNumber:i,method:s.method,provider:s.provider,parentAttemptId:i===1?null:n,edgeKind:i===1?null:o>0&&s.attemptNumber===1?"durable_resume":s.retryKind==="provider_fallback"?"provider_fallback":"same_method_retry",retryKind:i===1?null:o>0&&s.attemptNumber===1?"durable_resume":s.retryKind,retryReason:i===1?null:o>0&&s.attemptNumber===1?"durable_worker_resume":s.retryReason,retryOwner:i===1?null:o>0&&s.attemptNumber===1?"durable_reconciler":"harvest",normalizedFlags:{maxAttempts:s.maxAttempts,maxQuestions:s.maxQuestions,hasLocation:!!s.location,plannedProvider:s.provider,planReason:s.planReason??null,planTimeoutMs:s.planTimeoutMs??null},startedAt:s.startedAt}),l=await B({attemptId:u,provider:s.provider,providerProduct:s.provider==="brightdata"?"brightdata_browser_api":"kernel_browser",billingTier:s.provider==="brightdata"?Z("https://www.google.com/search"):null,accountScope:t,method:s.method,providerStatus:"started"});r.set(s.attemptNumber,{attemptId:u,receiptId:l.id,plannedProvider:s.provider,observedProvider:null,providerMismatch:!1,method:s.method,startedAtMs:Date.parse(s.startedAt),billingMode:null}),n=u;return}let a=r.get(s.attemptNumber);if(!a)throw new Error(`Missing operation attempt state for attempt ${s.attemptNumber}`);if(s.type==="provider_session_started"){let i=Zr(s.provider);if(a.observedProvider=i,a.billingMode=s.billingMode??null,i!==a.plannedProvider){a.providerMismatch=!0,await F({receiptId:a.receiptId,costState:"unavailable",providerStatus:"provider_mismatch",errorCode:"provider_mismatch",errorMessage:`Planned ${a.plannedProvider}; observed ${i}`,finalizedAt:s.observedAt});let u=i==="brightdata"?"brightdata_browser":"kernel_browser",l=await B({attemptId:a.attemptId,provider:i,providerProduct:i==="brightdata"?"brightdata_browser_api":"kernel_browser",billingTier:i==="brightdata"?Z("https://www.google.com/search"):null,accountScope:t,method:u,providerStatus:"observed_provider_mismatch"});a.receiptId=l.id,a.method=u}await U(a.receiptId,s.providerSessionId);return}if(await z({attemptId:a.attemptId,status:a.providerMismatch?"failed":Ke(s.outcome),errorCode:a.providerMismatch?"provider_mismatch":Ke(s.outcome)==="succeeded"?null:s.outcome,errorMessage:s.error,durationMs:s.durationMs,endedAt:s.completedAt}),(a.observedProvider??a.plannedProvider)==="kernel"){let i=a.billingMode==="headful",u=a.billingMode==="headless"||a.billingMode==="headful"?Et(s.durationMs,i):null;await F({receiptId:a.receiptId,costState:"estimated",units:s.durationMs/1e3,unitType:"wall_open_second",amountUsdNanos:u==null?null:Math.round(u*Yr),providerProduct:"kernel_browser",billingTier:a.billingMode,billableEvent:"provider_active_gb_second",measurementSource:"wall_open_time_proxy",rateSource:u!=null?`kernel_${a.billingMode}_usd_per_second:${i?Pt:Rt}`:null,providerStatus:a.providerMismatch?"provider_mismatch":s.outcome,providerDurationMs:s.durationMs,errorCode:a.providerMismatch?"provider_mismatch":Ke(s.outcome)==="succeeded"?null:s.outcome,errorMessage:s.error,finalizedAt:s.completedAt})}}catch(a){console.warn(JSON.stringify({event:"operation_attempt_telemetry_failed",operation_run_id:e,attempt_number:s.attemptNumber,message:a instanceof Error?a.message:String(a)}))}}}function je(e,t,o=null,r="production-default",n=0){let s=sr(o,r,n);return async a=>{if(await s(a),a.type==="started"){await bt({jobId:e,userId:t,attemptNumber:a.attemptNumber,maxAttempts:a.maxAttempts,query:a.query,location:a.location,maxQuestions:a.maxQuestions,startedAt:a.startedAt});return}if(a.type==="provider_session_started"){await _t({jobId:e,attemptNumber:a.attemptNumber,browserProvider:a.provider,providerSessionId:a.providerSessionId,observedAt:a.observedAt});return}await wt({jobId:e,attemptNumber:a.attemptNumber,outcome:a.outcome,kernelSessionId:a.kernelSessionId,questionCount:a.questionCount,durationMs:a.durationMs,error:a.error,willRetry:a.willRetry,kernelDeleteStarted:a.cleanup.kernelDeleteStarted,kernelDeleteSucceeded:a.cleanup.kernelDeleteSucceeded,kernelDeleteError:a.cleanup.kernelDeleteError,browserCloseSucceeded:a.cleanup.browserCloseSucceeded,browserCloseError:a.cleanup.browserCloseError,browserProvider:a.cleanup.provider??null,providerSessionId:a.cleanup.providerSessionId??null,providerLookupStatus:a.cleanup.providerSessionId?"pending":"not_applicable",providerErrorCode:a.cleanup.providerDisconnectObserved?"browser_session_interrupted":null,providerErrorMessage:a.cleanup.providerDisconnectMessage??null,providerReasonSource:a.cleanup.providerDisconnectObserved?"client_runtime":null,providerDisconnectObserved:a.cleanup.providerDisconnectObserved??!1,providerDisconnectMessage:a.cleanup.providerDisconnectMessage??null,debug:a.debug,completedAt:a.completedAt})}}var ir=2,Qe=0;function lr(e){if(!e||typeof e!="object")return 0;let t=e;return typeof t.totalQuestions=="number"?t.totalQuestions:Array.isArray(t.flat)?t.flat.length:0}function ur(e){return Ne.paa_base+Math.max(1,e)*Ne.paa}async function cr(e){Qe++;try{let t=typeof e.options=="string"?JSON.parse(e.options):e.options,o={value:null},r=null,n=t.serpOnly&&!t.serpIdentity,s=we(e.id),a=process.env.APIFY_API_TOKEN?.trim();if(n&&!ar({pages:t.pages===2?2:1,recency:t.recency==="week"||t.recency==="month"?t.recency:void 0,mode:t.mode,token:a}))throw new Error("Search providers are not configured");let i=await St({op:t.serpOnly?"serp":"paa",userId:Number(e.user_id),headlessSentOut:o,operationRunId:s,operationAccountScope:"production-default"},()=>n?Ge({query:t.query??e.query,pages:t.pages===2?2:1,token:a,mode:t.mode,recency:t.recency==="week"||t.recency==="month"?t.recency:void 0,operationRunId:s,accountScope:"production-default",signal:AbortSignal.timeout(pt(t.maxQuestions??30,!0,t.mode,t.pages,!!t.recency,n).serverMs),onAttemptEvent:je(e.id,e.user_id,s)}).then(l=>(r=l,l.result)):ue({...t,kernelApiKey:mt(),headless:!0,format:"json",outputDir:"/tmp/paa-output-api",onAttemptEvent:je(e.id,e.user_id,s)}));if(n&&!r)throw new Error("Search provider billing receipt is missing");await vt(e.id,i),await Je(s,"succeeded");let u=await ke(e.id,e.user_id);if(typeof t.billingHoldMc=="number"&&t.billingDebitKey){let l=t.serpOnly?n?qe(r.deliveredPages,r.deliveryProvider):De(o.value):ur(lr(i));await Ce(Number(e.user_id),t.billingDebitKey,l,t.serpOnly?"serp_refund":"paa_refund",t.serpOnly?"SERP search settlement":"PAA harvest settlement",t.serpOnly?"serp_search":"paa_harvest")}else if(!t.serpOnly&&typeof t.billingHoldMc=="number"){let l=ur(lr(i)),c=t.billingHoldMc-l;c>0?await le(e.user_id,c,"paa_refund","overestimate refund"):c<0&&await Te(e.user_id,-c,"paa",t.query??e.query)}else if(t.serpOnly&&typeof t.billingHoldMc=="number"){let l=n?qe(r.deliveredPages,r.deliveryProvider):De(o.value),c=t.billingHoldMc-l;c>0?await le(e.user_id,c,"serp_refund","headless-mode pricing settle"):c<0&&await Te(e.user_id,-c,"serp",t.query??e.query)}e.callback_url&&await Le(e.callback_url,{job_id:e.id,status:"done",result:Nt(i),attempts:Me(u)})}catch(t){console.error("[harvest/worker] failed",{jobId:e.id,error:t instanceof Error?t.message:String(t)});let o=Ve(t);await Je(we(e.id),o.error_code==="harvest_timeout"?"timed_out":o.terminalStatus,o.error_code);let r=typeof e.options=="string"?JSON.parse(e.options):e.options,n=typeof r.billingHoldMc=="number"&&r.billingHoldMc>0;await Ie(e.id,$e(o),ze(o,{chargeStatus:n?"refund_pending":"not_charged"}));let s=await ke(e.id,e.user_id);try{n&&(r.billingDebitKey?await Ce(Number(e.user_id),r.billingDebitKey,0,r.serpOnly?"serp_refund":"paa_refund","failed call",r.serpOnly?"serp_search":"paa_harvest"):await le(e.user_id,r.billingHoldMc,"refund","failed call"),await Ie(e.id,$e(o),ze(o,{chargeStatus:"refunded"})))}catch{}e.callback_url&&await Le(e.callback_url,{job_id:e.id,status:"failed",..._e(o),attempts:Me(s)})}finally{Qe--}}async function eo(){let e=await Oe();if(!e)return{claimed:!1};let t=Date.now();return await cr(e),{claimed:!0,jobId:e.id,completed:!0,durationMs:Date.now()-t}}async function sn(e){let t=[];for(let o=0;o<e.maxJobs&&!(Date.now()>=e.deadlineMs);o++){let r=await eo();if(t.push(r),!r.claimed)break}return t}function ln(){setInterval(async()=>{if(Qe>=ir)return;let e=await Oe();e&&cr(e)},2e3),console.log(`[worker] started \u2014 polling every 2s, max ${ir} concurrent`)}export{Z as a,me as b,Ar as c,so as d,ar as e,Ge as f,Ve as g,$e as h,_e as i,ze as j,qo as k,we as l,Lo as m,Je as n,Vo as o,je as p,Le as q,eo as r,sn as s,ln as t};
|