mcp-scraper 0.93.4 → 0.94.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +71 -28
  2. package/README.md +5 -5
  3. package/dist/{analytics-repository-IHOFBSUV.js → analytics-repository-A7VMA24D.js} +1 -1
  4. package/dist/bin/api-server.js +1 -1
  5. package/dist/bin/mcp-scraper-cli.js +1 -1
  6. package/dist/bin/mcp-scraper-core.js +1 -1
  7. package/dist/bin/mcp-scraper-install.js +1 -1
  8. package/dist/bin/mcp-stdio-server.js +1 -1
  9. package/dist/bin/paa-harvest.js +1 -1
  10. package/dist/{chunk-LI7WHOII.js → chunk-56C243PX.js} +1 -1
  11. package/dist/{chunk-4O2FOBZR.js → chunk-5HALFCAC.js} +1 -1
  12. package/dist/{chunk-W2T4NTCE.js → chunk-6ND5AFV5.js} +3 -3
  13. package/dist/{chunk-NLKA6SHC.js → chunk-72MWPIMR.js} +1 -1
  14. package/dist/{chunk-IJF4VXJ7.js → chunk-BRNTVOCW.js} +1 -1
  15. package/dist/{chunk-DFJT2YX6.js → chunk-C5MWTMRD.js} +4 -4
  16. package/dist/chunk-DOJMTFLB.js +1 -0
  17. package/dist/{chunk-LSFDB5XE.js → chunk-DYKSPHHR.js} +1 -1
  18. package/dist/chunk-JKHBLSXB.js +6 -0
  19. package/dist/{chunk-QTDLZTQ7.js → chunk-L552AQQY.js} +1 -1
  20. package/dist/chunk-M5QHXNFZ.js +3 -3
  21. package/dist/{chunk-2SP57VCG.js → chunk-NA76FFGD.js} +15 -15
  22. package/dist/chunk-NZSDHDX3.js +1 -0
  23. package/dist/{chunk-AFFMCO7R.js → chunk-QDP3XTES.js} +2 -2
  24. package/dist/{chunk-X7GZJU5P.js → chunk-QQ7OD7BJ.js} +1 -1
  25. package/dist/{chunk-M5VZVMPC.js → chunk-R6TD6OXQ.js} +6 -6
  26. package/dist/{chunk-IK5BG7MO.js → chunk-RJPQCPDX.js} +1 -1
  27. package/dist/chunk-RK2UKLWP.js +1 -0
  28. package/dist/chunk-SH5KB4P7.js +1 -1
  29. package/dist/{chunk-OYJ4HES6.js → chunk-VCK5BJMX.js} +187 -185
  30. package/dist/chunk-XPEJB4BZ.js +1 -0
  31. package/dist/chunk-ZGITQEFC.js +1 -0
  32. package/dist/{chunk-WJ4XFLS4.js → chunk-ZPK5LYEN.js} +105 -38
  33. package/dist/{chunk-5JKBFNYF.js → chunk-ZTEDLK3Y.js} +1 -1
  34. package/dist/{chunk-ORB4RHCK.js → chunk-ZUGTYD2I.js} +5 -5
  35. package/dist/{db-B5XJTOGN.js → db-B6RNXKEI.js} +1 -1
  36. package/dist/{extract-bundle-C3N6E6V7.js → extract-bundle-GMRZ2CW3.js} +1 -1
  37. package/dist/{gmail-service-6V5MFBYH.js → gmail-service-U7NP5IMY.js} +1 -1
  38. package/dist/index.cjs +41 -41
  39. package/dist/index.d.cts +14 -14
  40. package/dist/index.d.ts +14 -14
  41. package/dist/index.js +1 -1
  42. package/dist/{lead-list-enrichment-repository-AWZZMY2H.js → lead-list-enrichment-repository-SXYSS7ZX.js} +1 -1
  43. package/dist/{location-data-repository-6XAVI65Z.js → location-data-repository-NWWKDYG6.js} +1 -1
  44. package/dist/{operation-metering-DB5F7SXN.js → operation-metering-OBKSWJVJ.js} +1 -1
  45. package/dist/{server-HAV33FP7.js → server-4VUCH5QX.js} +537 -486
  46. package/dist/{site-extract-repository-2LFDP6FZ.js → site-extract-repository-DUBZXCRG.js} +1 -1
  47. package/dist/{stripe-event-worker-IJXA6T37.js → stripe-event-worker-2KOMIZXB.js} +1 -1
  48. package/dist/worker-5J4KF5RM.js +1 -0
  49. package/package.json +137 -17
  50. package/THIRD_PARTY_NOTICES.html +0 -203
  51. package/dist/chunk-DT2FYN6N.js +0 -1
  52. package/dist/chunk-POASCYQ7.js +0 -1
  53. package/dist/chunk-RK2VCTZI.js +0 -1
  54. package/dist/chunk-UN6FVDJQ.js +0 -1
  55. package/dist/chunk-Y6MKMSOC.js +0 -1
  56. package/dist/chunk-ZG52SKGD.js +0 -6
  57. package/dist/worker-P2BICG36.js +0 -1
package/CHANGELOG.md CHANGED
@@ -4,6 +4,44 @@ All notable changes to MCP Scraper are documented here. The format is based on [
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.94.1] - 2026-09-25
8
+
9
+ ### Changed
10
+
11
+ - Accept up to 1,000 reviews in a Maps place request across API, MCP, dashboard, and workflows. Scale browser collection time with the requested count and preserve each saved review batch through interrupted runs.
12
+ - Read the named business first during Bright Data service and area acquisition, matching the full Bright Data place path. Retry a challenged Bright Data browser once with a fresh session and meter both sessions. Keep a visible partial run and refund its hold if Google challenges both.
13
+
14
+ ### Fixed
15
+
16
+ - Treat a profile's exhausted review total as collected even when it has fewer reviews than the requested maximum. An unknown total no longer means the business has no reviews.
17
+
18
+ ## [0.94.0] - 2026-09-25
19
+
20
+ ### Added
21
+
22
+ - Recoverable Maps place runs with owner-scoped status, paginated saved reviews and images, and explicit resume tools. MCP now reads the same run after a client timeout.
23
+ - Incremental, revision-fenced Maps place checkpoints and five-minute repair of expired attempts and unsettled customer holds. Image ZIP packaging can resume from saved image URLs without reopening a browser.
24
+ - Google Maps place calls accept an `include` declaration, including `all`, to request reviews, configured services and areas served, and images within the existing review and image limits.
25
+ - Maps uses Bright Data Browser API for requested configured services and areas served, while Kernel remains the default for direct Google Maps discovery and place details.
26
+
27
+ ### Changed
28
+
29
+ - Maps place requests reserve the maximum declared base and review charge before browser work, then settle completed attempts to the base plus unique saved reviews. Stopped attempts are refunded; saved results survive a billing repair.
30
+ - Ordinary Maps searches now use state-targeted mobile browser egress when `location` names a US state, then direct egress after two failed mobile attempts. Searches without a state use direct egress; the MCP tool asks for a state without exposing proxy settings.
31
+
32
+ - Maps search and place requests now use a provider broker and report both the result provider and, for split place requests, the acquisition provider. Kernel discovery reads the Google Maps feed; Bright Data reads the visible Businesses pack or clicks More businesses for larger service-enriched searches.
33
+ - Align Bright Data Maps acquisition with the observed Google Places DOM: recognize the `udm=local` finder after More businesses, open exact `pv-` result cards, read the `#local-place-viewer` profile and its complete Services and Areas Served dialogs, and verify place identity from its `ftid` Maps link. Named service requests use Kernel's place category to reach the organic Businesses pack before Bright Data enrichment.
34
+ - Bright Data-only named place requests now begin on the organic Google SERP and use the matching knowledge panel's category to find the business in the organic Businesses results. A cold exact-name Places finder is reserved for profiles without a category, and a Google challenge is reported as a CAPTCHA rather than an unexplained missing business.
35
+ - Bright Data-only place extraction stays in the opened Google Search Places viewer to read profile details, expanded hours, declared services and areas, photos, and review cards up to the requested limits; a live run collected 30 services, 10 areas, 50 reviews, and 10 photo URLs. Maps SERP navigation preserves provider errors and allows up to two minutes for slow Browser API navigation.
36
+ - Bright Data Maps search, service acquisition, and place details can start one new browser session after the provider confirms `no_ready_cookies` with zero navigations, separately from Maps extraction retries; that recovery waits five seconds after connecting before navigation.
37
+ - Bright Data place review and gallery collection now scale their time and scroll allowances with the declared review and image limits. A default full declaration has a five-minute planning budget; the largest allowed request has nine minutes.
38
+
39
+ ### Fixed
40
+
41
+ - Attribute Maps provider cost to each browser actually used, including failed and retried Bright Data search, place, services acquisition, and startup recovery sessions. Kernel rotations and broker fallbacks are linked as retries for CEO reporting, which now shows failed-attempt spend separately. Kernel costs remain estimates; Bright Data sessions with delayed usage remain pending for reconciliation.
42
+ - Include web-app embedding requests in the paid-action inventory and scan web source during future cost-coverage checks.
43
+ - Include configured S3-compatible storage and remote media processing in the paid-action inventory, and show live Stripe account fee totals separately from MCP operation costs in the CEO report.
44
+
7
45
  ## [0.93.4] - 2026-09-24
8
46
 
9
47
  ### Fixed
@@ -16,13 +54,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
16
54
  ### Fixed
17
55
 
18
56
  - Render the CEO report headline and detail from one cost-ledger snapshot, and label cost-linked surcharge multiples as pricing targets rather than guaranteed realized margin.
19
- - Apply the owner-confirmed $1.50 per 1,000 successful Option 2 SERP requests to delivered provider receipts, with a guarded backfill for earlier successful requests. Failed attempts remain visibly unresolved until provider billing evidence is available.
57
+ - Apply the owner-confirmed $1.50 per 1,000 successful Bright Data SERP requests to delivered provider receipts, with a guarded backfill for earlier successful requests. Failed attempts remain visibly unresolved until provider billing evidence is available.
20
58
 
21
59
  ## [0.93.2] - 2026-09-24
22
60
 
23
61
  ### Fixed
24
62
 
25
- - Show Google Search and other provider costs in the CEO report with separate calculated, actual-or-measured, and unresolved receipt coverage. Failed queries now stop the report, missing Option 2 SERP rates leave an explicit delivered-but-unpriced receipt, and historical cost rows no longer imply a margin from subscription MRR.
63
+ - Show Google Search and other provider costs in the CEO report with separate calculated, actual-or-measured, and unresolved receipt coverage. Failed queries now stop the report, missing Bright Data SERP rates leave an explicit delivered-but-unpriced receipt, and historical cost rows no longer imply a margin from subscription MRR.
26
64
 
27
65
  ## [0.93.1] - 2026-09-24
28
66
 
@@ -121,7 +159,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
121
159
 
122
160
  - Retire the inactive Personal Assistant from server startup, HTTP and MCP routing, web navigation, Scheduler transitions, generated contracts, and Memory tool registration while preserving its source history, persisted data, secrets, and provider resources behind the verified archive branch.
123
161
  - Make two-page SERP capture perform a distinct page-two request, preserve page provenance and query/location intent, and report local-pack evidence as present, absent, incomplete, or unknown instead of silently claiming completeness.
124
- - Make Maps retries follow one immutable egress plan, retain location intent, close every browser/proxy attempt, and treat an already-gone Option 1 session as successful cleanup.
162
+ - Make Maps retries follow one immutable egress plan, retain location intent, close every browser/proxy attempt, and treat an already-gone Kernel session as successful cleanup.
125
163
 
126
164
  ### Fixed
127
165
 
@@ -135,13 +173,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
135
173
 
136
174
  - Add the fixture-tested operation, attempt, provider-receipt, reconciliation, and billing-link foundation needed to trace failed and retried SERP, PAA, Maps, extraction, and transcription work without treating missing provider cost as zero.
137
175
  - Add a generated MCP cost-coverage manifest and release gate that tracks every hosted input flag plus each named execution method's retry, timeout, billing, and cost-accounting owner.
138
- - Trace direct Option 2 SERP requests and Option 2/Option 1 browser attempts into the shared operation timeline, including provider identities, retry or fallback causality, immediate estimates, hourly due-gated Browser API reconciliation, and customer billing links.
139
- - Add a protected four-cell PAA benchmark runner for Option 1 and Option 2 at 20 and 40 complete questions, with a dry-run contract gate, one paid attempt per cell, sanitized provider evidence, and no automatic replacement run.
176
+ - Trace direct Bright Data SERP requests and Bright Data/Kernel browser attempts into the shared operation timeline, including provider identities, retry or fallback causality, immediate estimates, hourly due-gated Browser API reconciliation, and customer billing links.
177
+ - Add a protected four-cell PAA benchmark runner for Kernel and Bright Data at 20 and 40 complete questions, with a dry-run contract gate, one paid attempt per cell, sanitized provider evidence, and no automatic replacement run.
140
178
  - Add a once-daily due-gated purge of raw provider identifiers and verbose diagnostic errors after 30 days while preserving normalized cost facts and hashed correlation keys.
141
179
 
142
180
  ### Fixed
143
181
 
144
- - Skip Option 1 proxy resolution when the active attempt uses Option 2 Browser, removing avoidable Option 1 API traffic from Option 2-first SERP and PAA work.
182
+ - Skip Kernel proxy resolution when the active attempt uses Bright Data Browser, removing avoidable Kernel API traffic from Bright Data-first SERP and PAA work.
145
183
 
146
184
  ## [0.90.4] - 2026-09-22
147
185
 
@@ -153,21 +191,21 @@ All notable changes to MCP Scraper are documented here. The format is based on [
153
191
 
154
192
  ### Fixed
155
193
 
156
- - Route ordinary Option 2 SERP searches through the zone's native parsed proxy, returning Google results within the existing bounded deadline while preserving the REST endpoint as a configuration fallback.
194
+ - Route ordinary Bright Data SERP searches through the zone's native parsed proxy, returning Google results within the existing bounded deadline while preserving the REST endpoint as a configuration fallback.
157
195
  - Keep Browser API credentials and interactive PAA behavior independent from the native SERP transport, with one provider request and no hidden retry or browser fallback.
158
196
 
159
197
  ## [0.90.2] - 2026-09-21
160
198
 
161
199
  ### Fixed
162
200
 
163
- - Unwrap Option 2's object-valued REST response body before mapping parsed SERP fields, so successful direct searches return their organic results instead of an empty collection.
201
+ - Unwrap Bright Data's object-valued REST response body before mapping parsed SERP fields, so successful direct searches return their organic results instead of an empty collection.
164
202
 
165
203
  ## [0.90.1] - 2026-09-21
166
204
 
167
205
  ### Changed
168
206
 
169
- - Route ordinary `search_serp` calls through one parsed Option 2 SERP API request when the dedicated production zone is configured, avoiding browser startup and cleanup while leaving interactive PAA on Browser API.
170
- - Keep saved SERP identities on their existing Option 1-backed path and retain the browser path as the configuration fallback for local and unconfigured environments.
207
+ - Route ordinary `search_serp` calls through one parsed Bright Data SERP API request when the dedicated production zone is configured, avoiding browser startup and cleanup while leaving interactive PAA on Browser API.
208
+ - Keep saved SERP identities on their existing Kernel-backed path and retain the browser path as the configuration fallback for local and unconfigured environments.
171
209
 
172
210
  ### Fixed
173
211
 
@@ -210,7 +248,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
210
248
 
211
249
  ### Changed
212
250
 
213
- - Cap every Option 1 browser session at a ten-minute absolute lifetime and move expired-session reconciliation from the minute root cron to a dedicated hourly audit.
251
+ - Cap every Kernel browser session at a ten-minute absolute lifetime and move expired-session reconciliation from the minute root cron to a dedicated hourly audit.
214
252
  - Disable the inactive Personal Assistant reminder, reconciliation, and inbound cron schedules while preserving their routes and implementation.
215
253
  - Disable the inactive Personal Memory heartbeat and weekly rollup registrations in the production scheduler while preserving their implementation.
216
254
  - Pause twice-daily automatic memory optimization while preserving the workflow for deliberate use, preventing all-vault fan-out from consuming background execution capacity.
@@ -219,10 +257,10 @@ All notable changes to MCP Scraper are documented here. The format is based on [
219
257
 
220
258
  ### Fixed
221
259
 
222
- - Treat Option 1's not-found response during legacy session deletion as successful cleanup, preventing already-closed sessions from retrying forever.
260
+ - Treat Kernel's not-found response during legacy session deletion as successful cleanup, preventing already-closed sessions from retrying forever.
223
261
  - Show browser sessions awaiting provider cleanup separately in the CTO report instead of hiding them behind a non-null close timestamp.
224
262
  - Cast the analytics pruning clock before PostgreSQL interval arithmetic so the root cron no longer fails every minute while pruning scheduled occurrences.
225
- - Explicitly delete Option 1 screenshot sessions after capture, including when closing the browser connection fails.
263
+ - Explicitly delete Kernel screenshot sessions after capture, including when closing the browser connection fails.
226
264
 
227
265
  ## [0.89.7] - 2026-09-18
228
266
 
@@ -341,7 +379,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
341
379
 
342
380
  ### Fixed
343
381
 
344
- - Give each `reddit_thread` retrieval a 300-second end-to-end deadline, with two 60-second Option 1 attempts and two 60-second managed-browser backup attempts, instead of exhausting the full retry ladder in about 50 seconds. The MCP client now waits long enough to receive the endpoint's structured terminal result.
382
+ - Give each `reddit_thread` retrieval a 300-second end-to-end deadline, with two 60-second Kernel attempts and two 60-second managed-browser backup attempts, instead of exhausting the full retry ladder in about 50 seconds. The MCP client now waits long enough to receive the endpoint's structured terminal result.
345
383
 
346
384
  ## [0.88.2] - 2026-09-02
347
385
 
@@ -378,19 +416,19 @@ All notable changes to MCP Scraper are documented here. The format is based on [
378
416
 
379
417
  ### Fixed
380
418
 
381
- - Kept Option 2 telemetry lookup off the Reddit response critical path and reallocated the saved time to 17-second backup attempts, so all four provider attempts can finish before production ends the request.
419
+ - Kept Bright Data telemetry lookup off the Reddit response critical path and reallocated the saved time to 17-second backup attempts, so all four provider attempts can finish before production ends the request.
382
420
 
383
421
  ## [0.86.4] - 2026-09-02
384
422
 
385
423
  ### Fixed
386
424
 
387
- - Kept the complete two-primary, two-backup Reddit retry ladder inside the production request window by limiting Option 1 attempts to 8 seconds, Option 2 attempts to 14 seconds, and browser cleanup to 1 second.
425
+ - Kept the complete two-primary, two-backup Reddit retry ladder inside the production request window by limiting Kernel attempts to 8 seconds, Bright Data attempts to 14 seconds, and browser cleanup to 1 second.
388
426
 
389
427
  ## [0.86.3] - 2026-09-02
390
428
 
391
429
  ### Fixed
392
430
 
393
- - Applied 45-second Option 1 and 35-second Option 2 deadlines to the complete Reddit browser-attempt lifecycle, and made known-thread primary attempts find and click the target through DuckDuckGo before the residential landing.
431
+ - Applied 45-second Kernel and 35-second Bright Data deadlines to the complete Reddit browser-attempt lifecycle, and made known-thread primary attempts find and click the target through DuckDuckGo before the residential landing.
394
432
 
395
433
  ## [0.86.2] - 2026-09-02
396
434
 
@@ -539,13 +577,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
539
577
 
540
578
  ### Added
541
579
 
542
- - Added a Option 1-only Reddit workflow that searches DuckDuckGo with a `site:reddit.com` query, switches the same browser to a residential proxy before clicking the selected result, and reads modern Reddit posts plus bounded rendered-comment expansion through dedicated search, thread, and combined REST endpoints.
543
- - Added a bounded managed-browser backup for Reddit thread hydration after the primary Option 1 attempt fails or returns fewer than the semantic target, capped at 25 comments with measured bandwidth, duration, CAPTCHA, closure, and provider-cost telemetry.
580
+ - Added a Kernel-only Reddit workflow that searches DuckDuckGo with a `site:reddit.com` query, switches the same browser to a residential proxy before clicking the selected result, and reads modern Reddit posts plus bounded rendered-comment expansion through dedicated search, thread, and combined REST endpoints.
581
+ - Added a bounded managed-browser backup for Reddit thread hydration after the primary Kernel attempt fails or returns fewer than the semantic target, capped at 25 comments with measured bandwidth, duration, CAPTCHA, closure, and provider-cost telemetry.
544
582
 
545
583
  ### Changed
546
584
 
547
- - Routed the production `reddit_thread` and `reddit_trending` MCP tools through modern Reddit on Option 1 residential sessions, with DuckDuckGo site search for trend discovery; removed Google and old Reddit from their active execution path while preserving tool names, billing rates, bounded partial results, and refunds.
548
- - Cost probes now include Reddit Option 1 sessions and any managed-browser fallback bytes and cost in the same request receipt, and identify when the backup contributed to total cost.
585
+ - Routed the production `reddit_thread` and `reddit_trending` MCP tools through modern Reddit on Kernel residential sessions, with DuckDuckGo site search for trend discovery; removed Google and old Reddit from their active execution path while preserving tool names, billing rates, bounded partial results, and refunds.
586
+ - Cost probes now include Reddit Kernel sessions and any managed-browser fallback bytes and cost in the same request receipt, and identify when the backup contributed to total cost.
549
587
 
550
588
  ### Fixed
551
589
 
@@ -590,7 +628,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
590
628
 
591
629
  ### Fixed
592
630
 
593
- - Persisted per-control PAA dispatch and 0.7/1.0/1.4-second confirmation telemetry in durable checkpoints, exposed recent interaction and attempt correlation through MCP status, attached Option 2 session IDs immediately after browser launch, and finalized dangling attempt rows during lease recovery without blocking customer settlement.
631
+ - Persisted per-control PAA dispatch and 0.7/1.0/1.4-second confirmation telemetry in durable checkpoints, exposed recent interaction and attempt correlation through MCP status, attached Bright Data session IDs immediately after browser launch, and finalized dangling attempt rows during lease recovery without blocking customer settlement.
594
632
  - Prevented inline style, script, and hidden DOM text inside Google answer containers from falsely confirming that PAA answer material loaded.
595
633
  - Routed canonical `/assistant` page loads to the web app and the redacted private Assistant readiness endpoint to the main API function, preventing production 404s after the 0.79.1 launch.
596
634
 
@@ -614,7 +652,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
614
652
  - Added Scheduling as the canonical Personal Assistant setup surface, with connection readiness for Gmail, Calendar, Zoom, browser profiles, Memory, SMS, and email; exact schedule confirmation; approval and spend review; run history; and explicit watch/takeover states.
615
653
  - Added owner-scoped browser profiles that can hold multiple independently verified login bindings, while every browser schedule grant selects one exact profile, login, domain, and action set.
616
654
  - Added immutable schedule revisions, readiness receipts, append-only activation records, additive legacy schedule projection, and single-owner occurrence transition receipts so migration cannot silently infer browser authority or double-dispatch work.
617
- - Added Option 1 and private-Mac browser runtime boundaries with collision-resistant tenant namespaces, per-owner concurrency ceilings, bounded sessions, explicit and timeout cleanup, owner-qualified account deletion, and provider deletion readback.
655
+ - Added Kernel and private-Mac browser runtime boundaries with collision-resistant tenant namespaces, per-owner concurrency ceilings, bounded sessions, explicit and timeout cleanup, owner-qualified account deletion, and provider deletion readback.
618
656
  - Added an owner-controlled Personal Assistant that brings SMS/MMS, Gmail, Google Calendar, Zoom, browser work, reminders, and Memory context packets into one governed workflow with immutable plans, approval checkpoints, spend limits, and durable receipts.
619
657
  - Added Twilio number discovery, owned-number attachment, purchase and registration previews, Messaging Service readiness, signed inbound and delivery webhooks, safe MMS ingestion, deterministic opt-out handling, single and reviewed bulk messaging, and reconciliation for unknown provider outcomes.
620
658
  - Added immutable, revisioned Memory context packets with source and attachment provenance, Gmail full-message imports, MMS media metadata, lifecycle controls, and readback verification against the selected vault.
@@ -651,7 +689,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
651
689
  - Made `maxQuestions` an explicit target count rather than a traversal-depth control, with separate discovery and material-completeness diagnostics.
652
690
  - Preserved complete People Also Ask, AI Overview, and organic-result link provenance in JSON, structured MCP output, and CSV while classifying plain links and Google redirect links explicitly.
653
691
  - Resolved opaque Google `/goto` targets through bounded concurrent manual-redirect requests with active-browser interception as a fallback, without following publisher destinations and without dropping unresolved material.
654
- - Aligned the bounded PAA production-provider canary with the public `maxQuestions` contract and made Option 2 the default test provider.
692
+ - Aligned the bounded PAA production-provider canary with the public `maxQuestions` contract and made Bright Data the default test provider.
655
693
 
656
694
  ### Fixed
657
695
 
@@ -859,7 +897,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
859
897
 
860
898
  - Added portable `harvest_paa_start` and `harvest_paa_status` tools for durable long-running PAA research, with stable idempotency recovery, progress, attempt provenance, completeness, billing state, and bounded provider telemetry.
861
899
  - Added progressive PAA checkpoints that preserve and merge the best unique rows across browser retries and stale-job recovery instead of losing already captured questions when a provider session or caller is interrupted.
862
- - Added exact Option 2 browser-session identity, sanitized Session Logs enrichment, disconnect attribution, bandwidth usage telemetry, and retryable reconciliation without making provider telemetry a prerequisite for result delivery.
900
+ - Added exact Bright Data browser-session identity, sanitized Session Logs enrichment, disconnect attribution, bandwidth usage telemetry, and retryable reconciliation without making provider telemetry a prerequisite for result delivery.
863
901
 
864
902
  ### Changed
865
903
 
@@ -1592,7 +1630,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
1592
1630
 
1593
1631
  ### Changed
1594
1632
 
1595
- - PAA browser work now uses Option 1's co-located Playwright execution with stealth mode's default managed proxy and native browser metadata. Location is expressed only through Google UULE, CAPTCHA solver waiting is capped at 60 seconds, and a fresh session is allowed once only when no useful data was captured.
1633
+ - PAA browser work now uses Kernel's co-located Playwright execution with stealth mode's default managed proxy and native browser metadata. Location is expressed only through Google UULE, CAPTCHA solver waiting is capped at 60 seconds, and a fresh session is allowed once only when no useful data was captured.
1596
1634
  - PAA invocations stop browser work at 250 seconds inside the 280-second application budget, reserving 30 seconds for persistence, cleanup, and settlement. The legacy cron worker no longer claims Inngest-owned PAA jobs.
1597
1635
 
1598
1636
  ### Fixed
@@ -1828,7 +1866,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
1828
1866
 
1829
1867
  ### Changed
1830
1868
 
1831
- - `maps_search` now applies a transport ladder across its retry attempts so it can recover from Google soft-blocks instead of only retrying the same way. The first attempt is unchanged (Option 1's default stealth ISP proxy, direct navigation). Subsequent retries switch to direct egress and arrive at Google through a cross-site redirect (the combination that measurably clears blocks a cold navigation triggers); the final escalation attempt uses direct egress without the redirect and accepts any egress country. This only affects the `proxyMode: 'none'` default path and only its retries — a first-attempt success behaves exactly as before.
1869
+ - `maps_search` now applies a transport ladder across its retry attempts so it can recover from Google soft-blocks instead of only retrying the same way. The first attempt is unchanged (Kernel's default stealth ISP proxy, direct navigation). Subsequent retries switch to direct egress and arrive at Google through a cross-site redirect (the combination that measurably clears blocks a cold navigation triggers); the final escalation attempt uses direct egress without the redirect and accepts any egress country. This only affects the `proxyMode: 'none'` default path and only its retries — a first-attempt success behaves exactly as before.
1832
1870
 
1833
1871
  ## [0.32.1] - 2026-07-22
1834
1872
 
@@ -2075,7 +2113,12 @@ All notable changes to MCP Scraper are documented here. The format is based on [
2075
2113
  - Write actions remain unavailable until the account owner explicitly enables them.
2076
2114
  - Provider-specific connection data is normalized into one agent-facing contract.
2077
2115
 
2078
- [Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.0...HEAD
2116
+ [Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.94.0...HEAD
2117
+ [0.94.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.4...v0.94.0
2118
+ [0.93.4]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.3...v0.93.4
2119
+ [0.93.3]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.2...v0.93.3
2120
+ [0.93.2]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.1...v0.93.2
2121
+ [0.93.1]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.0...v0.93.1
2079
2122
  [0.93.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.92.1...v0.93.0
2080
2123
  [0.92.1]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.92.0...v0.92.1
2081
2124
  [0.92.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.91.0...v0.92.0
package/README.md CHANGED
@@ -175,7 +175,7 @@ Build the branded one-click bundle:
175
175
  npm run build:mcpb
176
176
  ```
177
177
 
178
- The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.93.4`, SHA-256 `cd176d3180aad0cdd4899bb6d2422d9eafbf2c5af30af24f9668ab6d3060e989`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
178
+ The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.94.1`, SHA-256 `4a5f8a8dc8cb7743ffca82fee5a9c981a9146e5e55e8f50ac7e028400913404d`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
179
179
 
180
180
  The MCPB install exposes every tool — web-intelligence plus all `browser_*` tools — through the one `mcp-scraper` server.
181
181
 
@@ -267,7 +267,7 @@ Check `pagination.requestedPages`, `pagination.capturedPages`, and `pagination.p
267
267
  - `instagram_profile_content` — discover Instagram profile grid content links for a handle or profile URL, optionally through a saved hosted browser `profile` for authenticated access. Returns collected post/reel/tv URLs, profile counts, type counts, shortcodes, browser details, pagination attempts, stop reason, and limitations.
268
268
  - `instagram_media_download` — extract and download one Instagram post/reel/tv URL, optionally through a saved hosted browser `profile` for authenticated access. Returns text/caption, image URL/downloads, selected video/audio MP4 tracks, optional muxed MP4 when `ffmpeg` is available, optional transcript, and browser details.
269
269
  - `maps_search` — search Google's localized local-results list for multiple business/profile candidates. Use for GMB/GBP prospect lists, competitors, categories, and anything needing more than the Google 3-pack. It opens the rendered business card, reads the profile dialog, then closes it before continuing to the next ranked card. Set `includeServices: true` to return services and areas served without collecting review cards. `maxResults` defaults to 10 and is capped at 50.
270
- - `maps_place_intel` — hydrate one known/named Google Maps business with profile details, entity IDs/CID, services, service areas, review aggregates/cards, and optional photos. Set `includeImages:true`, choose `imageScope:"owner"` or `"all"`, and tune `maxImages`; results include ownership evidence, completion state, bounded AI image blocks, and an owner-scoped ZIP manifest readable with `archive_read`.
270
+ - `maps_place_intel` — hydrate one known/named Google Maps business with profile details, entity IDs/CID, services, service areas, review aggregates/cards, and optional photos. Set `includeImages:true`, choose `imageScope:"owner"` or `"all"`, and tune `maxImages`; results include a recoverable run ID, field status, ownership evidence, and an owner-scoped ZIP readable with `archive_read`. Use `maps_place_status` to read saved results and `maps_place_resume` to explicitly retry a partial run.
271
271
  - `directory_workflow` — build city-by-city directory/prospecting datasets from Census place selection plus localized Google business searches. Use it for requests like "all cities over 100k population in Tennessee, then get 20 roofers from Maps." Supply the business category, state, and market limits; MCP Scraper manages search transport and retry behavior internally. The saved CSV includes `source_location`, `result_position`, `business_name`, `review_stars`, `review_count`, `category`, `address`, `phone`, `hours_status`, `website_url`, `directions_url`, `place_url`, `cid`, `cid_decimal`, Census population, and ZIP groups.
272
272
  - `workflow_list` — list higher-level workflow IDs plus AI-facing recipes for market analysis, ICP research, forum/review acquisition, brand design briefings, CRO audits, positioning briefs, content gaps, and AI search visibility audits.
273
273
  - `workflow_suggest` — route a high-level business goal to the right workflow/tool chain before spending credits.
@@ -351,7 +351,7 @@ The `mcp-scraper` server (and the MCPB bundle, which runs it) exposes both secti
351
351
 
352
352
  All MCP tools return `structuredContent` with the IDs, URLs, CSV paths, transcripts, browser session handles, replay paths, artifacts, recipe fields, or blueprint fields needed by the next step, plus readable text content for compatibility. Runtime `tools/list` omits output schemas so strict clients can register the complete catalog; the generated developer manifest retains every canonical output schema for validation and typed SDK generation. All tools carry MCP annotations; file-writing tools such as replay downloads and annotations state their filesystem side effects.
353
353
 
354
- The canonical tool inventory is generated at `docs/mcp-tool-manifest.generated.json`. The unified server exposes 362 tools: 236 scraper, browser, workflow, billing, and connected-service tools plus 126 durable-memory tools. Personal Assistant tools and routes are retired from active builds; their source history and persisted data remain recoverable from the archived pre-retirement branch. The scraper-side inventory includes complete Gmail selection, message, attachment, export, bulk-action, and Memory-import workflows; durable SERP, PAA, and single-page extraction starts and status; rendered site-content similarity; governed Local Sourcebook and Transparent Commons workflows; direct site-export reads; Editorial Reading Room and News Publisher templates; and production X-Ray setup, analytics, seven-model attribution, structured post-purchase surveys, reported impact, truthful view-evidence status, CRM policy and receipt, campaign, export, and scheduled-report tools. Provider setup remains absent until its authorization, ingestion, reconciliation, privacy, canary, cleanup, and deployment receipts are complete. Successful evidence-compiled Local Sourcebook revisions publish automatically to their canonical `localsourcebook.com` category profile and review URLs; administrator controls handle exceptional rejection or unpublishing. Release verification compares the exact local and hosted tool-name sets, not only the count.
354
+ The canonical tool inventory is generated at `docs/mcp-tool-manifest.generated.json`. The unified server exposes 364 tools: 238 scraper, browser, workflow, billing, and connected-service tools plus 126 durable-memory tools. Personal Assistant tools and routes are retired from active builds; their source history and persisted data remain recoverable from the archived pre-retirement branch. The scraper-side inventory includes complete Gmail selection, message, attachment, export, bulk-action, and Memory-import workflows; durable SERP, PAA, and single-page extraction starts and status; rendered site-content similarity; governed Local Sourcebook and Transparent Commons workflows; direct site-export reads; Editorial Reading Room and News Publisher templates; and production X-Ray setup, analytics, seven-model attribution, structured post-purchase surveys, reported impact, truthful view-evidence status, CRM policy and receipt, campaign, export, and scheduled-report tools. Provider setup remains absent until its authorization, ingestion, reconciliation, privacy, canary, cleanup, and deployment receipts are complete. Successful evidence-compiled Local Sourcebook revisions publish automatically to their canonical `localsourcebook.com` category profile and review URLs; administrator controls handle exceptional rejection or unpublishing. Release verification compares the exact local and hosted tool-name sets, not only the count.
355
355
 
356
356
  For contract parity, stdio and MCPB memory calls invoke the matching public tool on the hosted MCP Scraper `/mcp` endpoint. The hosted aggregate runtime owns MCP Scraper-specific billing, scheduling, credential, and in-process cutover policy; its internal `/memory/mcp-call` bridge is a fallback to the standalone Memory service, not a second customer setup path. Existing direct Memory credentials remain compatible for one release, but all new customer setup uses the root endpoint and `MCP_SCRAPER_API_KEY`.
357
357
 
@@ -366,8 +366,8 @@ The `mcp-scraper` NPX stdio server also exposes saved reports as MCP resources:
366
366
  - `MCP_SCRAPER_OUTPUT_DIR` is optional and defaults to `~/Downloads/mcp-scraper`.
367
367
  - `MCP_SCRAPER_SAVE_REPORTS=false` disables automatic Markdown report files.
368
368
  - `MCP_SCRAPER_KEY_PATH` is optional. When no API key env var is set, the server also reads `~/.mcp-scraper-key` for compatibility with older installs.
369
- - `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `mcp-scraper` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `Option 1_BROWSER_PROFILE_NAME`, `Option 1_PROFILE_NAME`.
370
- - `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `Option 1_BROWSER_PROFILE_SAVE_CHANGES`, `Option 1_PROFILE_SAVE_CHANGES`.
369
+ - `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `mcp-scraper` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `KERNEL_BROWSER_PROFILE_NAME`, `KERNEL_PROFILE_NAME`.
370
+ - `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `KERNEL_BROWSER_PROFILE_SAVE_CHANGES`, `KERNEL_PROFILE_SAVE_CHANGES`.
371
371
 
372
372
  Hosted operators can isolate authorization state in a dedicated Turso/libSQL database without changing the public MCP tool catalog or API-key authentication. The secured store uses atomic authorization-code exchange and refresh rotation, keyed secret lookup, bounded encrypted replay receipts, authority epochs, and fail-closed maintenance behavior. Production migration and rollback are controlled data moves, not ordinary mode flips; see [MCP OAuth operations](docs/operations/mcp-oauth-runbook.md). Existing client setup and reconnect behavior are unchanged in Phase 1.
373
373
 
@@ -1 +1 @@
1
- import{$ as L,$a as La,A as k,Aa as ka,B as l,Ba as la,C as m,Ca as ma,D as n,Da as na,E as o,Ea as oa,F as p,Fa as pa,G as q,Ga as qa,H as r,Ha as ra,I as s,Ia as sa,J as t,Ja as ta,K as u,Ka as ua,L as v,La as va,M as w,Ma as wa,N as x,Na as xa,O as y,Oa as ya,P as z,Pa as za,Q as A,Qa as Aa,R as B,Ra as Ba,S as C,Sa as Ca,T as D,Ta as Da,U as E,Ua as Ea,V as F,Va as Fa,W as G,Wa as Ga,X as H,Xa as Ha,Y as I,Ya as Ia,Z as J,Za as Ja,_ as K,_a as Ka,aa as M,ab as Ma,ba as N,bb as Na,ca as O,da as P,ea as Q,fa as R,ga as S,ha as T,ia as U,ja as V,ka as W,la as X,ma as Y,na as Z,oa as _,pa as $,q as a,qa as aa,r as b,ra as ba,s as c,sa as ca,t as d,ta as da,u as e,ua as ea,v as f,va as fa,w as g,wa as ga,x as h,xa as ha,y as i,ya as ia,z as j,za as ja}from"./chunk-LI7WHOII.js";import"./chunk-QTDLZTQ7.js";import"./chunk-WJ4XFLS4.js";export{P as ANALYTICS_CONTENT_SORTS,a as AnalyticsRepositoryError,B as ENGAGED_SESSION_MS,A as MAX_ENGAGED_MS,N as analyticsAcquisition,l as analyticsBusinessMetrics,O as analyticsChannelBreakdown,R as analyticsContent,T as analyticsConversions,Ma as analyticsCsvCell,V as analyticsDimensions,S as analyticsEventCounts,m as analyticsForecast,Ja as analyticsHealth,x as analyticsIdentityPromotionAllowed,w as analyticsIdentityResolutionAllowed,L as analyticsOverview,U as analyticsPaths,M as analyticsTimeseries,H as appendAnalyticsAuthoritativeOutcomeVersion,ta as archiveAnalyticsActivationDestination,Y as archiveAnalyticsCampaignLink,ga as assignAnalyticsIdentityNode,fa as backfillAnalyticsConfirmedHistory,oa as claimAnalyticsCrmImportRows,Ea as claimAnalyticsFormDeliveryJobs,c as closeAnalyticsPool,pa as completeAnalyticsCrmImportRow,Ga as completeAnalyticsFormBridgeDelivery,Fa as completeAnalyticsFormDelivery,J as consumeAnalyticsSurveyInvite,ra as createAnalyticsActivationDestination,W as createAnalyticsCampaignLink,K as createAnalyticsConversion,ma as createAnalyticsCrmImport,Na as createAnalyticsExport,_ as createAnalyticsForm,o as createAnalyticsPixel,h as createAnalyticsSite,qa as deferAnalyticsCrmImportRow,Ha as deferAnalyticsFormDelivery,n as deleteAnalyticsSite,ea as deterministicAnalyticsCrmEntityId,ja as enrichAnalyticsExistingCrmIdentity,ua as getAnalyticsActivationDestinationConnectionRef,la as getAnalyticsPersonJourney,b as getAnalyticsPool,aa as getPublicAnalyticsForm,da as identityHmac,F as ingestAnalyticsEvents,G as insertAnalyticsRevenueSetupRevision,Ia as isAnalyticsFormPlacementApproved,ia as linkAnalyticsFormIdentity,ha as linkAnalyticsIdentityInTransaction,sa as listAnalyticsActivationDestinations,xa as listAnalyticsActivationReceipts,X as listAnalyticsCampaignLinks,na as listAnalyticsCrmImports,$ as listAnalyticsForms,t as listAnalyticsHostGroups,ka as listAnalyticsPeople,p as listAnalyticsPixels,i as listAnalyticsSites,d as migrateAnalytics,C as normalizeAnalyticsPath,D as normalizeAnalyticsUrl,Q as normalizeContentOptions,e as normalizeObservedHostname,za as pollAnalyticsActivationDiagnostics,u as prepareAnalyticsLinkerIssue,I as projectAnalyticsAuthoritativeConversion,y as projectAnalyticsPixelEventConsent,Aa as queueAnalyticsActivation,Ca as queueAnalyticsFormBridgeTransaction,Ba as queueAnalyticsFormDelivery,ba as recordAnalyticsFormSubmission,v as redeemAnalyticsLinkerRecord,Ka as refreshAnalyticsDailyRollups,La as refreshAnalyticsDailyRollupsIfDue,f as requireAnalyticsAccess,g as requireAnalyticsEditor,Z as resolveAnalyticsCampaignLink,z as resolveAnalyticsConfirmedActivationIdentity,ya as retryAnalyticsActivationJob,E as sanitizeAnalyticsProperties,ca as sanitizeClickIds,va as setAnalyticsActivationReadiness,r as setAnalyticsPixelDomainState,Da as sweepAnalyticsRestrictedRetention,wa as testAnalyticsActivationDestination,j as updateAnalyticsBusinessModel,q as updateAnalyticsPixel,k as upsertAnalyticsAdSpend,s as upsertAnalyticsHostGroup};
1
+ import{$ as L,$a as La,A as k,Aa as ka,B as l,Ba as la,C as m,Ca as ma,D as n,Da as na,E as o,Ea as oa,F as p,Fa as pa,G as q,Ga as qa,H as r,Ha as ra,I as s,Ia as sa,J as t,Ja as ta,K as u,Ka as ua,L as v,La as va,M as w,Ma as wa,N as x,Na as xa,O as y,Oa as ya,P as z,Pa as za,Q as A,Qa as Aa,R as B,Ra as Ba,S as C,Sa as Ca,T as D,Ta as Da,U as E,Ua as Ea,V as F,Va as Fa,W as G,Wa as Ga,X as H,Xa as Ha,Y as I,Ya as Ia,Z as J,Za as Ja,_ as K,_a as Ka,aa as M,ab as Ma,ba as N,bb as Na,ca as O,da as P,ea as Q,fa as R,ga as S,ha as T,ia as U,ja as V,ka as W,la as X,ma as Y,na as Z,oa as _,pa as $,q as a,qa as aa,r as b,ra as ba,s as c,sa as ca,t as d,ta as da,u as e,ua as ea,v as f,va as fa,w as g,wa as ga,x as h,xa as ha,y as i,ya as ia,z as j,za as ja}from"./chunk-56C243PX.js";import"./chunk-L552AQQY.js";import"./chunk-ZPK5LYEN.js";export{P as ANALYTICS_CONTENT_SORTS,a as AnalyticsRepositoryError,B as ENGAGED_SESSION_MS,A as MAX_ENGAGED_MS,N as analyticsAcquisition,l as analyticsBusinessMetrics,O as analyticsChannelBreakdown,R as analyticsContent,T as analyticsConversions,Ma as analyticsCsvCell,V as analyticsDimensions,S as analyticsEventCounts,m as analyticsForecast,Ja as analyticsHealth,x as analyticsIdentityPromotionAllowed,w as analyticsIdentityResolutionAllowed,L as analyticsOverview,U as analyticsPaths,M as analyticsTimeseries,H as appendAnalyticsAuthoritativeOutcomeVersion,ta as archiveAnalyticsActivationDestination,Y as archiveAnalyticsCampaignLink,ga as assignAnalyticsIdentityNode,fa as backfillAnalyticsConfirmedHistory,oa as claimAnalyticsCrmImportRows,Ea as claimAnalyticsFormDeliveryJobs,c as closeAnalyticsPool,pa as completeAnalyticsCrmImportRow,Ga as completeAnalyticsFormBridgeDelivery,Fa as completeAnalyticsFormDelivery,J as consumeAnalyticsSurveyInvite,ra as createAnalyticsActivationDestination,W as createAnalyticsCampaignLink,K as createAnalyticsConversion,ma as createAnalyticsCrmImport,Na as createAnalyticsExport,_ as createAnalyticsForm,o as createAnalyticsPixel,h as createAnalyticsSite,qa as deferAnalyticsCrmImportRow,Ha as deferAnalyticsFormDelivery,n as deleteAnalyticsSite,ea as deterministicAnalyticsCrmEntityId,ja as enrichAnalyticsExistingCrmIdentity,ua as getAnalyticsActivationDestinationConnectionRef,la as getAnalyticsPersonJourney,b as getAnalyticsPool,aa as getPublicAnalyticsForm,da as identityHmac,F as ingestAnalyticsEvents,G as insertAnalyticsRevenueSetupRevision,Ia as isAnalyticsFormPlacementApproved,ia as linkAnalyticsFormIdentity,ha as linkAnalyticsIdentityInTransaction,sa as listAnalyticsActivationDestinations,xa as listAnalyticsActivationReceipts,X as listAnalyticsCampaignLinks,na as listAnalyticsCrmImports,$ as listAnalyticsForms,t as listAnalyticsHostGroups,ka as listAnalyticsPeople,p as listAnalyticsPixels,i as listAnalyticsSites,d as migrateAnalytics,C as normalizeAnalyticsPath,D as normalizeAnalyticsUrl,Q as normalizeContentOptions,e as normalizeObservedHostname,za as pollAnalyticsActivationDiagnostics,u as prepareAnalyticsLinkerIssue,I as projectAnalyticsAuthoritativeConversion,y as projectAnalyticsPixelEventConsent,Aa as queueAnalyticsActivation,Ca as queueAnalyticsFormBridgeTransaction,Ba as queueAnalyticsFormDelivery,ba as recordAnalyticsFormSubmission,v as redeemAnalyticsLinkerRecord,Ka as refreshAnalyticsDailyRollups,La as refreshAnalyticsDailyRollupsIfDue,f as requireAnalyticsAccess,g as requireAnalyticsEditor,Z as resolveAnalyticsCampaignLink,z as resolveAnalyticsConfirmedActivationIdentity,ya as retryAnalyticsActivationJob,E as sanitizeAnalyticsProperties,ca as sanitizeClickIds,va as setAnalyticsActivationReadiness,r as setAnalyticsPixelDomainState,Da as sweepAnalyticsRestrictedRetention,wa as testAnalyticsActivationDestination,j as updateAnalyticsBusinessModel,q as updateAnalyticsPixel,k as upsertAnalyticsAdSpend,s as upsertAnalyticsHostGroup};
@@ -1,3 +1,3 @@
1
1
  #!/usr/bin/env node
2
2
  import{readFileSync as s}from"fs";function c(){try{for(let r of s(".env","utf8").split(`
3
- `)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-HAV33FP7.js"),import("../worker-P2BICG36.js"),import("../db-B5XJTOGN.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-IHOFBSUV.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
3
+ `)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-4VUCH5QX.js"),import("../worker-5J4KF5RM.js"),import("../db-B6RNXKEI.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-A7VMA24D.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-W2T4NTCE.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-Y6MKMSOC.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import{a as P}from"../chunk-POASCYQ7.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
2
+ import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-6ND5AFV5.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-XPEJB4BZ.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import{a as P}from"../chunk-RK2UKLWP.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
3
3
  `)){let l=a.match(/^\s{2,}([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);if(!l){if(a.trim().length&&!/^\s{2,}/.test(a))break;continue}p[l[1]]=l[2]}return{command:n,args:i,env:p}}function j(e){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(e.env))n.push("--env",`${c}=${i}`);return n.push("--",e.command,...e.args),n}function G(e={}){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(pe(e)))n.push("--env",`${c}=${i}`);return n.push("--","npx",...A(e)),n}function O(e){if(e==="claude-code")return"claude";if(e==="claude"||D.hosts.some(n=>n.id===e))return e;throw new Error('Unknown host "'+e+'". Use: codex, claude, claude-code, claude-desktop, cursor, windsurf, cline, or user-action-only')}function ue(e){return K(e==="claude"?"claude-code":e)}function W(e,n={}){let c=O(e),i=ue(c),p="Restart the MCP client so it starts a fresh npx process.",u='MCP_SCRAPER_API_KEY="$MCP_SCRAPER_API_KEY" npx -y -p mcp-scraper@latest mcp-scraper-cli agent install claude --apply',a=`X-Ray install protocol: ${v} (${b})`;return c==="codex"?["# Codex MCP config",a,i.exactConfig,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
4
4
  `):c==="claude"?["# Claude Code command",a,i.exactConfig,"","# One-command Claude Code setup",u,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
5
5
  `):c==="claude-desktop"?["# Claude Desktop config",a,i.exactConfig,"","Desktop Extension: https://mcpscraper.dev/downloads/mcp-scraper.mcpb",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,p].join(`
@@ -1,2 +1,2 @@
1
1
  #!/usr/bin/env node
2
- import{a as e}from"../chunk-4O2FOBZR.js";import"../chunk-OYJ4HES6.js";import"../chunk-DT2FYN6N.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-UN6FVDJQ.js";import"../chunk-SH5KB4P7.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-DFJT2YX6.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-POASCYQ7.js";import"../chunk-WJ4XFLS4.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
2
+ import{a as e}from"../chunk-5HALFCAC.js";import"../chunk-VCK5BJMX.js";import"../chunk-NZSDHDX3.js";import"../chunk-W2BVJ7S2.js";import"../chunk-DOJMTFLB.js";import"../chunk-ZUGTYD2I.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-ZGITQEFC.js";import"../chunk-SH5KB4P7.js";import"../chunk-NA76FFGD.js";import"../chunk-6DTXIZY2.js";import"../chunk-C5MWTMRD.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-RK2UKLWP.js";import"../chunk-ZPK5LYEN.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
@@ -1,3 +1,3 @@
1
1
  #!/usr/bin/env node
2
- import{a as s}from"../chunk-DFJT2YX6.js";import{a as e}from"../chunk-POASCYQ7.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
2
+ import{a as s}from"../chunk-C5MWTMRD.js";import{a as e}from"../chunk-RK2UKLWP.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
3
3
  `)),process.exit(0));process.stdout.write(s({version:e,color:!r,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()}));
@@ -1,2 +1,2 @@
1
1
  #!/usr/bin/env node
2
- import{a as r}from"../chunk-4O2FOBZR.js";import"../chunk-OYJ4HES6.js";import"../chunk-DT2FYN6N.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-UN6FVDJQ.js";import"../chunk-SH5KB4P7.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-DFJT2YX6.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-POASCYQ7.js";import"../chunk-WJ4XFLS4.js";r();
2
+ import{a as r}from"../chunk-5HALFCAC.js";import"../chunk-VCK5BJMX.js";import"../chunk-NZSDHDX3.js";import"../chunk-W2BVJ7S2.js";import"../chunk-DOJMTFLB.js";import"../chunk-ZUGTYD2I.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-ZGITQEFC.js";import"../chunk-SH5KB4P7.js";import"../chunk-NA76FFGD.js";import"../chunk-6DTXIZY2.js";import"../chunk-C5MWTMRD.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-RK2UKLWP.js";import"../chunk-ZPK5LYEN.js";r();
@@ -1,2 +1,2 @@
1
1
  #!/usr/bin/env node
2
- import{w as t}from"../chunk-M5VZVMPC.js";import"../chunk-M5QHXNFZ.js";import{b as r}from"../chunk-SH5KB4P7.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-Y6MKMSOC.js";import"../chunk-WJ4XFLS4.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--\u006b\u0065\u0072\u006e\u0065\u006c-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,\u006b\u0065\u0072\u006e\u0065\u006cApiKey:e.browserApiKey??e.\u006b\u0065\u0072\u006e\u0065\u006cApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
2
+ import{z as t}from"../chunk-R6TD6OXQ.js";import"../chunk-M5QHXNFZ.js";import{b as r}from"../chunk-SH5KB4P7.js";import"../chunk-NA76FFGD.js";import"../chunk-6DTXIZY2.js";import"../chunk-XPEJB4BZ.js";import"../chunk-ZPK5LYEN.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--kernel-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,kernelApiKey:e.browserApiKey??e.kernelApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
@@ -1,4 +1,4 @@
1
- import{a as Q,d as Se}from"./chunk-QTDLZTQ7.js";import{createHash as pe,createHmac as Yt,randomBytes as P,randomUUID as y}from"crypto";import{Pool as Wt}from"pg";var ct=[[/perplexity/i,"llm","Perplexity"],[/chatgpt|chat\.openai|openai/i,"llm","ChatGPT"],[/claude|anthropic/i,"llm","Claude"],[/\bgrok\b|xai/i,"llm","Grok"],[/ai[ _-]?overview|google[ _-]?ai|\bsge\b|gemini/i,"llm","Google AI Overview"],[/instagram/i,"social","Instagram"],[/facebook|\bfb\b|meta/i,"social","Facebook"],[/linkedin/i,"social","LinkedIn"],[/tiktok/i,"social","TikTok"],[/youtube|youtu\.be/i,"social","YouTube"],[/twitter|x\.com/i,"social","X"],[/\bg2\b|g2\.com/i,"review","G2"],[/trustpilot/i,"review","Trustpilot"],[/\bbbb\b|better business bureau/i,"review","BBB"],[/capterra/i,"review","Capterra"],[/yelp/i,"review","Yelp"],[/google|bing|duckduckgo|yahoo/i,"search","Search"]];function q(e,t=240){return typeof e=="string"&&e.trim()?e.trim().slice(0,t):null}function Ce(e){try{return e?new URL(e).hostname.replace(/^www\./,""):""}catch{return""}}function Re(e){let t=e.properties??{},i=q(e.source,180)||Ce(e.referrer)||"(direct)",n=q(e.medium,180)||(i==="(direct)"?"(none)":"referral"),s=e.clickIds??{},a=`${i} ${n} ${Ce(e.referrer)}`,r=i==="(direct)"?"direct":/email|newsletter/i.test(n)?"email":"referral",o=i==="(direct)"?"Direct":i;for(let[d,_,u]of ct)if(d.test(a)){r=_,o=u;break}return s.fbclid?(r="social",o=/instagram/i.test(a)?"Instagram":"Facebook"):s.ttclid?(r="social",o="TikTok"):s.rdt_cid?(r="social",o="Reddit"):s.gclid||s.gbraid||s.wbraid?(r="search",o="Google"):s.msclkid&&(r="search",o="Microsoft Ads"),{channelFamily:r,platform:o,source:i,medium:n,campaign:q(e.campaign),campaignId:q(t.campaign_id??t.utm_id),adSetId:q(t.adset_id??t.ad_set_id??t.adgroup_id??t.ad_group_id),adId:q(t.ad_id),creativeId:q(t.creative_id),placement:q(t.placement),term:q(t.utm_term),content:q(t.utm_content)}}import{createHash as yt,randomUUID as ee}from"crypto";import{createCipheriv as _t,createDecipheriv as ut,createHmac as fe,randomBytes as Et,scryptSync as lt,timingSafeEqual as Nt}from"crypto";var Z="xray1";function mt(){return process.env.ANALYTICS_RESTRICTED_DATA_SECRET?.trim()||Q()}function _e(e){return lt(mt(),`xray-restricted:${e}:v1`,32)}function De(e,t){let i=t.trim().slice(0,80);if(!i)throw new Error("analytics encryption purpose required");let n=Et(12),s=_t("aes-256-gcm",_e(i),n);s.setAAD(Buffer.from(`${Z}:${i}`,"utf8"));let a=Buffer.concat([s.update(e,"utf8"),s.final()]),r=s.getAuthTag();return[Z,Buffer.from(i,"utf8").toString("base64url"),n.toString("base64url"),r.toString("base64url"),a.toString("base64url")].join(".")}function Ue(e,t){try{let[i,n,s,a,r]=e.split(".");if(i!==Z||!n||!s||!a||!r)return null;let o=Buffer.from(n,"base64url").toString("utf8");if(o!==t)return null;let d=ut("aes-256-gcm",_e(o),Buffer.from(s,"base64url"));return d.setAAD(Buffer.from(`${Z}:${o}`,"utf8")),d.setAuthTag(Buffer.from(a,"base64url")),Buffer.concat([d.update(Buffer.from(r,"base64url")),d.final()]).toString("utf8")}catch{return null}}function bi(e,t){return fe("sha256",_e(`identity:${e}`)).update(t.trim().toLowerCase()).digest("hex")}function hi(e,t,i,n){return fe("sha256",n).update(`${e}.${t}.${i}`).digest("hex")}function xi(e,t){if(!e||e.length!==t.length)return!1;try{return Nt(Buffer.from(e),Buffer.from(t))}catch{return!1}}var we="enhanced_matching",pt=90,Tt=10,H=class extends Error{constructor(i,n,s){super(n);this.code=i;this.status=s;this.name="AnalyticsIdentityProfileError"}code;status};function te(e,t){return e?.trim().replace(/\s+/g," ").slice(0,t)||null}function Lt(e){if(!e)return null;let[t,i]=e.split("@");return!t||!i?null:`${t.slice(0,1)}${"*".repeat(Math.min(6,Math.max(2,t.length-1)))}@${i}`}function At(e){if(!e)return null;let t=e.replace(/\D/g,"");return t.length<4?null:`***-***-${t.slice(-4)}`}function ue(e){return yt("sha256").update(e).digest("hex")}async function ie(e,t){if(!t.signals.length)return;let i=t.signals.map(a=>a.kind),n=t.signals.map(a=>a.valueHmac);if((await e.query(`SELECT 1 FROM analytics_identity_tombstones
1
+ import{a as Q,d as Se}from"./chunk-L552AQQY.js";import{createHash as pe,createHmac as Yt,randomBytes as P,randomUUID as y}from"crypto";import{Pool as Wt}from"pg";var ct=[[/perplexity/i,"llm","Perplexity"],[/chatgpt|chat\.openai|openai/i,"llm","ChatGPT"],[/claude|anthropic/i,"llm","Claude"],[/\bgrok\b|xai/i,"llm","Grok"],[/ai[ _-]?overview|google[ _-]?ai|\bsge\b|gemini/i,"llm","Google AI Overview"],[/instagram/i,"social","Instagram"],[/facebook|\bfb\b|meta/i,"social","Facebook"],[/linkedin/i,"social","LinkedIn"],[/tiktok/i,"social","TikTok"],[/youtube|youtu\.be/i,"social","YouTube"],[/twitter|x\.com/i,"social","X"],[/\bg2\b|g2\.com/i,"review","G2"],[/trustpilot/i,"review","Trustpilot"],[/\bbbb\b|better business bureau/i,"review","BBB"],[/capterra/i,"review","Capterra"],[/yelp/i,"review","Yelp"],[/google|bing|duckduckgo|yahoo/i,"search","Search"]];function q(e,t=240){return typeof e=="string"&&e.trim()?e.trim().slice(0,t):null}function Ce(e){try{return e?new URL(e).hostname.replace(/^www\./,""):""}catch{return""}}function Re(e){let t=e.properties??{},i=q(e.source,180)||Ce(e.referrer)||"(direct)",n=q(e.medium,180)||(i==="(direct)"?"(none)":"referral"),s=e.clickIds??{},a=`${i} ${n} ${Ce(e.referrer)}`,r=i==="(direct)"?"direct":/email|newsletter/i.test(n)?"email":"referral",o=i==="(direct)"?"Direct":i;for(let[d,_,u]of ct)if(d.test(a)){r=_,o=u;break}return s.fbclid?(r="social",o=/instagram/i.test(a)?"Instagram":"Facebook"):s.ttclid?(r="social",o="TikTok"):s.rdt_cid?(r="social",o="Reddit"):s.gclid||s.gbraid||s.wbraid?(r="search",o="Google"):s.msclkid&&(r="search",o="Microsoft Ads"),{channelFamily:r,platform:o,source:i,medium:n,campaign:q(e.campaign),campaignId:q(t.campaign_id??t.utm_id),adSetId:q(t.adset_id??t.ad_set_id??t.adgroup_id??t.ad_group_id),adId:q(t.ad_id),creativeId:q(t.creative_id),placement:q(t.placement),term:q(t.utm_term),content:q(t.utm_content)}}import{createHash as yt,randomUUID as ee}from"crypto";import{createCipheriv as _t,createDecipheriv as ut,createHmac as fe,randomBytes as Et,scryptSync as lt,timingSafeEqual as Nt}from"crypto";var Z="xray1";function mt(){return process.env.ANALYTICS_RESTRICTED_DATA_SECRET?.trim()||Q()}function _e(e){return lt(mt(),`xray-restricted:${e}:v1`,32)}function De(e,t){let i=t.trim().slice(0,80);if(!i)throw new Error("analytics encryption purpose required");let n=Et(12),s=_t("aes-256-gcm",_e(i),n);s.setAAD(Buffer.from(`${Z}:${i}`,"utf8"));let a=Buffer.concat([s.update(e,"utf8"),s.final()]),r=s.getAuthTag();return[Z,Buffer.from(i,"utf8").toString("base64url"),n.toString("base64url"),r.toString("base64url"),a.toString("base64url")].join(".")}function Ue(e,t){try{let[i,n,s,a,r]=e.split(".");if(i!==Z||!n||!s||!a||!r)return null;let o=Buffer.from(n,"base64url").toString("utf8");if(o!==t)return null;let d=ut("aes-256-gcm",_e(o),Buffer.from(s,"base64url"));return d.setAAD(Buffer.from(`${Z}:${o}`,"utf8")),d.setAuthTag(Buffer.from(a,"base64url")),Buffer.concat([d.update(Buffer.from(r,"base64url")),d.final()]).toString("utf8")}catch{return null}}function bi(e,t){return fe("sha256",_e(`identity:${e}`)).update(t.trim().toLowerCase()).digest("hex")}function hi(e,t,i,n){return fe("sha256",n).update(`${e}.${t}.${i}`).digest("hex")}function xi(e,t){if(!e||e.length!==t.length)return!1;try{return Nt(Buffer.from(e),Buffer.from(t))}catch{return!1}}var we="enhanced_matching",pt=90,Tt=10,H=class extends Error{constructor(i,n,s){super(n);this.code=i;this.status=s;this.name="AnalyticsIdentityProfileError"}code;status};function te(e,t){return e?.trim().replace(/\s+/g," ").slice(0,t)||null}function Lt(e){if(!e)return null;let[t,i]=e.split("@");return!t||!i?null:`${t.slice(0,1)}${"*".repeat(Math.min(6,Math.max(2,t.length-1)))}@${i}`}function At(e){if(!e)return null;let t=e.replace(/\D/g,"");return t.length<4?null:`***-***-${t.slice(-4)}`}function ue(e){return yt("sha256").update(e).digest("hex")}async function ie(e,t){if(!t.signals.length)return;let i=t.signals.map(a=>a.kind),n=t.signals.map(a=>a.valueHmac);if((await e.query(`SELECT 1 FROM analytics_identity_tombstones
2
2
  WHERE site_id=$1 AND (kind,value_hmac) IN (
3
3
  SELECT * FROM unnest($2::text[],$3::text[])
4
4
  ) LIMIT 1`,[t.siteId,i,n])).rowCount)throw new H("analytics_identity_deleted","This identity cannot be automatically recreated.",410)}async function ne(e,t){let i=te(t.displayName,160),n=te(t.email,320)?.toLowerCase()??null,s=te(t.phone,40);if(!i&&!n&&!s)return!1;let r=(await e.query(`SELECT p.public_ref FROM analytics_people p
@@ -1,4 +1,4 @@
1
- import{A as f,Ga as T,Ia as h,Ja as w,Ka as _,La as y,Ma as C,ga as g,ia as S,va as R,wa as E,ya as v,z as u}from"./chunk-OYJ4HES6.js";import{a as P}from"./chunk-DFJT2YX6.js";import{a as m}from"./chunk-POASCYQ7.js";import{readFileSync as N}from"fs";import{homedir as I}from"os";import{join as A}from"path";import{serveStdio as b}from"@modelcontextprotocol/server/stdio";import{McpServer as k}from"@modelcontextprotocol/server";var x=new Map(w.map(r=>[r.upstreamName,r.id]));function O(r){let s=r.split(`
1
+ import{A as f,Ga as T,Ia as h,Ja as w,Ka as _,La as y,Ma as C,ga as g,ia as S,va as R,wa as E,ya as v,z as u}from"./chunk-VCK5BJMX.js";import{a as P}from"./chunk-C5MWTMRD.js";import{a as m}from"./chunk-RK2UKLWP.js";import{readFileSync as N}from"fs";import{homedir as I}from"os";import{join as A}from"path";import{serveStdio as b}from"@modelcontextprotocol/server/stdio";import{McpServer as k}from"@modelcontextprotocol/server";var x=new Map(w.map(r=>[r.upstreamName,r.id]));function O(r){let s=r.split(`
2
2
  `).filter(o=>o.startsWith("data:")).map(o=>o.slice(5).trim()).filter(Boolean);if(!s.length)return JSON.parse(r);for(let o of s){let e=JSON.parse(o);if(e.result||e.error)return e}throw new Error("hosted MCP returned no JSON-RPC result")}function a(r){return{content:[{type:"text",text:JSON.stringify({ok:!1,error:r})}],isError:!0}}var c=class{baseUrl;apiKey;constructor(s,o){this.baseUrl=s.replace(/\/$/,""),this.apiKey=o}async callMemoryTool(s,o){let e=x.get(s);if(!e)return a(`unknown memory tool: ${s}`);try{let n=await fetch(`${this.baseUrl}/mcp`,{method:"POST",headers:{"content-type":"application/json",accept:"application/json, text/event-stream","x-api-key":this.apiKey},body:JSON.stringify({jsonrpc:"2.0",id:`memory:${e}`,method:"tools/call",params:{name:e,arguments:o}})}),l=await n.text();if(!n.ok)return a(`hosted memory ${e} failed (HTTP ${n.status})`);let t=O(l);return t.error?a(t.error.message??`hosted memory ${e} failed`):t.result??a(`hosted memory ${e} returned no result`)}catch(n){return a(n instanceof Error?n.message:`hosted memory ${e} call failed`)}}};function M(r,s){if(s.length===0)throw new Error("Restricted MCP tool allowlist must contain at least one tool name");let o=new Set;for(let t of s){if(typeof t!="string"||t.trim()!==t||t.length===0)throw new Error("Restricted MCP tool allowlist contains a malformed tool name");if(o.has(t))throw new Error(`Restricted MCP tool allowlist contains duplicate tool name: ${t}`);o.add(t)}let e=new Set,n=r,l=n.registerTool.bind(r);return n.registerTool=(t,i,p)=>{if(o.has(t))return e.add(t),l(t,i,p)},{assertComplete(){let t=[...o].filter(i=>!e.has(i));if(t.length>0)throw new Error(`Restricted MCP tool allowlist contains unknown or unavailable tool names: ${t.join(", ")}`)},registeredToolNames(){return[...e]}}}var L=new Set(["paa","serp","browser-agent","scheduled-results","memory"]);function $(){let s=[process.env.MCP_SCRAPER_KEY_PATH?.trim(),A(I(),".mcp-scraper-key")].filter(Boolean);for(let o of s)try{let e=N(o,"utf8").trim();if(e)return e}catch{}}function Y(r,s={}){let o=s.toolsets??L,e=process.env.MCP_SCRAPER_BASE_URL?.trim()||process.env.MCP_BASE_URL?.trim()||"https://mcpscraper.dev",n=g(),l=S({deploymentProfile:n,transportProfile:"stdio",baseUrl:e,explicitlyEnabled:process.env.MCP_SCRAPER_ALLOW_PRIVATE_NETWORK==="1"}),t=process.env.BROWSER_AGENT_CONSOLE_URL?.trim()||e,i=new k({name:"mcp-scraper",version:m},{instructions:u,cacheHints:{"server/discover":{ttlMs:3e5,cacheScope:"private"},"tools/list":{ttlMs:3e5,cacheScope:"private"},"prompts/list":{ttlMs:3e5,cacheScope:"private"},"resources/list":{ttlMs:3e5,cacheScope:"private"},"resources/templates/list":{ttlMs:3e5,cacheScope:"private"},"resources/read":{ttlMs:6e4,cacheScope:"private"}}});f(i);let p=s.allowedToolNames?M(i,s.allowedToolNames):void 0,d=s.httpExecutor??new T(e,r,{localNetworkAccess:l});return o.has("paa")&&v(i,d,{ownerId:R(r),deploymentProfile:n,transportProfile:"stdio",baseUrl:e,localNetworkAccess:l,taskHandleSecret:r}),o.has("serp")&&E(i,d,{exposeDevelopmentDiagnostics:n==="development"||n==="test"}),o.has("browser-agent")&&h(i,{baseUrl:e,apiKey:r,consoleBaseUrl:t}),o.has("scheduled-results")&&C(i,new y(e,r)),o.has("memory")&&_(i,new c(e,r)),p?.assertComplete(),i}function se(r={}){let s=process.argv.includes("--stdio")||process.env.MCP_SCRAPER_FORCE_STDIO==="1",o=!!(process.stdin.isTTY&&process.stdout.isTTY),e=process.argv.includes("--help")||process.argv.includes("-h");if(!s&&(o||e)){let l=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY;process.stdout.write(P({version:m,color:!l,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()})),process.exit(0)}let n=(process.env.MCP_SCRAPER_API_KEY??process.env.MCP_SCRAPER_KEY??process.env.MCP_API_KEY??$())?.trim();n||(process.stderr.write(`MCP_SCRAPER_API_KEY env var or ~/.mcp-scraper-key is required
3
3
  `),process.exit(1)),b(()=>Y(n,r),{legacy:"serve",onerror(l){process.stderr.write(`${l.message}
4
4
  `)}})}export{se as a};