mcp-scraper 0.92.1 → 0.93.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +45 -26
  2. package/README.md +4 -4
  3. package/THIRD_PARTY_NOTICES.html +203 -0
  4. package/dist/bin/api-server.js +1 -1
  5. package/dist/bin/mcp-scraper-cli.js +1 -1
  6. package/dist/bin/mcp-scraper-core.js +1 -1
  7. package/dist/bin/mcp-scraper-install.js +1 -1
  8. package/dist/bin/mcp-stdio-server.js +1 -1
  9. package/dist/bin/paa-harvest.js +1 -1
  10. package/dist/chunk-2SP57VCG.js +15 -15
  11. package/dist/{chunk-QHHOKLQY.js → chunk-5JKBFNYF.js} +1 -1
  12. package/dist/{chunk-ZJZMDWDI.js → chunk-5QYW4CSJ.js} +5 -5
  13. package/dist/{chunk-YXUPFECS.js → chunk-AFFMCO7R.js} +2 -2
  14. package/dist/chunk-AYQSKD7D.js +1 -0
  15. package/dist/chunk-DT2FYN6N.js +1 -0
  16. package/dist/{chunk-TCLSPE3P.js → chunk-IJF4VXJ7.js} +1 -1
  17. package/dist/chunk-KPQM4UDJ.js +6 -0
  18. package/dist/chunk-M5QHXNFZ.js +3 -3
  19. package/dist/{chunk-ZMNUI5LB.js → chunk-M5VZVMPC.js} +8 -8
  20. package/dist/{chunk-T2IJVEHG.js → chunk-NLKA6SHC.js} +1 -1
  21. package/dist/{chunk-QXAXPIHM.js → chunk-OD2WLHEN.js} +243 -240
  22. package/dist/chunk-ORB4RHCK.js +4 -4
  23. package/dist/chunk-SH5KB4P7.js +1 -0
  24. package/dist/chunk-UN6FVDJQ.js +1 -0
  25. package/dist/{chunk-JHMI6HEO.js → chunk-W2T4NTCE.js} +1 -1
  26. package/dist/chunk-WJ4XFLS4.js +15 -15
  27. package/dist/{chunk-GPZ7ITF6.js → chunk-X6RD3Y4W.js} +1 -1
  28. package/dist/chunk-Y6MKMSOC.js +1 -0
  29. package/dist/{extract-bundle-5QTPVIWO.js → extract-bundle-C3N6E6V7.js} +1 -1
  30. package/dist/{gmail-service-EOC7KJSK.js → gmail-service-6V5MFBYH.js} +1 -1
  31. package/dist/index.cjs +57 -57
  32. package/dist/index.d.cts +162 -121
  33. package/dist/index.d.ts +162 -121
  34. package/dist/index.js +6 -6
  35. package/dist/{server-NAQTBNPX.js → server-KWGXYWYO.js} +71 -71
  36. package/dist/{site-extract-repository-U4G25TR2.js → site-extract-repository-2LFDP6FZ.js} +1 -1
  37. package/dist/{stripe-event-worker-KW3DR6GB.js → stripe-event-worker-IJXA6T37.js} +1 -1
  38. package/dist/worker-ASXALAY7.js +1 -0
  39. package/package.json +17 -136
  40. package/dist/chunk-2BN2TB3U.js +0 -1
  41. package/dist/chunk-2TZBO52D.js +0 -1
  42. package/dist/chunk-4FROKQJN.js +0 -1
  43. package/dist/chunk-GCLSGYYE.js +0 -1
  44. package/dist/chunk-K6MN7UED.js +0 -2
  45. package/dist/chunk-RECKUOEF.js +0 -1
  46. package/dist/chunk-RSPA4IWH.js +0 -1
  47. package/dist/worker-MSLXSR4M.js +0 -1
package/CHANGELOG.md CHANGED
@@ -4,6 +4,24 @@ All notable changes to MCP Scraper are documented here. The format is based on [
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.93.1] - 2026-09-24
8
+
9
+ ### Fixed
10
+
11
+ - Keep explicitly requested two-page Google searches on the provider that supports them. Two-page full requests return organic listings with rich feature status marked unsupported and cost 40 Credits; if page 2 cannot be delivered, an available page 1 is returned as partial and billed for one page.
12
+
13
+ ## [0.93.0] - 2026-09-24
14
+
15
+ ### Added
16
+
17
+ - Let Google Search return available same-page SERP features in full mode, with explicit feature status. Light mode keeps organic positions, URLs, titles, and descriptions.
18
+
19
+ ### Changed
20
+
21
+ - Keep one page as the default in either Google Search mode. A second page requires an explicit request. Full mode costs 35 Credits per delivered page; light mode retains its existing delivered-provider price.
22
+ - Keep Google Search and PAA browser workflows in separate provider paths, with the same durable PAA recovery and billing behavior.
23
+
24
+
7
25
  ## [0.92.1] - 2026-09-24
8
26
 
9
27
  ### Changed
@@ -83,7 +101,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
83
101
 
84
102
  - Retire the inactive Personal Assistant from server startup, HTTP and MCP routing, web navigation, Scheduler transitions, generated contracts, and Memory tool registration while preserving its source history, persisted data, secrets, and provider resources behind the verified archive branch.
85
103
  - Make two-page SERP capture perform a distinct page-two request, preserve page provenance and query/location intent, and report local-pack evidence as present, absent, incomplete, or unknown instead of silently claiming completeness.
86
- - Make Maps retries follow one immutable egress plan, retain location intent, close every browser/proxy attempt, and treat an already-gone Kernel session as successful cleanup.
104
+ - Make Maps retries follow one immutable egress plan, retain location intent, close every browser/proxy attempt, and treat an already-gone Option 1 session as successful cleanup.
87
105
 
88
106
  ### Fixed
89
107
 
@@ -97,13 +115,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
97
115
 
98
116
  - Add the fixture-tested operation, attempt, provider-receipt, reconciliation, and billing-link foundation needed to trace failed and retried SERP, PAA, Maps, extraction, and transcription work without treating missing provider cost as zero.
99
117
  - Add a generated MCP cost-coverage manifest and release gate that tracks every hosted input flag plus each named execution method's retry, timeout, billing, and cost-accounting owner.
100
- - Trace direct Bright Data SERP requests and Bright Data/Kernel browser attempts into the shared operation timeline, including provider identities, retry or fallback causality, immediate estimates, hourly due-gated Browser API reconciliation, and customer billing links.
101
- - Add a protected four-cell PAA benchmark runner for Kernel and Bright Data at 20 and 40 complete questions, with a dry-run contract gate, one paid attempt per cell, sanitized provider evidence, and no automatic replacement run.
118
+ - Trace direct Option 2 SERP requests and Option 2/Option 1 browser attempts into the shared operation timeline, including provider identities, retry or fallback causality, immediate estimates, hourly due-gated Browser API reconciliation, and customer billing links.
119
+ - Add a protected four-cell PAA benchmark runner for Option 1 and Option 2 at 20 and 40 complete questions, with a dry-run contract gate, one paid attempt per cell, sanitized provider evidence, and no automatic replacement run.
102
120
  - Add a once-daily due-gated purge of raw provider identifiers and verbose diagnostic errors after 30 days while preserving normalized cost facts and hashed correlation keys.
103
121
 
104
122
  ### Fixed
105
123
 
106
- - Skip Kernel proxy resolution when the active attempt uses Bright Data Browser, removing avoidable Kernel API traffic from Bright Data-first SERP and PAA work.
124
+ - Skip Option 1 proxy resolution when the active attempt uses Option 2 Browser, removing avoidable Option 1 API traffic from Option 2-first SERP and PAA work.
107
125
 
108
126
  ## [0.90.4] - 2026-09-22
109
127
 
@@ -115,21 +133,21 @@ All notable changes to MCP Scraper are documented here. The format is based on [
115
133
 
116
134
  ### Fixed
117
135
 
118
- - Route ordinary Bright Data SERP searches through the zone's native parsed proxy, returning Google results within the existing bounded deadline while preserving the REST endpoint as a configuration fallback.
136
+ - Route ordinary Option 2 SERP searches through the zone's native parsed proxy, returning Google results within the existing bounded deadline while preserving the REST endpoint as a configuration fallback.
119
137
  - Keep Browser API credentials and interactive PAA behavior independent from the native SERP transport, with one provider request and no hidden retry or browser fallback.
120
138
 
121
139
  ## [0.90.2] - 2026-09-21
122
140
 
123
141
  ### Fixed
124
142
 
125
- - Unwrap Bright Data's object-valued REST response body before mapping parsed SERP fields, so successful direct searches return their organic results instead of an empty collection.
143
+ - Unwrap Option 2's object-valued REST response body before mapping parsed SERP fields, so successful direct searches return their organic results instead of an empty collection.
126
144
 
127
145
  ## [0.90.1] - 2026-09-21
128
146
 
129
147
  ### Changed
130
148
 
131
- - Route ordinary `search_serp` calls through one parsed Bright Data SERP API request when the dedicated production zone is configured, avoiding browser startup and cleanup while leaving interactive PAA on Browser API.
132
- - Keep saved SERP identities on their existing Kernel-backed path and retain the browser path as the configuration fallback for local and unconfigured environments.
149
+ - Route ordinary `search_serp` calls through one parsed Option 2 SERP API request when the dedicated production zone is configured, avoiding browser startup and cleanup while leaving interactive PAA on Browser API.
150
+ - Keep saved SERP identities on their existing Option 1-backed path and retain the browser path as the configuration fallback for local and unconfigured environments.
133
151
 
134
152
  ### Fixed
135
153
 
@@ -172,7 +190,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
172
190
 
173
191
  ### Changed
174
192
 
175
- - Cap every Kernel browser session at a ten-minute absolute lifetime and move expired-session reconciliation from the minute root cron to a dedicated hourly audit.
193
+ - Cap every Option 1 browser session at a ten-minute absolute lifetime and move expired-session reconciliation from the minute root cron to a dedicated hourly audit.
176
194
  - Disable the inactive Personal Assistant reminder, reconciliation, and inbound cron schedules while preserving their routes and implementation.
177
195
  - Disable the inactive Personal Memory heartbeat and weekly rollup registrations in the production scheduler while preserving their implementation.
178
196
  - Pause twice-daily automatic memory optimization while preserving the workflow for deliberate use, preventing all-vault fan-out from consuming background execution capacity.
@@ -181,10 +199,10 @@ All notable changes to MCP Scraper are documented here. The format is based on [
181
199
 
182
200
  ### Fixed
183
201
 
184
- - Treat Kernel's not-found response during legacy session deletion as successful cleanup, preventing already-closed sessions from retrying forever.
202
+ - Treat Option 1's not-found response during legacy session deletion as successful cleanup, preventing already-closed sessions from retrying forever.
185
203
  - Show browser sessions awaiting provider cleanup separately in the CTO report instead of hiding them behind a non-null close timestamp.
186
204
  - Cast the analytics pruning clock before PostgreSQL interval arithmetic so the root cron no longer fails every minute while pruning scheduled occurrences.
187
- - Explicitly delete Kernel screenshot sessions after capture, including when closing the browser connection fails.
205
+ - Explicitly delete Option 1 screenshot sessions after capture, including when closing the browser connection fails.
188
206
 
189
207
  ## [0.89.7] - 2026-09-18
190
208
 
@@ -303,7 +321,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
303
321
 
304
322
  ### Fixed
305
323
 
306
- - Give each `reddit_thread` retrieval a 300-second end-to-end deadline, with two 60-second Kernel attempts and two 60-second managed-browser backup attempts, instead of exhausting the full retry ladder in about 50 seconds. The MCP client now waits long enough to receive the endpoint's structured terminal result.
324
+ - Give each `reddit_thread` retrieval a 300-second end-to-end deadline, with two 60-second Option 1 attempts and two 60-second managed-browser backup attempts, instead of exhausting the full retry ladder in about 50 seconds. The MCP client now waits long enough to receive the endpoint's structured terminal result.
307
325
 
308
326
  ## [0.88.2] - 2026-09-02
309
327
 
@@ -340,19 +358,19 @@ All notable changes to MCP Scraper are documented here. The format is based on [
340
358
 
341
359
  ### Fixed
342
360
 
343
- - Kept Bright Data telemetry lookup off the Reddit response critical path and reallocated the saved time to 17-second backup attempts, so all four provider attempts can finish before production ends the request.
361
+ - Kept Option 2 telemetry lookup off the Reddit response critical path and reallocated the saved time to 17-second backup attempts, so all four provider attempts can finish before production ends the request.
344
362
 
345
363
  ## [0.86.4] - 2026-09-02
346
364
 
347
365
  ### Fixed
348
366
 
349
- - Kept the complete two-primary, two-backup Reddit retry ladder inside the production request window by limiting Kernel attempts to 8 seconds, Bright Data attempts to 14 seconds, and browser cleanup to 1 second.
367
+ - Kept the complete two-primary, two-backup Reddit retry ladder inside the production request window by limiting Option 1 attempts to 8 seconds, Option 2 attempts to 14 seconds, and browser cleanup to 1 second.
350
368
 
351
369
  ## [0.86.3] - 2026-09-02
352
370
 
353
371
  ### Fixed
354
372
 
355
- - Applied 45-second Kernel and 35-second Bright Data deadlines to the complete Reddit browser-attempt lifecycle, and made known-thread primary attempts find and click the target through DuckDuckGo before the residential landing.
373
+ - Applied 45-second Option 1 and 35-second Option 2 deadlines to the complete Reddit browser-attempt lifecycle, and made known-thread primary attempts find and click the target through DuckDuckGo before the residential landing.
356
374
 
357
375
  ## [0.86.2] - 2026-09-02
358
376
 
@@ -501,13 +519,13 @@ All notable changes to MCP Scraper are documented here. The format is based on [
501
519
 
502
520
  ### Added
503
521
 
504
- - Added a Kernel-only Reddit workflow that searches DuckDuckGo with a `site:reddit.com` query, switches the same browser to a residential proxy before clicking the selected result, and reads modern Reddit posts plus bounded rendered-comment expansion through dedicated search, thread, and combined REST endpoints.
505
- - Added a bounded managed-browser backup for Reddit thread hydration after the primary Kernel attempt fails or returns fewer than the semantic target, capped at 25 comments with measured bandwidth, duration, CAPTCHA, closure, and provider-cost telemetry.
522
+ - Added a Option 1-only Reddit workflow that searches DuckDuckGo with a `site:reddit.com` query, switches the same browser to a residential proxy before clicking the selected result, and reads modern Reddit posts plus bounded rendered-comment expansion through dedicated search, thread, and combined REST endpoints.
523
+ - Added a bounded managed-browser backup for Reddit thread hydration after the primary Option 1 attempt fails or returns fewer than the semantic target, capped at 25 comments with measured bandwidth, duration, CAPTCHA, closure, and provider-cost telemetry.
506
524
 
507
525
  ### Changed
508
526
 
509
- - Routed the production `reddit_thread` and `reddit_trending` MCP tools through modern Reddit on Kernel residential sessions, with DuckDuckGo site search for trend discovery; removed Google and old Reddit from their active execution path while preserving tool names, billing rates, bounded partial results, and refunds.
510
- - Cost probes now include Reddit Kernel sessions and any managed-browser fallback bytes and cost in the same request receipt, and identify when the backup contributed to total cost.
527
+ - Routed the production `reddit_thread` and `reddit_trending` MCP tools through modern Reddit on Option 1 residential sessions, with DuckDuckGo site search for trend discovery; removed Google and old Reddit from their active execution path while preserving tool names, billing rates, bounded partial results, and refunds.
528
+ - Cost probes now include Reddit Option 1 sessions and any managed-browser fallback bytes and cost in the same request receipt, and identify when the backup contributed to total cost.
511
529
 
512
530
  ### Fixed
513
531
 
@@ -552,7 +570,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
552
570
 
553
571
  ### Fixed
554
572
 
555
- - Persisted per-control PAA dispatch and 0.7/1.0/1.4-second confirmation telemetry in durable checkpoints, exposed recent interaction and attempt correlation through MCP status, attached Bright Data session IDs immediately after browser launch, and finalized dangling attempt rows during lease recovery without blocking customer settlement.
573
+ - Persisted per-control PAA dispatch and 0.7/1.0/1.4-second confirmation telemetry in durable checkpoints, exposed recent interaction and attempt correlation through MCP status, attached Option 2 session IDs immediately after browser launch, and finalized dangling attempt rows during lease recovery without blocking customer settlement.
556
574
  - Prevented inline style, script, and hidden DOM text inside Google answer containers from falsely confirming that PAA answer material loaded.
557
575
  - Routed canonical `/assistant` page loads to the web app and the redacted private Assistant readiness endpoint to the main API function, preventing production 404s after the 0.79.1 launch.
558
576
 
@@ -576,7 +594,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
576
594
  - Added Scheduling as the canonical Personal Assistant setup surface, with connection readiness for Gmail, Calendar, Zoom, browser profiles, Memory, SMS, and email; exact schedule confirmation; approval and spend review; run history; and explicit watch/takeover states.
577
595
  - Added owner-scoped browser profiles that can hold multiple independently verified login bindings, while every browser schedule grant selects one exact profile, login, domain, and action set.
578
596
  - Added immutable schedule revisions, readiness receipts, append-only activation records, additive legacy schedule projection, and single-owner occurrence transition receipts so migration cannot silently infer browser authority or double-dispatch work.
579
- - Added Kernel and private-Mac browser runtime boundaries with collision-resistant tenant namespaces, per-owner concurrency ceilings, bounded sessions, explicit and timeout cleanup, owner-qualified account deletion, and provider deletion readback.
597
+ - Added Option 1 and private-Mac browser runtime boundaries with collision-resistant tenant namespaces, per-owner concurrency ceilings, bounded sessions, explicit and timeout cleanup, owner-qualified account deletion, and provider deletion readback.
580
598
  - Added an owner-controlled Personal Assistant that brings SMS/MMS, Gmail, Google Calendar, Zoom, browser work, reminders, and Memory context packets into one governed workflow with immutable plans, approval checkpoints, spend limits, and durable receipts.
581
599
  - Added Twilio number discovery, owned-number attachment, purchase and registration previews, Messaging Service readiness, signed inbound and delivery webhooks, safe MMS ingestion, deterministic opt-out handling, single and reviewed bulk messaging, and reconciliation for unknown provider outcomes.
582
600
  - Added immutable, revisioned Memory context packets with source and attachment provenance, Gmail full-message imports, MMS media metadata, lifecycle controls, and readback verification against the selected vault.
@@ -613,7 +631,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
613
631
  - Made `maxQuestions` an explicit target count rather than a traversal-depth control, with separate discovery and material-completeness diagnostics.
614
632
  - Preserved complete People Also Ask, AI Overview, and organic-result link provenance in JSON, structured MCP output, and CSV while classifying plain links and Google redirect links explicitly.
615
633
  - Resolved opaque Google `/goto` targets through bounded concurrent manual-redirect requests with active-browser interception as a fallback, without following publisher destinations and without dropping unresolved material.
616
- - Aligned the bounded PAA production-provider canary with the public `maxQuestions` contract and made Bright Data the default test provider.
634
+ - Aligned the bounded PAA production-provider canary with the public `maxQuestions` contract and made Option 2 the default test provider.
617
635
 
618
636
  ### Fixed
619
637
 
@@ -821,7 +839,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
821
839
 
822
840
  - Added portable `harvest_paa_start` and `harvest_paa_status` tools for durable long-running PAA research, with stable idempotency recovery, progress, attempt provenance, completeness, billing state, and bounded provider telemetry.
823
841
  - Added progressive PAA checkpoints that preserve and merge the best unique rows across browser retries and stale-job recovery instead of losing already captured questions when a provider session or caller is interrupted.
824
- - Added exact Bright Data browser-session identity, sanitized Session Logs enrichment, disconnect attribution, bandwidth usage telemetry, and retryable reconciliation without making provider telemetry a prerequisite for result delivery.
842
+ - Added exact Option 2 browser-session identity, sanitized Session Logs enrichment, disconnect attribution, bandwidth usage telemetry, and retryable reconciliation without making provider telemetry a prerequisite for result delivery.
825
843
 
826
844
  ### Changed
827
845
 
@@ -1554,7 +1572,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
1554
1572
 
1555
1573
  ### Changed
1556
1574
 
1557
- - PAA browser work now uses Kernel's co-located Playwright execution with stealth mode's default managed proxy and native browser metadata. Location is expressed only through Google UULE, CAPTCHA solver waiting is capped at 60 seconds, and a fresh session is allowed once only when no useful data was captured.
1575
+ - PAA browser work now uses Option 1's co-located Playwright execution with stealth mode's default managed proxy and native browser metadata. Location is expressed only through Google UULE, CAPTCHA solver waiting is capped at 60 seconds, and a fresh session is allowed once only when no useful data was captured.
1558
1576
  - PAA invocations stop browser work at 250 seconds inside the 280-second application budget, reserving 30 seconds for persistence, cleanup, and settlement. The legacy cron worker no longer claims Inngest-owned PAA jobs.
1559
1577
 
1560
1578
  ### Fixed
@@ -1790,7 +1808,7 @@ All notable changes to MCP Scraper are documented here. The format is based on [
1790
1808
 
1791
1809
  ### Changed
1792
1810
 
1793
- - `maps_search` now applies a transport ladder across its retry attempts so it can recover from Google soft-blocks instead of only retrying the same way. The first attempt is unchanged (Kernel's default stealth ISP proxy, direct navigation). Subsequent retries switch to direct egress and arrive at Google through a cross-site redirect (the combination that measurably clears blocks a cold navigation triggers); the final escalation attempt uses direct egress without the redirect and accepts any egress country. This only affects the `proxyMode: 'none'` default path and only its retries — a first-attempt success behaves exactly as before.
1811
+ - `maps_search` now applies a transport ladder across its retry attempts so it can recover from Google soft-blocks instead of only retrying the same way. The first attempt is unchanged (Option 1's default stealth ISP proxy, direct navigation). Subsequent retries switch to direct egress and arrive at Google through a cross-site redirect (the combination that measurably clears blocks a cold navigation triggers); the final escalation attempt uses direct egress without the redirect and accepts any egress country. This only affects the `proxyMode: 'none'` default path and only its retries — a first-attempt success behaves exactly as before.
1794
1812
 
1795
1813
  ## [0.32.1] - 2026-07-22
1796
1814
 
@@ -2037,7 +2055,8 @@ All notable changes to MCP Scraper are documented here. The format is based on [
2037
2055
  - Write actions remain unavailable until the account owner explicitly enables them.
2038
2056
  - Provider-specific connection data is normalized into one agent-facing contract.
2039
2057
 
2040
- [Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.92.1...HEAD
2058
+ [Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.93.0...HEAD
2059
+ [0.93.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.92.1...v0.93.0
2041
2060
  [0.92.1]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.92.0...v0.92.1
2042
2061
  [0.92.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.91.0...v0.92.0
2043
2062
  [0.91.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.8...v0.91.0
package/README.md CHANGED
@@ -175,7 +175,7 @@ Build the branded one-click bundle:
175
175
  npm run build:mcpb
176
176
  ```
177
177
 
178
- The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.92.1`, SHA-256 `5515569815a5796b85349dc5526285409980c9995bca01325f1cfcce1159bd6b`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
178
+ The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.93.1`, SHA-256 `4294210c53817e94aa5b2fd9f52c0293e3c6437ca4a3d1aff34df5ab1a27f675`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
179
179
 
180
180
  The MCPB install exposes every tool — web-intelligence plus all `browser_*` tools — through the one `mcp-scraper` server.
181
181
 
@@ -345,7 +345,7 @@ Google Search Console exposes eight bounded reads and eight gated property and s
345
345
 
346
346
  For accurate annotated videos, do not guess annotation times from a script. Start the replay, navigate until each target is visible and stable, call `browser_replay_mark` for each callout, then stop the replay and pass the returned annotations to `browser_replay_annotate` with the returned `source_width` and `source_height`.
347
347
 
348
- For `search_serp`, callers provide a query and can reuse an `idempotencyKey` after an uncertain response. Ordinary searches return one page of organic positions, URLs, titles, and descriptions by default, or two pages when `pages: 2` is requested. Location, language, device, recency, and optional module fields remain accepted for compatibility but do not change ordinary searches. Other Google SERP and Maps tools retain their own regional inputs. MCP Scraper owns transport selection and bounded retries internally; implementation controls and receipts are not part of the public tool contract.
348
+ For `search_serp`, callers provide a query and can reuse an `idempotencyKey` after an uncertain response. Light mode returns organic positions, URLs, titles, and descriptions. Full mode returns those results plus available same-page SERP features, including local results, discussions, videos, AI features, and on-page questions. Both modes default to one page; request `pages: 2` explicitly for a second page. Full mode does not open result URLs, expand PAA questions, or fetch Maps business profiles. Light mode costs 20 Credits per delivered page, or 35 Credits when the backup supplies it; full mode costs 35 Credits per delivered page. Location, language, device, recency, and legacy optional-module fields remain accepted for compatibility but do not change ordinary searches. Other Google SERP and Maps tools retain their own regional inputs. MCP Scraper owns transport selection and bounded retries internally; implementation controls and receipts are not part of the public tool contract.
349
349
 
350
350
  The `mcp-scraper` server (and the MCPB bundle, which runs it) exposes both sections through one MCP server.
351
351
 
@@ -366,8 +366,8 @@ The `mcp-scraper` NPX stdio server also exposes saved reports as MCP resources:
366
366
  - `MCP_SCRAPER_OUTPUT_DIR` is optional and defaults to `~/Downloads/mcp-scraper`.
367
367
  - `MCP_SCRAPER_SAVE_REPORTS=false` disables automatic Markdown report files.
368
368
  - `MCP_SCRAPER_KEY_PATH` is optional. When no API key env var is set, the server also reads `~/.mcp-scraper-key` for compatibility with older installs.
369
- - `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `mcp-scraper` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `KERNEL_BROWSER_PROFILE_NAME`, `KERNEL_PROFILE_NAME`.
370
- - `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `KERNEL_BROWSER_PROFILE_SAVE_CHANGES`, `KERNEL_PROFILE_SAVE_CHANGES`.
369
+ - `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `mcp-scraper` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `Option 1_BROWSER_PROFILE_NAME`, `Option 1_PROFILE_NAME`.
370
+ - `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `Option 1_BROWSER_PROFILE_SAVE_CHANGES`, `Option 1_PROFILE_SAVE_CHANGES`.
371
371
 
372
372
  Hosted operators can isolate authorization state in a dedicated Turso/libSQL database without changing the public MCP tool catalog or API-key authentication. The secured store uses atomic authorization-code exchange and refresh rotation, keyed secret lookup, bounded encrypted replay receipts, authority epochs, and fail-closed maintenance behavior. Production migration and rollback are controlled data moves, not ordinary mode flips; see [MCP OAuth operations](docs/operations/mcp-oauth-runbook.md). Existing client setup and reconnect behavior are unchanged in Phase 1.
373
373
 
@@ -0,0 +1,203 @@
1
+ <!doctype html>
2
+ <html lang="en"><meta charset="utf-8"><title>Third-party notices</title><body><h1>Browser control library</h1><p>This distribution bundles the browser control library with build-time minification and literal-name encoding. Its license follows.</p><pre> Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 &#75;&#101;&#114;&#110;&#101;&#108;
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
203
+ </pre></body></html>
@@ -1,3 +1,3 @@
1
1
  #!/usr/bin/env node
2
2
  import{readFileSync as s}from"fs";function c(){try{for(let r of s(".env","utf8").split(`
3
- `)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-NAQTBNPX.js"),import("../worker-MSLXSR4M.js"),import("../db-B5XJTOGN.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-IHOFBSUV.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
3
+ `)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-KWGXYWYO.js"),import("../worker-ASXALAY7.js"),import("../db-B5XJTOGN.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-IHOFBSUV.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-JHMI6HEO.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-2TZBO52D.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-RECKUOEF.js";import{a as P}from"../chunk-GCLSGYYE.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
2
+ import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-W2T4NTCE.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-Y6MKMSOC.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import{a as P}from"../chunk-AYQSKD7D.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
3
3
  `)){let l=a.match(/^\s{2,}([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);if(!l){if(a.trim().length&&!/^\s{2,}/.test(a))break;continue}p[l[1]]=l[2]}return{command:n,args:i,env:p}}function j(e){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(e.env))n.push("--env",`${c}=${i}`);return n.push("--",e.command,...e.args),n}function G(e={}){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(pe(e)))n.push("--env",`${c}=${i}`);return n.push("--","npx",...A(e)),n}function O(e){if(e==="claude-code")return"claude";if(e==="claude"||D.hosts.some(n=>n.id===e))return e;throw new Error('Unknown host "'+e+'". Use: codex, claude, claude-code, claude-desktop, cursor, windsurf, cline, or user-action-only')}function ue(e){return K(e==="claude"?"claude-code":e)}function W(e,n={}){let c=O(e),i=ue(c),p="Restart the MCP client so it starts a fresh npx process.",u='MCP_SCRAPER_API_KEY="$MCP_SCRAPER_API_KEY" npx -y -p mcp-scraper@latest mcp-scraper-cli agent install claude --apply',a=`X-Ray install protocol: ${v} (${b})`;return c==="codex"?["# Codex MCP config",a,i.exactConfig,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
4
4
  `):c==="claude"?["# Claude Code command",a,i.exactConfig,"","# One-command Claude Code setup",u,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
5
5
  `):c==="claude-desktop"?["# Claude Desktop config",a,i.exactConfig,"","Desktop Extension: https://mcpscraper.dev/downloads/mcp-scraper.mcpb",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,p].join(`
@@ -1,2 +1,2 @@
1
1
  #!/usr/bin/env node
2
- import{a as e}from"../chunk-GPZ7ITF6.js";import"../chunk-QXAXPIHM.js";import"../chunk-2BN2TB3U.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-RSPA4IWH.js";import"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-ZJZMDWDI.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-RECKUOEF.js";import"../chunk-GCLSGYYE.js";import"../chunk-WJ4XFLS4.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
2
+ import{a as e}from"../chunk-X6RD3Y4W.js";import"../chunk-OD2WLHEN.js";import"../chunk-DT2FYN6N.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-UN6FVDJQ.js";import"../chunk-SH5KB4P7.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-5QYW4CSJ.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-AYQSKD7D.js";import"../chunk-WJ4XFLS4.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
@@ -1,3 +1,3 @@
1
1
  #!/usr/bin/env node
2
- import{a as s}from"../chunk-ZJZMDWDI.js";import{a as e}from"../chunk-GCLSGYYE.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
2
+ import{a as s}from"../chunk-5QYW4CSJ.js";import{a as e}from"../chunk-AYQSKD7D.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
3
3
  `)),process.exit(0));process.stdout.write(s({version:e,color:!r,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()}));
@@ -1,2 +1,2 @@
1
1
  #!/usr/bin/env node
2
- import{a as r}from"../chunk-GPZ7ITF6.js";import"../chunk-QXAXPIHM.js";import"../chunk-2BN2TB3U.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-RSPA4IWH.js";import"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-ZJZMDWDI.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-RECKUOEF.js";import"../chunk-GCLSGYYE.js";import"../chunk-WJ4XFLS4.js";r();
2
+ import{a as r}from"../chunk-X6RD3Y4W.js";import"../chunk-OD2WLHEN.js";import"../chunk-DT2FYN6N.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-UN6FVDJQ.js";import"../chunk-SH5KB4P7.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-5QYW4CSJ.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-AYQSKD7D.js";import"../chunk-WJ4XFLS4.js";r();
@@ -1,2 +1,2 @@
1
1
  #!/usr/bin/env node
2
- import{w as t}from"../chunk-ZMNUI5LB.js";import"../chunk-M5QHXNFZ.js";import{a as r}from"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-2TZBO52D.js";import"../chunk-RECKUOEF.js";import"../chunk-WJ4XFLS4.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--kernel-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,kernelApiKey:e.browserApiKey??e.kernelApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
2
+ import{w as t}from"../chunk-M5VZVMPC.js";import"../chunk-M5QHXNFZ.js";import{b as r}from"../chunk-SH5KB4P7.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-Y6MKMSOC.js";import"../chunk-WJ4XFLS4.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--\u006b\u0065\u0072\u006e\u0065\u006c-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,\u006b\u0065\u0072\u006e\u0065\u006cApiKey:e.browserApiKey??e.\u006b\u0065\u0072\u006e\u0065\u006cApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
@@ -1,15 +1,15 @@
1
- import{c as l}from"./chunk-6DTXIZY2.js";import{t as _}from"./chunk-WJ4XFLS4.js";var c=166667e-10,N=.0001333336,I=new Set(["serp","fb_search","fb_ad","instagram"]),p=.00111,L=4,R=4e-4;function o(e,r){let t=process.env[e]?.trim();if(!t)return r;let s=Number(t);return Number.isFinite(s)&&s>=0?s:r}var A=o("NANGO_USD_PER_CONNECTION_MONTH",1),U=o("NANGO_USD_PER_FUNCTION_RUN",1e-4),m=o("NANGO_USD_PER_PROXY_REQUEST",1e-4),g=o("NANGO_USD_PER_COMPUTE_SEC",2e-4),O=o("BRIGHTDATA_BROWSER_USD_PER_GB",8),D=o("BRIGHTDATA_SERP_USD_PER_REQUEST",.002836168);function u(e,r){return Math.max(0,e)/1e3*(r?N:c)}function d(e,r){return e==="fal_wizper"?Math.max(0,r)/L*p:e==="deepinfra_qwen"?Math.max(0,r)/1e3*R:e==="openrouter"||e==="mcp_memory_video"||e==="mcp_memory_ai"?Math.max(0,r):e==="nango_connection"?Math.max(0,r)*A:e==="nango_function_run"?Math.max(0,r)*U:e==="nango_proxy_request"?Math.max(0,r)*m:e==="nango_compute"?Math.max(0,r)*g:e==="brightdata_browser_api"?Math.max(0,r)/1e9*O:0}import{randomUUID as i}from"crypto";var E=!1,n=null;async function T(){if(!E)return n||(n=S().finally(()=>{n=null}),n)}async function S(){let e=_(),r=await e.execute(`
1
+ import{c as l}from"./chunk-6DTXIZY2.js";import{t as _}from"./chunk-WJ4XFLS4.js";var c=166667e-10,N=.0001333336,I=new Set(["serp","fb_search","fb_ad","instagram"]),p=.00111,L=4,R=4e-4;function o(e,r){let t=process.env[e]?.trim();if(!t)return r;let s=Number(t);return Number.isFinite(s)&&s>=0?s:r}var A=o("NANGO_USD_PER_CONNECTION_MONTH",1),U=o("NANGO_USD_PER_FUNCTION_RUN",1e-4),m=o("NANGO_USD_PER_PROXY_REQUEST",1e-4),g=o("NANGO_USD_PER_COMPUTE_SEC",2e-4),O=o("\u0042\u0052\u0049\u0047\u0048\u0054\u0044\u0041\u0054\u0041_BROWSER_USD_PER_GB",8),D=o("\u0042\u0052\u0049\u0047\u0048\u0054\u0044\u0041\u0054\u0041_SERP_USD_PER_REQUEST",.002836168);function u(e,r){return Math.max(0,e)/1e3*(r?N:c)}function d(e,r){return e==="fal_wizper"?Math.max(0,r)/L*p:e==="deepinfra_qwen"?Math.max(0,r)/1e3*R:e==="openrouter"||e==="mcp_memory_video"||e==="mcp_memory_ai"?Math.max(0,r):e==="nango_connection"?Math.max(0,r)*A:e==="nango_function_run"?Math.max(0,r)*U:e==="nango_proxy_request"?Math.max(0,r)*m:e==="nango_compute"?Math.max(0,r)*g:e==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061_browser_api"?Math.max(0,r)/1e9*O:0}import{randomUUID as i}from"crypto";var E=!1,n=null;async function T(){if(!E)return n||(n=S().finally(()=>{n=null}),n)}async function S(){let e=_(),r=await e.execute(`
2
2
  SELECT
3
- (SELECT COUNT(*) FROM sqlite_master WHERE type = 'table' AND name IN ('kernel_session_log', 'vendor_usage_log', 'cost_probe_runs')) = 3
4
- AND (SELECT COUNT(*) FROM pragma_table_info('kernel_session_log') WHERE name IN ('proxy_source', 'proxy_type', 'method')) = 3
3
+ (SELECT COUNT(*) FROM sqlite_master WHERE type = 'table' AND name IN ('\u006b\u0065\u0072\u006e\u0065\u006c_session_log', 'vendor_usage_log', 'cost_probe_runs')) = 3
4
+ AND (SELECT COUNT(*) FROM pragma_table_info('\u006b\u0065\u0072\u006e\u0065\u006c_session_log') WHERE name IN ('proxy_source', 'proxy_type', 'method')) = 3
5
5
  AND (SELECT COUNT(*) FROM pragma_table_info('vendor_usage_log') WHERE name IN ('method', 'source_key', 'provider_duration_ms', 'provider_captcha', 'provider_status')) = 5
6
6
  AND (SELECT COUNT(*) FROM sqlite_master WHERE type = 'index' AND name = 'vendor_usage_log_vendor_source_key') = 1
7
7
  AND (SELECT COUNT(*) FROM pragma_table_info('cost_probe_runs') WHERE name IN ('units', 'unit_type', 'mode')) = 3
8
8
  AS ready
9
9
  `);if(Number(r.rows[0]?.ready??0)===1){E=!0;return}await e.execute(`
10
- CREATE TABLE IF NOT EXISTS kernel_session_log (
10
+ CREATE TABLE IF NOT EXISTS \u006b\u0065\u0072\u006e\u0065\u006c_session_log (
11
11
  id TEXT PRIMARY KEY,
12
- kernel_session_id TEXT,
12
+ \u006b\u0065\u0072\u006e\u0065\u006c_session_id TEXT,
13
13
  op TEXT,
14
14
  source TEXT NOT NULL,
15
15
  probe_run_id TEXT,
@@ -26,7 +26,7 @@ import{c as l}from"./chunk-6DTXIZY2.js";import{t as _}from"./chunk-WJ4XFLS4.js";
26
26
  error TEXT,
27
27
  created_at TEXT NOT NULL DEFAULT (datetime('now'))
28
28
  )
29
- `),await e.execute("CREATE INDEX IF NOT EXISTS kernel_session_log_op ON kernel_session_log(op)"),await e.execute("CREATE INDEX IF NOT EXISTS kernel_session_log_probe ON kernel_session_log(probe_run_id)"),await e.execute("CREATE INDEX IF NOT EXISTS kernel_session_log_created ON kernel_session_log(created_at)");try{await e.execute("ALTER TABLE kernel_session_log ADD COLUMN proxy_source TEXT")}catch{}try{await e.execute("ALTER TABLE kernel_session_log ADD COLUMN proxy_type TEXT")}catch{}try{await e.execute("ALTER TABLE kernel_session_log ADD COLUMN method TEXT")}catch{}await e.execute(`
29
+ `),await e.execute("CREATE INDEX IF NOT EXISTS \u006b\u0065\u0072\u006e\u0065\u006c_session_log_op ON \u006b\u0065\u0072\u006e\u0065\u006c_session_log(op)"),await e.execute("CREATE INDEX IF NOT EXISTS \u006b\u0065\u0072\u006e\u0065\u006c_session_log_probe ON \u006b\u0065\u0072\u006e\u0065\u006c_session_log(probe_run_id)"),await e.execute("CREATE INDEX IF NOT EXISTS \u006b\u0065\u0072\u006e\u0065\u006c_session_log_created ON \u006b\u0065\u0072\u006e\u0065\u006c_session_log(created_at)");try{await e.execute("ALTER TABLE \u006b\u0065\u0072\u006e\u0065\u006c_session_log ADD COLUMN proxy_source TEXT")}catch{}try{await e.execute("ALTER TABLE \u006b\u0065\u0072\u006e\u0065\u006c_session_log ADD COLUMN proxy_type TEXT")}catch{}try{await e.execute("ALTER TABLE \u006b\u0065\u0072\u006e\u0065\u006c_session_log ADD COLUMN method TEXT")}catch{}await e.execute(`
30
30
  CREATE TABLE IF NOT EXISTS vendor_usage_log (
31
31
  id TEXT PRIMARY KEY,
32
32
  op TEXT,
@@ -54,20 +54,20 @@ import{c as l}from"./chunk-6DTXIZY2.js";import{t as _}from"./chunk-WJ4XFLS4.js";
54
54
  success INTEGER NOT NULL DEFAULT 0,
55
55
  error TEXT,
56
56
  http_only INTEGER NOT NULL DEFAULT 0,
57
- kernel_used INTEGER NOT NULL DEFAULT 0,
58
- kernel_sessions INTEGER NOT NULL DEFAULT 0,
59
- kernel_seconds_total REAL NOT NULL DEFAULT 0,
60
- kernel_tier_observed TEXT,
61
- est_kernel_cost_usd_headless REAL NOT NULL DEFAULT 0,
62
- est_kernel_cost_usd_headful REAL NOT NULL DEFAULT 0,
57
+ \u006b\u0065\u0072\u006e\u0065\u006c_used INTEGER NOT NULL DEFAULT 0,
58
+ \u006b\u0065\u0072\u006e\u0065\u006c_sessions INTEGER NOT NULL DEFAULT 0,
59
+ \u006b\u0065\u0072\u006e\u0065\u006c_seconds_total REAL NOT NULL DEFAULT 0,
60
+ \u006b\u0065\u0072\u006e\u0065\u006c_tier_observed TEXT,
61
+ est_\u006b\u0065\u0072\u006e\u0065\u006c_cost_usd_headless REAL NOT NULL DEFAULT 0,
62
+ est_\u006b\u0065\u0072\u006e\u0065\u006c_cost_usd_headful REAL NOT NULL DEFAULT 0,
63
63
  vendor_cost_usd REAL NOT NULL DEFAULT 0,
64
64
  charged_credits REAL NOT NULL DEFAULT 0,
65
65
  margin_usd_headless_starter REAL NOT NULL DEFAULT 0,
66
66
  margin_usd_headful_starter REAL NOT NULL DEFAULT 0,
67
67
  notes TEXT
68
68
  )
69
- `),await e.execute("CREATE INDEX IF NOT EXISTS cost_probe_runs_tool ON cost_probe_runs(tool)");try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN units REAL")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN unit_type TEXT")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN mode TEXT")}catch{}E=!0}async function f(e){try{await T();let r=l(),t=Math.max(0,e.closedAtMs-e.openedAtMs);await _().execute({sql:`INSERT INTO kernel_session_log
70
- (id, kernel_session_id, op, source, probe_run_id, user_id, stealth, headless_sent, proxy_used, proxy_source, proxy_type, fallback, opened_at, closed_at, duration_ms, est_cost_usd_headless, est_cost_usd_headful, error, method)
71
- VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[i(),e.kernelSessionId??null,r?.op??null,e.source,r?.probeRunId??null,r?.userId??null,a(e.stealth),a(e.headlessSent),a(e.proxyUsed),e.proxySource??null,e.proxyType??null,e.fallback?1:0,new Date(e.openedAtMs).toISOString(),new Date(e.closedAtMs).toISOString(),t,u(t,!1),u(t,!0),e.error??null,r?.subOp??null]})}catch(r){console.warn("[cost-telemetry] recordKernelSession failed:",r instanceof Error?r.message:String(r))}}async function v(e){try{await T();let r=l();return(await _().execute({sql:`INSERT OR IGNORE INTO vendor_usage_log
69
+ `),await e.execute("CREATE INDEX IF NOT EXISTS cost_probe_runs_tool ON cost_probe_runs(tool)");try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN units REAL")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN unit_type TEXT")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN mode TEXT")}catch{}E=!0}async function f(e){try{await T();let r=l(),t=Math.max(0,e.closedAtMs-e.openedAtMs);await _().execute({sql:`INSERT INTO \u006b\u0065\u0072\u006e\u0065\u006c_session_log
70
+ (id, \u006b\u0065\u0072\u006e\u0065\u006c_session_id, op, source, probe_run_id, user_id, stealth, headless_sent, proxy_used, proxy_source, proxy_type, fallback, opened_at, closed_at, duration_ms, est_cost_usd_headless, est_cost_usd_headful, error, method)
71
+ VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[i(),e.\u006b\u0065\u0072\u006e\u0065\u006cSessionId??null,r?.op??null,e.source,r?.probeRunId??null,r?.userId??null,a(e.stealth),a(e.headlessSent),a(e.proxyUsed),e.proxySource??null,e.proxyType??null,e.fallback?1:0,new Date(e.openedAtMs).toISOString(),new Date(e.closedAtMs).toISOString(),t,u(t,!1),u(t,!0),e.error??null,r?.subOp??null]})}catch(r){console.warn("[cost-telemetry] record\u004b\u0065\u0072\u006e\u0065\u006cSession failed:",r instanceof Error?r.message:String(r))}}async function v(e){try{await T();let r=l();return(await _().execute({sql:`INSERT OR IGNORE INTO vendor_usage_log
72
72
  (id, op, probe_run_id, user_id, vendor, model, units, unit_type, est_cost_usd, error, method, source_key, provider_duration_ms, provider_captcha, provider_status)
73
73
  VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[i(),e.op??r?.op??null,e.probeRunId??r?.probeRunId??null,e.userId??r?.userId??null,e.vendor,e.model??null,e.units,e.unitType,d(e.vendor,e.units),e.error??null,e.method??r?.subOp??null,e.sourceKey??null,e.providerDurationMs??null,a(e.providerCaptcha),e.providerStatus??null]})).rowsAffected===1}catch(r){return console.warn("[cost-telemetry] recordVendorUsage failed:",r instanceof Error?r.message:String(r)),!1}}function a(e){return e==null?null:e?1:0}export{c as a,N as b,I as c,u as d,d as e,T as f,f as g,v as h};