mcp-scraper 0.34.1 → 0.36.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/README.md +10 -9
  2. package/dist/bin/api-server.cjs +24342 -17931
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +3 -3
  5. package/dist/bin/mcp-scraper-cli.cjs +51 -7
  6. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  7. package/dist/bin/mcp-scraper-cli.js +48 -5
  8. package/dist/bin/mcp-scraper-cli.js.map +1 -1
  9. package/dist/bin/mcp-scraper-install.cjs +2 -2
  10. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  11. package/dist/bin/mcp-scraper-install.js +2 -2
  12. package/dist/bin/mcp-stdio-server.cjs +2997 -2070
  13. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  14. package/dist/bin/mcp-stdio-server.js +8 -8
  15. package/dist/bin/paa-harvest.cjs +125 -70
  16. package/dist/bin/paa-harvest.cjs.map +1 -1
  17. package/dist/bin/paa-harvest.js +4 -4
  18. package/dist/chunk-345BQXZH.js +712 -0
  19. package/dist/chunk-345BQXZH.js.map +1 -0
  20. package/dist/{chunk-M2S27J6Z.js → chunk-44HZLHDV.js} +10 -1
  21. package/dist/chunk-44HZLHDV.js.map +1 -0
  22. package/dist/chunk-4HO66323.js +7 -0
  23. package/dist/chunk-4HO66323.js.map +1 -0
  24. package/dist/{chunk-BWXLTWF7.js → chunk-4ZIJ3BKZ.js} +6 -4
  25. package/dist/chunk-4ZIJ3BKZ.js.map +1 -0
  26. package/dist/{chunk-6OPHG76G.js → chunk-5RULXBJ7.js} +3166 -2322
  27. package/dist/chunk-5RULXBJ7.js.map +1 -0
  28. package/dist/chunk-AN3VQARU.js +684 -0
  29. package/dist/chunk-AN3VQARU.js.map +1 -0
  30. package/dist/{chunk-XVVNKASZ.js → chunk-ANCGXUQJ.js} +118 -73
  31. package/dist/chunk-ANCGXUQJ.js.map +1 -0
  32. package/dist/{chunk-3HBPKR5G.js → chunk-D7LM5QZN.js} +3 -3
  33. package/dist/{chunk-4ZB3X6BQ.js → chunk-E5UEELA7.js} +16 -2
  34. package/dist/{chunk-4ZB3X6BQ.js.map → chunk-E5UEELA7.js.map} +1 -1
  35. package/dist/{chunk-ZID3WQID.js → chunk-FQI5PFE7.js} +9 -71
  36. package/dist/chunk-FQI5PFE7.js.map +1 -0
  37. package/dist/chunk-G3P3ZDB4.js +69 -0
  38. package/dist/chunk-G3P3ZDB4.js.map +1 -0
  39. package/dist/{chunk-7AYRWAEK.js → chunk-G7KAVJ3F.js} +2 -2
  40. package/dist/{chunk-7AYRWAEK.js.map → chunk-G7KAVJ3F.js.map} +1 -1
  41. package/dist/{chunk-62DQAWPF.js → chunk-IFYER7O4.js} +367 -42
  42. package/dist/chunk-IFYER7O4.js.map +1 -0
  43. package/dist/chunk-O2MCWFXQ.js +499 -0
  44. package/dist/chunk-O2MCWFXQ.js.map +1 -0
  45. package/dist/{chunk-JWIE5NCR.js → chunk-QZXKQB7Y.js} +140 -10
  46. package/dist/chunk-QZXKQB7Y.js.map +1 -0
  47. package/dist/{db-YAI5AQOI.js → db-N6MPVMEF.js} +8 -2
  48. package/dist/extract-bundle-M4SDJG3V.js +568 -0
  49. package/dist/extract-bundle-M4SDJG3V.js.map +1 -0
  50. package/dist/index.cjs +129 -70
  51. package/dist/index.cjs.map +1 -1
  52. package/dist/index.d.cts +11 -0
  53. package/dist/index.d.ts +11 -0
  54. package/dist/index.js +4 -4
  55. package/dist/location-data-repository-Z4NQOU5Y.js +35 -0
  56. package/dist/{server-IWDHTES2.js → server-ZIGAFKOL.js} +10844 -7480
  57. package/dist/server-ZIGAFKOL.js.map +1 -0
  58. package/dist/site-extract-repository-PGQZNW6V.js +62 -0
  59. package/dist/site-extract-repository-PGQZNW6V.js.map +1 -0
  60. package/dist/{worker-645BZPEK.js → worker-EBB6CTGW.js} +7 -7
  61. package/docs/hosted-location-data.md +108 -0
  62. package/docs/mcp-tool-craft-lint.generated.md +6 -3
  63. package/docs/mcp-tool-manifest.generated.json +2073 -551
  64. package/docs/mcp-tool-quality-spec.md +1 -1
  65. package/docs/specs/connected-services-control-plane-decoupling-spec.md +1044 -0
  66. package/docs/specs/kernel-stealth-captcha-test-matrix.md +278 -0
  67. package/docs/specs/multimodal-image-memory-architecture-spec.md +1022 -0
  68. package/docs/specs/query-fanout-transport-contract-fix.md +9 -1
  69. package/docs/specs/unified-credit-and-scheduled-execution-billing-spec.md +36 -27
  70. package/package.json +6 -5
  71. package/dist/chunk-5PZ6N2QM.js +0 -130
  72. package/dist/chunk-5PZ6N2QM.js.map +0 -1
  73. package/dist/chunk-62DQAWPF.js.map +0 -1
  74. package/dist/chunk-6OPHG76G.js.map +0 -1
  75. package/dist/chunk-7N2KYL4U.js +0 -7
  76. package/dist/chunk-7N2KYL4U.js.map +0 -1
  77. package/dist/chunk-BWXLTWF7.js.map +0 -1
  78. package/dist/chunk-JWIE5NCR.js.map +0 -1
  79. package/dist/chunk-M2S27J6Z.js.map +0 -1
  80. package/dist/chunk-R7EETU7Z.js +0 -419
  81. package/dist/chunk-R7EETU7Z.js.map +0 -1
  82. package/dist/chunk-XVVNKASZ.js.map +0 -1
  83. package/dist/chunk-ZID3WQID.js.map +0 -1
  84. package/dist/extract-bundle-GPZRLBVY.js +0 -331
  85. package/dist/extract-bundle-GPZRLBVY.js.map +0 -1
  86. package/dist/server-IWDHTES2.js.map +0 -1
  87. package/dist/site-extract-repository-OLVWMOU2.js +0 -30
  88. /package/dist/{chunk-3HBPKR5G.js.map → chunk-D7LM5QZN.js.map} +0 -0
  89. /package/dist/{db-YAI5AQOI.js.map → db-N6MPVMEF.js.map} +0 -0
  90. /package/dist/{site-extract-repository-OLVWMOU2.js.map → location-data-repository-Z4NQOU5Y.js.map} +0 -0
  91. /package/dist/{worker-645BZPEK.js.map → worker-EBB6CTGW.js.map} +0 -0
package/README.md CHANGED
@@ -39,7 +39,9 @@ npx -y -p mcp-scraper@latest mcp-scraper-cli agent install codex
39
39
  npx -y -p mcp-scraper@latest mcp-scraper-cli agent prompt agent-packet
40
40
  ```
41
41
 
42
- `agent install claude --apply` upserts the Claude Code user-scope `mcp-scraper` entry to `npx -y -p mcp-scraper@latest mcp-scraper`. Fully exit Claude Code and open a new Claude terminal after applying; MCP servers are attached when Claude starts.
42
+ `agent install claude --apply` upserts the Claude Code user-scope `mcp-scraper` entry to `npx -y --package mcp-scraper@latest mcp-scraper`. Fully exit Claude Code and open a new Claude terminal after applying; MCP servers are attached when Claude starts.
43
+
44
+ The registered command uses the long `--package` flag deliberately. Claude Code's `mcp add` leaks short flags that appear after `--` back into its own option parsing, so a registered `-p` makes it reject its own `--scope`/`-s` argument with a misleading `unknown option` error. If the registration ever fails, the previous entry is captured beforehand and restored automatically.
43
45
 
44
46
  Check usage and upgrade concurrency from a normal terminal:
45
47
 
@@ -152,9 +154,10 @@ env = { MCP_SCRAPER_API_KEY = "sk_live_your_key" }
152
154
 
153
155
  - `harvest_paa`
154
156
  - `search_serp`
155
- - `extract_url`
157
+ - `extract_url` — extract normal or Wayback-replayed page copy; Wayback results omit playback chrome and can include a timestamp-matched featured image.
156
158
  - `map_site_urls`
157
- - `extract_site`
159
+ - `map_wayback_snapshots` — count and inventory Wayback captures across an inclusive date range without downloading page bodies. Supports exact pages, prefixes, hosts, domains, or selected URLs; reports exact versus lower-bound counts, unique URLs/content digests, monthly coverage, missing months, and optional timestamp rows.
160
+ - `extract_site` — crawl a live site, batch one archived site snapshot from a Wayback replay URL, or pass a `wayback` plan for whole-site, single-page, or selected-page timelines across explicit months or a `from`/`to` range. Timeline ZIPs include month folders and a capture matrix.
158
161
  - `youtube_harvest`
159
162
  - `youtube_transcribe`
160
163
  - `facebook_ad_search`
@@ -165,7 +168,7 @@ env = { MCP_SCRAPER_API_KEY = "sk_live_your_key" }
165
168
  - `instagram_media_download` — extract and download one Instagram post/reel/tv URL, optionally through a saved hosted browser `profile` for authenticated access. Returns text/caption, image URL/downloads, selected video/audio MP4 tracks, optional muxed MP4 when `ffmpeg` is available, optional transcript, and browser details.
166
169
  - `maps_search` — search Google's localized local-results list for multiple business/profile candidates. Use for GMB/GBP prospect lists, competitors, categories, and anything needing more than the Google 3-pack. It opens the rendered business card, reads the profile dialog, then closes it before continuing to the next ranked card. Set `includeServices: true` to return services and areas served without collecting review cards. `maxResults` defaults to 10 and is capped at 50.
167
170
  - `maps_place_intel` — hydrate one known/named Google Maps business with profile details and optional reviews. Use after `maps_search` when a selected candidate needs full details.
168
- - `directory_workflow` — build city-by-city directory/prospecting datasets from Census place selection plus localized Google business searches. Use it for requests like "all cities over 100k population in Tennessee, then get 20 roofers from Maps." The default direct route uses city-in-query plus UULE, `gl`, and `hl`; residential location proxying remains available only as an explicit override. The saved CSV includes `source_location`, `result_position`, `business_name`, `review_stars`, `review_count`, `category`, `address`, `phone`, `hours_status`, `website_url`, `directions_url`, `place_url`, `cid`, `cid_decimal`, Census population, and ZIP groups.
171
+ - `directory_workflow` — build city-by-city directory/prospecting datasets from Census place selection plus localized Google business searches. Use it for requests like "all cities over 100k population in Tennessee, then get 20 roofers from Maps." Supply the business category, state, and market limits; MCP Scraper manages search transport and retry behavior internally. The saved CSV includes `source_location`, `result_position`, `business_name`, `review_stars`, `review_count`, `category`, `address`, `phone`, `hours_status`, `website_url`, `directions_url`, `place_url`, `cid`, `cid_decimal`, Census population, and ZIP groups.
169
172
  - `workflow_list` — list higher-level workflow IDs plus AI-facing recipes for market analysis, ICP research, forum/review acquisition, brand design briefings, CRO audits, positioning briefs, content gaps, and AI search visibility audits.
170
173
  - `workflow_suggest` — route a high-level business goal to the right workflow/tool chain before spending credits.
171
174
  - `workflow_run` — run hosted workflows such as `agent-packet`, `local-competitive-audit`, `map-comparison`, `serp-comparison`, `paa-expansion-brief`, and `ai-overview-language`; returns run metadata, summary, and artifact IDs.
@@ -178,7 +181,7 @@ env = { MCP_SCRAPER_API_KEY = "sk_live_your_key" }
178
181
 
179
182
  - `list_service_connections` — list this caller's tenant-owned Nango OAuth and official remote MCP connections, including verified provider-side account email/name when exposed, exact live reads, gated actions, permanently blocked administrative tools, credential transport, and schema-discovery metadata. Provider identity is distinct from the MCP Scraper login, and connections are never shared between customers.
180
183
  - `describe_service_connection_tool` — fetch the sanitized live MCP Tool definition for one tool listed on one tenant-owned connection, including its current callability, input schema, optional output schema, safe annotations, and schema hash. Use this before constructing provider-native arguments; provider functions stay behind the generic bridges instead of becoming dozens of permanent top-level tools.
181
- - `export_connected_service_data` — fetch a fresh bounded Gmail, Google Calendar, Google Search Console, Zoom, Resend, or Meta time range in one MCP call. Search Console's `search_console_performance` dataset walks accessible properties and bounded live Search Analytics pages with signed continuation. Small exports return inline; larger exports become private JSONL retained for seven days with a 15-minute signed URL. For relationship work, gather source evidence first: inspect existing People records, resolve the exact provider account, use RFC3339 `from`/`to` for Gmail ranges longer than 90 days, preserve provider provenance when writing a linked Communication, and never treat an export as permission to mutate the source account.
184
+ - `export_connected_service_data` — fetch a fresh bounded Gmail, Google Calendar, Google Search Console, Zoom, Resend, or Meta time range in one MCP call. Zoom's `zoom_transcripts` dataset resolves VTT files from recording metadata and downloads them through the authenticated connection without looping the separately rate-limited `get-meeting-transcript` function. Search Console's `search_console_performance` dataset walks accessible properties and bounded live Search Analytics pages with signed continuation. Small exports return inline; larger exports become private JSONL retained for seven days with a 15-minute signed URL. For relationship work, gather source evidence first: inspect existing People records, resolve the exact provider account, use RFC3339 `from`/`to` for Gmail ranges longer than 90 days, preserve provider provenance when writing a linked Communication, and never treat an export as permission to mutate the source account.
182
185
  - `export_search_console_table_data` — filter up to 50,000 Search Console rows already persisted by a scheduled `connection_sync` and create a private renewable JSONL artifact without calling Google again. Get the typed `gsc_performance_*` table name from `list_service_connections`, inspect it with `table-describe`, and use the same filters with `table-query` for interactive analysis.
183
186
  - `renew_connected_data_download` — issue a fresh 15-minute signed URL for an unexpired private export artifact without pulling the provider again.
184
187
  - `read_service_connection` — run one small live read by exact allowlisted name across Nango OAuth or official remote MCP connections, including bounded Google Drive inventory, change, Doc, Sheet, and text-file tools. Do not loop it over a time range when `export_connected_service_data` supports that provider's collection.
@@ -219,15 +222,13 @@ Google Search Console exposes eight bounded reads and eight gated property and s
219
222
 
220
223
  For accurate annotated videos, do not guess annotation times from a script. Start the replay, navigate until each target is visible and stable, call `browser_replay_mark` for each callout, then stop the replay and pass the returned annotations to `browser_replay_annotate` with the returned `source_width` and `source_height`.
221
224
 
222
- For general SERP tools (`harvest_paa`, `search_serp`, and hosted SERP capture), omit `proxyMode` for normal use. The default is `configured`, which uses the configured browser-service proxy without city/ZIP targeting for the highest general success rate. Use `proxyMode: "location"` only when the user explicitly needs city/ZIP-targeted residential proxy evidence. When Google shows a CAPTCHA/challenge, browser-service sessions briefly wait for automatic challenge solving first, then rotate to a new proxy/session if the challenge does not clear. Proxy tunnel failure and wrong-location evidence are also retryable before returning.
223
-
224
- For Google Maps tools (`maps_search` and `directory_workflow`), leave `proxyMode` unset for the default direct Google route. Localization is carried by the city in the query, UULE, `gl`, and `hl`; do not treat proxy geolocation as the source of truth. Use `proxyMode: "location"` only when an explicit residential-proxy experiment or compatibility path needs it. Retryable failures open a new browser session; location mode can additionally rotate its disposable proxy. Successful structured responses include sanitized attempt telemetry without exposing full proxy or browser IDs.
225
+ For Google SERP and Maps tools, callers provide the query, two-letter country code (`gl`), language (`hl`), device when relevant, and an optional city or region. MCP Scraper owns transport selection, anti-bot handling, and bounded retries internally; those implementation controls and receipts are not part of the public tool contract.
225
226
 
226
227
  The `mcp-scraper` server (and the MCPB bundle, which runs it) exposes both sections through one MCP server.
227
228
 
228
229
  All MCP tools expose output schemas and return `structuredContent` with the IDs, URLs, CSV paths, transcripts, browser session handles, replay paths, artifacts, recipe fields, or blueprint fields needed by the next step. Browser Agent tools keep a JSON text block for older clients, but structured data is the primary contract. All tools carry MCP annotations; file-writing tools such as replay downloads and annotations state their filesystem side effects.
229
230
 
230
- The canonical tool inventory is generated at `docs/mcp-tool-manifest.generated.json`. Both the `mcp-scraper` stdio server and the hosted endpoint at `https://mcpscraper.dev/mcp` expose the same 167 tools: 78 scraper, browser, workflow, billing, and connected-service tools plus 89 durable-memory tools. Release verification compares the exact local and remote tool-name sets, not only the count.
231
+ The canonical tool inventory is generated at `docs/mcp-tool-manifest.generated.json`. The current local unified server exposes 170 tools: 81 scraper, browser, workflow, billing, and connected-service tools plus 89 durable-memory tools. Release verification compares the exact local and hosted tool-name sets, not only the count.
231
232
 
232
233
  For contract parity, stdio and MCPB memory calls invoke the matching public tool on the hosted MCP Scraper `/mcp` endpoint. The hosted aggregate runtime owns MCP Scraper-specific billing, scheduling, credential, and in-process cutover policy; its internal `/memory/mcp-call` bridge is a fallback to the standalone memory service, not the public stdio execution path. Direct `mcp-memory` OAuth and stdio clients continue to use `memory.mcpscraper.dev` and must be verified as a separate dependent release surface.
233
234