mcp-scraper 0.90.6 → 0.90.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -2
- package/README.md +2 -2
- package/dist/{analytics-repository-WNJESD4A.js → analytics-repository-IHOFBSUV.js} +1 -1
- package/dist/bin/api-server.js +1 -1
- package/dist/bin/mcp-scraper-cli.js +1 -1
- package/dist/bin/mcp-scraper-core.js +1 -1
- package/dist/bin/mcp-scraper-install.js +1 -1
- package/dist/bin/mcp-stdio-server.js +1 -1
- package/dist/bin/paa-harvest.js +1 -1
- package/dist/{chunk-MADTJXNI.js → chunk-2F35B4U5.js} +105 -105
- package/dist/{chunk-56UT5M5Y.js → chunk-2SP57VCG.js} +5 -5
- package/dist/{chunk-XARSC34L.js → chunk-53IEEOKA.js} +167 -167
- package/dist/chunk-6DTXIZY2.js +1 -0
- package/dist/{chunk-WDB5CTRY.js → chunk-6FMIMP5Z.js} +1 -1
- package/dist/chunk-7SHEBDEB.js +1 -0
- package/dist/{chunk-NWRRVZKW.js → chunk-DKXZIIA5.js} +1 -1
- package/dist/{chunk-XXK5PLRB.js → chunk-IK5BG7MO.js} +9 -8
- package/dist/{chunk-AQSMJCTF.js → chunk-J374DIKA.js} +1 -1
- package/dist/{chunk-AAJG5OHF.js → chunk-KZV2FLGG.js} +1 -1
- package/dist/{chunk-4OTWTGKC.js → chunk-LI7WHOII.js} +1 -1
- package/dist/{chunk-BVSK6YHX.js → chunk-LSFDB5XE.js} +1 -1
- package/dist/{chunk-5MMN27Q6.js → chunk-OIMPG6M6.js} +1 -1
- package/dist/{chunk-YDOBZMLS.js → chunk-OMYTA34L.js} +1 -1
- package/dist/chunk-ORB4RHCK.js +63 -0
- package/dist/chunk-OVTHAIDS.js +1 -0
- package/dist/{chunk-Q5SUDDSQ.js → chunk-QTDLZTQ7.js} +1 -1
- package/dist/chunk-RK2VCTZI.js +1 -0
- package/dist/{chunk-KKHTMYCD.js → chunk-WJ4XFLS4.js} +160 -115
- package/dist/{chunk-COIA4RW4.js → chunk-X7GZJU5P.js} +1 -1
- package/dist/{chunk-D6BTAT5Y.js → chunk-YGBTTW5D.js} +1 -1
- package/dist/{chunk-AQXYXVOP.js → chunk-ZMNUI5LB.js} +7 -7
- package/dist/db-B5XJTOGN.js +1 -0
- package/dist/{extract-bundle-CP6WBOJI.js → extract-bundle-II3RPATK.js} +10 -10
- package/dist/gmail-service-E5KORS6C.js +1 -0
- package/dist/index.cjs +17 -17
- package/dist/index.js +1 -1
- package/dist/{lead-list-enrichment-repository-AIMOP7YM.js → lead-list-enrichment-repository-AWZZMY2H.js} +1 -1
- package/dist/{location-data-repository-25D3GHXE.js → location-data-repository-6XAVI65Z.js} +1 -1
- package/dist/operation-metering-DB5F7SXN.js +1 -0
- package/dist/server-MI6XN25Z.js +4961 -0
- package/dist/{site-extract-repository-KG6FQU55.js → site-extract-repository-JHSVJJ62.js} +1 -1
- package/dist/stripe-event-worker-7KNF3VH5.js +1 -0
- package/dist/worker-3VTFFYHH.js +1 -0
- package/package.json +1 -1
- package/dist/chunk-NOUHD54C.js +0 -48
- package/dist/chunk-UM4WSLKQ.js +0 -1
- package/dist/db-XN7YPDGC.js +0 -1
- package/dist/gmail-service-5NH2GVQ5.js +0 -1
- package/dist/server-TQJ6Z3ZF.js +0 -4931
- package/dist/stripe-event-worker-ZNNQEVK4.js +0 -1
- package/dist/worker-4LFWAEX2.js +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,52 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.90.8] - 2026-09-24
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Record observed provider product, billing tier, usage units, and applied rate snapshots on normalized cost receipts; preserve older receipts through the schema migration.
|
|
12
|
+
- Link direct SERP page requests, Maps browser rotations, extraction sessions, shared site-crawl batches, and FAL jobs to operation attempts before provider contact.
|
|
13
|
+
- Store provider account-window totals and explicit unattributed variance without distributing the remainder across customer operations.
|
|
14
|
+
- Link transcription, Instagram media, and Maps place debit receipts to their operation runs, including additional duration charges.
|
|
15
|
+
- Record local caption, media, plain-page, Wayback, Maps detail, and export phases without creating duplicate provider charges.
|
|
16
|
+
- Start synchronous and durable site cost runs before discovery, and identify browser discovery retries under the site operation.
|
|
17
|
+
- Persist receipt-creation coverage defects when the database remains writable, so missing provider-cost evidence appears in operator reports.
|
|
18
|
+
- Check scoped provider metering calls against the generated method registry and reject unregistered methods during local verification.
|
|
19
|
+
|
|
20
|
+
### Changed
|
|
21
|
+
|
|
22
|
+
- Price measured browser bandwidth at the applicable standard or premium-domain tier, and record the selected tier and list snapshot on the provider receipt.
|
|
23
|
+
- Report receipt identity and known-cost closure separately. CEO and CTO cost views distinguish actual, measured, estimated, pending, and unavailable amounts, with product, method, and retry breakdowns.
|
|
24
|
+
- Keep existing wall-time estimates visible as estimates while waiting for provider usage evidence and accepted account rates.
|
|
25
|
+
- Attribute Facebook video fallback to the browser session it actually uses, and keep optional rendered capture as a separate browser session cost.
|
|
26
|
+
- Keep Instagram browser acquisition and optional FAL transcription under one operation run, with separate provider receipts.
|
|
27
|
+
- Count Maps services fallback sessions separately, and stop browser wall-time estimates at session close when a route continues with other work.
|
|
28
|
+
|
|
29
|
+
### Fixed
|
|
30
|
+
|
|
31
|
+
- Recover completed single-page extraction results from their original paid job: hosted and installed MCP status expose owner-scoped artifact readback, same-key replay keeps the job receipt, and dashboard history opens the durable job instead of a duplicate activity row.
|
|
32
|
+
- Keep the provider-receipt table rewrite out of hosted startup; releases now apply that constraint change through the explicit database migration command, while application instances fail closed when the operator migration is missing.
|
|
33
|
+
- Keep the five-minute and hourly maintenance passes within the hosted function's time limit when they start on a cold instance, so hourly billing, retention, and daily cleanup complete and record their runs instead of timing out.
|
|
34
|
+
- Stop creating a provider-usage index during hosted startup; the provider-usage lookback now uses the existing timestamp index instead.
|
|
35
|
+
- Stop the hourly job-retention sweep from holding the database write lock for minutes: expired completed and failed jobs are now found by insertion order instead of by scanning every job's stored result, which had stalled schema changes, cold starts, and scheduled maintenance for several minutes after each run.
|
|
36
|
+
|
|
37
|
+
## [0.90.7] - 2026-09-23
|
|
38
|
+
|
|
39
|
+
### Changed
|
|
40
|
+
|
|
41
|
+
- Run hosted background maintenance on schedules matched to how quickly each result matters: scheduled workflows and queued search jobs every minute, delivery and recovery checks every five minutes, and billing, retention, and cleanup work hourly or daily.
|
|
42
|
+
- Deliver submitted X-Ray forms, apply subscription changes from billing webhooks, and index saved Commons entities as soon as the request completes instead of waiting for the next scheduled pass; scheduled passes remain the retry path.
|
|
43
|
+
- Run daily cleanup once a full day has passed since its last successful run, so a single missed pass no longer postpones it for a day.
|
|
44
|
+
- Reconcile connected-account billing once a day per subscriber, while connecting or disconnecting an account still reconciles immediately.
|
|
45
|
+
|
|
46
|
+
### Fixed
|
|
47
|
+
|
|
48
|
+
- Restore X-Ray privacy retention, which had stopped running since 28 August, so expired analytics, identity, form, and delivery data is again deleted on its configured schedule.
|
|
49
|
+
- Stop the X-Ray scheduled-delivery backlog from growing and clear the accumulated backlog of stale scheduled work.
|
|
50
|
+
- Stop background recovery from repeatedly rechecking completed jobs and unrecoverable provider records, reducing wasted database and provider traffic.
|
|
51
|
+
- Report a scheduled-maintenance failure only when work actually fails, rather than whenever work is intentionally deferred to the next pass.
|
|
52
|
+
|
|
7
53
|
## [0.90.6] - 2026-09-22
|
|
8
54
|
|
|
9
55
|
### Added
|
|
@@ -1970,7 +2016,9 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
1970
2016
|
- Write actions remain unavailable until the account owner explicitly enables them.
|
|
1971
2017
|
- Provider-specific connection data is normalized into one agent-facing contract.
|
|
1972
2018
|
|
|
1973
|
-
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.
|
|
2019
|
+
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.8...HEAD
|
|
2020
|
+
[0.90.8]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.7...v0.90.8
|
|
2021
|
+
[0.90.7]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.6...v0.90.7
|
|
1974
2022
|
[0.88.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.87.0...v0.88.0
|
|
1975
2023
|
[0.87.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.86.5...v0.87.0
|
|
1976
2024
|
[0.86.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.85.0...v0.86.0
|
|
@@ -2115,7 +2163,6 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
2115
2163
|
[0.11.0]: https://github.com/VilovietaSEO/mcp-scraper/releases/tag/v0.11.0
|
|
2116
2164
|
|
|
2117
2165
|
[0.89.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.88.3...v0.89.0
|
|
2118
|
-
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.89.8...HEAD
|
|
2119
2166
|
[0.89.8]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.89.7...v0.89.8
|
|
2120
2167
|
[0.89.1]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.89.0...v0.89.1
|
|
2121
2168
|
[0.89.2]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.89.1...v0.89.2
|
package/README.md
CHANGED
|
@@ -175,7 +175,7 @@ Build the branded one-click bundle:
|
|
|
175
175
|
npm run build:mcpb
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
-
The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.90.
|
|
178
|
+
The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.90.8`, SHA-256 `2604bb96cf02bfc054c932408cc2006ab542808b0572815da2b38da6f8e61f85`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
|
|
179
179
|
|
|
180
180
|
The MCPB install exposes every tool — web-intelligence plus all `browser_*` tools — through the one `mcp-scraper` server.
|
|
181
181
|
|
|
@@ -252,7 +252,7 @@ Check `pagination.requestedPages`, `pagination.capturedPages`, and `pagination.p
|
|
|
252
252
|
- `harvest_paa` — expand People Also Ask on the original first result page. Optional `pages: 2` captures a second organic result page first; the default is one page.
|
|
253
253
|
- `harvest_paa_start` — start the same harvest as a durable job, with the same optional `pages: 2`.
|
|
254
254
|
- `search_serp`
|
|
255
|
-
- `extract_url` — start a durable normal or Wayback-replayed page extraction and receive a job receipt immediately; poll `extract_url_status` with the returned job ID until it reaches a terminal state. Wayback results omit playback chrome and can include a timestamp-matched featured image. Set `preserveMedia:true` to union static and rendered/lazy media, collapse responsive variants, attach up to `maxInlineImages` AI-readable images, and receive an owner-scoped ZIP manifest readable with `archive_read`. Branding output ranks the site logo separately from evidence-bounded proof images such as certifications, awards, memberships, partner/customer marks, and press mentions.
|
|
255
|
+
- `extract_url` — start a durable normal or Wayback-replayed page extraction and receive a job receipt immediately; poll `extract_url_status` with the returned job ID until it reaches a terminal state. For artifact delivery, use the returned `artifact.artifactId` with `report_artifact_read` to retrieve the full saved page in windows. Polling and reading do not start or bill another extraction. Hosted readback lasts while the completed job is retained (up to 45 days); an installed MCP client mirrors the page into its local artifact store for 24 hours from each successful status read. Wayback results omit playback chrome and can include a timestamp-matched featured image. Set `preserveMedia:true` to union static and rendered/lazy media, collapse responsive variants, attach up to `maxInlineImages` AI-readable images, and receive an owner-scoped ZIP manifest readable with `archive_read`. Branding output ranks the site logo separately from evidence-bounded proof images such as certifications, awards, memberships, partner/customer marks, and press mentions.
|
|
256
256
|
- `map_site_urls`
|
|
257
257
|
- `map_wayback_snapshots` — count and inventory Wayback captures across an inclusive date range without downloading page bodies. Supports exact pages, prefixes, hosts, domains, or selected URLs; reports exact versus lower-bound counts, unique URLs/content digests, monthly coverage, missing months, and optional timestamp rows.
|
|
258
258
|
- `extract_site` — crawl a live site, batch one archived site snapshot from a Wayback replay URL, or pass a `wayback` plan for whole-site, single-page, or selected-page timelines across explicit months or a `from`/`to` range. Timeline ZIPs include month folders and a capture matrix.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
import{$ as L,$a as La,A as k,Aa as ka,B as l,Ba as la,C as m,Ca as ma,D as n,Da as na,E as o,Ea as oa,F as p,Fa as pa,G as q,Ga as qa,H as r,Ha as ra,I as s,Ia as sa,J as t,Ja as ta,K as u,Ka as ua,L as v,La as va,M as w,Ma as wa,N as x,Na as xa,O as y,Oa as ya,P as z,Pa as za,Q as A,Qa as Aa,R as B,Ra as Ba,S as C,Sa as Ca,T as D,Ta as Da,U as E,Ua as Ea,V as F,Va as Fa,W as G,Wa as Ga,X as H,Xa as Ha,Y as I,Ya as Ia,Z as J,Za as Ja,_ as K,_a as Ka,aa as M,ab as Ma,ba as N,bb as Na,ca as O,da as P,ea as Q,fa as R,ga as S,ha as T,ia as U,ja as V,ka as W,la as X,ma as Y,na as Z,oa as _,pa as $,q as a,qa as aa,r as b,ra as ba,s as c,sa as ca,t as d,ta as da,u as e,ua as ea,v as f,va as fa,w as g,wa as ga,x as h,xa as ha,y as i,ya as ia,z as j,za as ja}from"./chunk-
|
|
1
|
+
import{$ as L,$a as La,A as k,Aa as ka,B as l,Ba as la,C as m,Ca as ma,D as n,Da as na,E as o,Ea as oa,F as p,Fa as pa,G as q,Ga as qa,H as r,Ha as ra,I as s,Ia as sa,J as t,Ja as ta,K as u,Ka as ua,L as v,La as va,M as w,Ma as wa,N as x,Na as xa,O as y,Oa as ya,P as z,Pa as za,Q as A,Qa as Aa,R as B,Ra as Ba,S as C,Sa as Ca,T as D,Ta as Da,U as E,Ua as Ea,V as F,Va as Fa,W as G,Wa as Ga,X as H,Xa as Ha,Y as I,Ya as Ia,Z as J,Za as Ja,_ as K,_a as Ka,aa as M,ab as Ma,ba as N,bb as Na,ca as O,da as P,ea as Q,fa as R,ga as S,ha as T,ia as U,ja as V,ka as W,la as X,ma as Y,na as Z,oa as _,pa as $,q as a,qa as aa,r as b,ra as ba,s as c,sa as ca,t as d,ta as da,u as e,ua as ea,v as f,va as fa,w as g,wa as ga,x as h,xa as ha,y as i,ya as ia,z as j,za as ja}from"./chunk-LI7WHOII.js";import"./chunk-QTDLZTQ7.js";import"./chunk-WJ4XFLS4.js";export{P as ANALYTICS_CONTENT_SORTS,a as AnalyticsRepositoryError,B as ENGAGED_SESSION_MS,A as MAX_ENGAGED_MS,N as analyticsAcquisition,l as analyticsBusinessMetrics,O as analyticsChannelBreakdown,R as analyticsContent,T as analyticsConversions,Ma as analyticsCsvCell,V as analyticsDimensions,S as analyticsEventCounts,m as analyticsForecast,Ja as analyticsHealth,x as analyticsIdentityPromotionAllowed,w as analyticsIdentityResolutionAllowed,L as analyticsOverview,U as analyticsPaths,M as analyticsTimeseries,H as appendAnalyticsAuthoritativeOutcomeVersion,ta as archiveAnalyticsActivationDestination,Y as archiveAnalyticsCampaignLink,ga as assignAnalyticsIdentityNode,fa as backfillAnalyticsConfirmedHistory,oa as claimAnalyticsCrmImportRows,Ea as claimAnalyticsFormDeliveryJobs,c as closeAnalyticsPool,pa as completeAnalyticsCrmImportRow,Ga as completeAnalyticsFormBridgeDelivery,Fa as completeAnalyticsFormDelivery,J as consumeAnalyticsSurveyInvite,ra as createAnalyticsActivationDestination,W as createAnalyticsCampaignLink,K as createAnalyticsConversion,ma as createAnalyticsCrmImport,Na as createAnalyticsExport,_ as createAnalyticsForm,o as createAnalyticsPixel,h as createAnalyticsSite,qa as deferAnalyticsCrmImportRow,Ha as deferAnalyticsFormDelivery,n as deleteAnalyticsSite,ea as deterministicAnalyticsCrmEntityId,ja as enrichAnalyticsExistingCrmIdentity,ua as getAnalyticsActivationDestinationConnectionRef,la as getAnalyticsPersonJourney,b as getAnalyticsPool,aa as getPublicAnalyticsForm,da as identityHmac,F as ingestAnalyticsEvents,G as insertAnalyticsRevenueSetupRevision,Ia as isAnalyticsFormPlacementApproved,ia as linkAnalyticsFormIdentity,ha as linkAnalyticsIdentityInTransaction,sa as listAnalyticsActivationDestinations,xa as listAnalyticsActivationReceipts,X as listAnalyticsCampaignLinks,na as listAnalyticsCrmImports,$ as listAnalyticsForms,t as listAnalyticsHostGroups,ka as listAnalyticsPeople,p as listAnalyticsPixels,i as listAnalyticsSites,d as migrateAnalytics,C as normalizeAnalyticsPath,D as normalizeAnalyticsUrl,Q as normalizeContentOptions,e as normalizeObservedHostname,za as pollAnalyticsActivationDiagnostics,u as prepareAnalyticsLinkerIssue,I as projectAnalyticsAuthoritativeConversion,y as projectAnalyticsPixelEventConsent,Aa as queueAnalyticsActivation,Ca as queueAnalyticsFormBridgeTransaction,Ba as queueAnalyticsFormDelivery,ba as recordAnalyticsFormSubmission,v as redeemAnalyticsLinkerRecord,Ka as refreshAnalyticsDailyRollups,La as refreshAnalyticsDailyRollupsIfDue,f as requireAnalyticsAccess,g as requireAnalyticsEditor,Z as resolveAnalyticsCampaignLink,z as resolveAnalyticsConfirmedActivationIdentity,ya as retryAnalyticsActivationJob,E as sanitizeAnalyticsProperties,ca as sanitizeClickIds,va as setAnalyticsActivationReadiness,r as setAnalyticsPixelDomainState,Da as sweepAnalyticsRestrictedRetention,wa as testAnalyticsActivationDestination,j as updateAnalyticsBusinessModel,q as updateAnalyticsPixel,k as upsertAnalyticsAdSpend,s as upsertAnalyticsHostGroup};
|
package/dist/bin/api-server.js
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import{readFileSync as s}from"fs";function c(){try{for(let r of s(".env","utf8").split(`
|
|
3
|
-
`)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-
|
|
3
|
+
`)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-MI6XN25Z.js"),import("../worker-3VTFFYHH.js"),import("../db-B5XJTOGN.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-IHOFBSUV.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-JHMI6HEO.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-2TZBO52D.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import{a as P}from"../chunk-
|
|
2
|
+
import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-JHMI6HEO.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-2TZBO52D.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import{a as P}from"../chunk-7SHEBDEB.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
|
|
3
3
|
`)){let l=a.match(/^\s{2,}([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);if(!l){if(a.trim().length&&!/^\s{2,}/.test(a))break;continue}p[l[1]]=l[2]}return{command:n,args:i,env:p}}function j(e){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(e.env))n.push("--env",`${c}=${i}`);return n.push("--",e.command,...e.args),n}function G(e={}){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(pe(e)))n.push("--env",`${c}=${i}`);return n.push("--","npx",...A(e)),n}function O(e){if(e==="claude-code")return"claude";if(e==="claude"||D.hosts.some(n=>n.id===e))return e;throw new Error('Unknown host "'+e+'". Use: codex, claude, claude-code, claude-desktop, cursor, windsurf, cline, or user-action-only')}function ue(e){return K(e==="claude"?"claude-code":e)}function W(e,n={}){let c=O(e),i=ue(c),p="Restart the MCP client so it starts a fresh npx process.",u='MCP_SCRAPER_API_KEY="$MCP_SCRAPER_API_KEY" npx -y -p mcp-scraper@latest mcp-scraper-cli agent install claude --apply',a=`X-Ray install protocol: ${v} (${b})`;return c==="codex"?["# Codex MCP config",a,i.exactConfig,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
|
|
4
4
|
`):c==="claude"?["# Claude Code command",a,i.exactConfig,"","# One-command Claude Code setup",u,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
|
|
5
5
|
`):c==="claude-desktop"?["# Claude Desktop config",a,i.exactConfig,"","Desktop Extension: https://mcpscraper.dev/downloads/mcp-scraper.mcpb",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,p].join(`
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as e}from"../chunk-
|
|
2
|
+
import{a as e}from"../chunk-J374DIKA.js";import"../chunk-53IEEOKA.js";import"../chunk-KZV2FLGG.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-CTX3NMZQ.js";import"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-OIMPG6M6.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import"../chunk-7SHEBDEB.js";import"../chunk-WJ4XFLS4.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as s}from"../chunk-
|
|
2
|
+
import{a as s}from"../chunk-OIMPG6M6.js";import{a as e}from"../chunk-7SHEBDEB.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
|
|
3
3
|
`)),process.exit(0));process.stdout.write(s({version:e,color:!r,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()}));
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as r}from"../chunk-
|
|
2
|
+
import{a as r}from"../chunk-J374DIKA.js";import"../chunk-53IEEOKA.js";import"../chunk-KZV2FLGG.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-CTX3NMZQ.js";import"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-OIMPG6M6.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import"../chunk-7SHEBDEB.js";import"../chunk-WJ4XFLS4.js";r();
|
package/dist/bin/paa-harvest.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{w as t}from"../chunk-
|
|
2
|
+
import{w as t}from"../chunk-ZMNUI5LB.js";import"../chunk-M5QHXNFZ.js";import{a as r}from"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-2TZBO52D.js";import"../chunk-2PXP7TMY.js";import"../chunk-WJ4XFLS4.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--\u006b\u0065\u0072\u006e\u0065\u006c-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,\u006b\u0065\u0072\u006e\u0065\u006cApiKey:e.browserApiKey??e.\u006b\u0065\u0072\u006e\u0065\u006cApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
|