mcp-scraper 0.89.1 → 0.89.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +59 -15
- package/README.md +3 -3
- package/THIRD_PARTY_NOTICES.html +203 -0
- package/dist/{analytics-repository-J25XR5E7.js → analytics-repository-5CAFUXYD.js} +1 -1
- package/dist/bin/api-server.js +1 -1
- package/dist/bin/mcp-scraper-cli.js +1 -1
- package/dist/bin/mcp-scraper-core.js +1 -1
- package/dist/bin/mcp-scraper-install.js +1 -1
- package/dist/bin/mcp-stdio-server.js +1 -1
- package/dist/bin/paa-harvest.js +1 -1
- package/dist/{chunk-FFPD2DOF.js → chunk-2O3UYISB.js} +1 -1
- package/dist/chunk-3GP5CYZX.js +1 -1
- package/dist/chunk-4FROKQJN.js +1 -1
- package/dist/{chunk-YUPLOOTB.js → chunk-CYOEMHTB.js} +1 -1
- package/dist/{chunk-RYNHNEF3.js → chunk-D4NOJAUN.js} +15 -15
- package/dist/{chunk-IJ2AO7AO.js → chunk-DMCYIWB3.js} +1 -1
- package/dist/{chunk-ZHORX44A.js → chunk-EN66HBAW.js} +1 -1
- package/dist/{chunk-PRWZNHQB.js → chunk-HYQDXTS7.js} +1 -1
- package/dist/{chunk-6QO4F2KC.js → chunk-J7EVLE6H.js} +3 -3
- package/dist/{chunk-57AHYL2O.js → chunk-K7KMLT33.js} +1 -1
- package/dist/{chunk-CCYWSJNG.js → chunk-OBDMJCCF.js} +4 -4
- package/dist/{chunk-645LQBE7.js → chunk-OXFWXWQG.js} +3 -3
- package/dist/{chunk-MEM4D2QN.js → chunk-QN3QHTVK.js} +88 -88
- package/dist/chunk-QPWPR5XG.js +3 -3
- package/dist/chunk-RNDORDA2.js +1 -0
- package/dist/chunk-S7RA23GR.js +1 -0
- package/dist/{chunk-DGBLGFNJ.js → chunk-VEJQ4XPI.js} +1 -1
- package/dist/{chunk-D6IQJP64.js → chunk-Z27SYWYD.js} +1 -1
- package/dist/{chunk-63T4W2LX.js → chunk-Z66Z2HV4.js} +87 -83
- package/dist/{db-JJWJLAVQ.js → db-NLGNTQC3.js} +1 -1
- package/dist/{extract-bundle-TTSEGWYF.js → extract-bundle-TXH3VNXH.js} +1 -1
- package/dist/{gmail-service-34PIJRLK.js → gmail-service-J2DWSQJX.js} +1 -1
- package/dist/index.cjs +49 -37
- package/dist/index.d.cts +14 -14
- package/dist/index.d.ts +14 -14
- package/dist/index.js +1 -1
- package/dist/{lead-list-enrichment-repository-VJBBMBCG.js → lead-list-enrichment-repository-G7Z5DQZR.js} +1 -1
- package/dist/{location-data-repository-W3LPT6G3.js → location-data-repository-TEXZ3I2N.js} +1 -1
- package/dist/{server-UEY5YQL6.js → server-RORLM2FG.js} +719 -712
- package/dist/{site-extract-repository-7W5ZUNPR.js → site-extract-repository-MBSCAP4D.js} +1 -1
- package/dist/worker-SUCJKWU2.js +1 -0
- package/package.json +17 -131
- package/dist/chunk-5H3MCTTP.js +0 -1
- package/dist/chunk-FXEWHK54.js +0 -1
- package/dist/worker-ZOTOKG56.js +0 -1
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
var s={message:"
|
|
1
|
+
var s={message:"Account limits now report accurately, repeated URL extraction requests behave consistently across REST and MCP clients, and retry messages point to the correct recovery path."};var _={reset:"\x1B[0m",cyan:"\x1B[36m",lime:"\x1B[32m",amber:"\x1B[33m",red:"\x1B[31m",muted:"\x1B[90m",bold:"\x1B[1m"};function r(t,e,n){return n?`${_[e]}${t}${_.reset}`:t}function o(t,e,n){let a=t.padEnd(9," ");return` ${r(a,"muted",n)} ${e.join(r(" . ","muted",n))}`}function p(t){let e=t.color??!0,n=t.apiKeyConfigured?"$MCP_SCRAPER_API_KEY":"sk_live_your_key",a=String.raw`
|
|
2
2
|
__ __ ____ ____
|
|
3
3
|
| \/ |/ ___| _ \
|
|
4
4
|
| |\/| | | | |_) |
|
|
@@ -10,7 +10,7 @@ var s={message:"Discover public Facebook Reel URLs, retain partial inventories,
|
|
|
10
10
|
\___ \| | | |_) | / _ \ | |_) | _| | |_) |
|
|
11
11
|
___) | |___| _ < / ___ \| __/| |___| _ <
|
|
12
12
|
|____/ \____|_| \_\/_/ \_\_| |_____|_| \_\
|
|
13
|
-
`,
|
|
14
|
-
`),
|
|
15
|
-
`);return[r(`mcp-scraper v${t.version}`,"bold",e),r("> mcp-scraper-install","muted",e),r(a,"amber",e),`${r("MCP Scraper Agent","cyan",e)} . v${t.version} . mcpscraper.dev`,"1/1 install surfaces ready",r(`Newest in v${t.version}: ${s.message}`,"lime",e),"",`${r("Tools","cyan",e)} ${r("(376 MCP tools)","muted",e)}`,o("search",["harvest_paa","search_serp","maps_search","maps_place_intel"],e),o("extract",["extract_url","map_site_urls","extract_site","audit_site","directory_workflow"],e),o("build",["create_editorial_reading_room","rank_tracker_workflow","portable HTML"],e),o("media",["youtube_harvest","youtube_transcribe","facebook_reels_inventory","facebook_ad_search","facebook_page_intel","facebook_ad_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","reddit_thread"],e),o("browser",["serp_identity_create","serp_identity_list","browser_open","browser_profile_connect","browser_profile_list","browser_close","browser_screenshot","browser_read","browser_locate","browser_replay_mark","browser_replay_annotate"],e),o("connect",["list_service_connections","describe_service_connection_tool","import_service_connection_to_memory","export_connected_service_data","renew_connected_data_download","read_service_connection","call_service_connection_action"],e),o("commons",["commons_search_entities","commons_get_entity_linkset","commons_prepare_entity","commons_submit_entity","commons_prepare_publication","commons_claim_publication","commons_publish_editorial","commons_get_publication"],e),o("account",["credits_info","reports","MCP resources"],e),o("memory",["memory-put","memory-get","memory-search","list-vaults","record-fact","list-scheduled-actions"],e),`${r("Workflows","cyan",e)} ${r("(MCP + CLI + API)","muted",e)}`,o("route",["workflow_list","workflow_suggest","workflow_run","workflow_step","workflow_status","workflow_artifact_read"],e),o("seo",["directory","agent-packet","competitive audit","map/serp comparison","PAA/AIO briefs","scheduled runs"],e),"",r("Usage tips:","amber",e),"Run mcp-scraper-install for this visible card. Run mcp-scraper-cli for setup utilities and subcommands.","Run mcp-scraper in a human terminal to print this card; MCP clients get the same command as a silent stdio server.","Explicit card command: npx -y -p mcp-scraper@latest mcp-scraper-install","Hosted browser sessions use direct/no-proxy egress by default.","Customer auth setup: run browser_profile_connect, send the watch_url, let the user sign in, then call browser_profile_list until AUTHENTICATED.","Connected account ranges: call export_connected_service_data once. It handles Gmail, Calendar, Zoom, and Resend pagination; do not loop read_service_connection over individual records.","Connected account RAG: call import_service_connection_to_memory for one bounded approved read. It writes a redacted, untrusted snapshot to a stable Memory path and embeds it for search.","Stack logins / reconnect: run browser_profile_connect again with the same profile name and another domain to add accounts or refresh a login.","Start with workflow_suggest for broad jobs like market analysis, ICP research, CRO audits, brand briefs, content gaps, and AI visibility.","For MCP clients, use mcp-scraper so one install can mix SERP, Maps, browser, reports, and saved MCP resources.","If you hit the concurrency limit, add 2 browsers for $5/month with mcp-scraper-cli billing concurrency checkout.","",`${r("Ready.","lime",e)} Install the combined MCP server with one command:`,"",r("Setup doctor","amber",e),"npx -y -p mcp-scraper@latest mcp-scraper-cli doctor","",r("Hosted profile setup","amber",e),'In your MCP client, call browser_profile_connect with email="seo@example.com" and domain="chatgpt.com".',"Give the returned watch_url to the user. After they sign in, call browser_profile_list, then browser_open with the returned profile. Add more logins by calling browser_profile_connect again with the same profile and a new domain.","",r("Claude Code one-command setup","amber",e),
|
|
13
|
+
`,c=[`MCP_SCRAPER_API_KEY=${n} npx -y -p mcp-scraper@latest \\`," mcp-scraper-cli agent install claude --apply"].join(`
|
|
14
|
+
`),i=["[mcp_servers.mcp-scraper]",'command = "npx"','args = ["-y", "-p", "mcp-scraper@latest", "mcp-scraper"]',`env = { MCP_SCRAPER_API_KEY = "${n}" }`].join(`
|
|
15
|
+
`);return[r(`mcp-scraper v${t.version}`,"bold",e),r("> mcp-scraper-install","muted",e),r(a,"amber",e),`${r("MCP Scraper Agent","cyan",e)} . v${t.version} . mcpscraper.dev`,"1/1 install surfaces ready",r(`Newest in v${t.version}: ${s.message}`,"lime",e),"",`${r("Tools","cyan",e)} ${r("(376 MCP tools)","muted",e)}`,o("search",["harvest_paa","search_serp","maps_search","maps_place_intel"],e),o("extract",["extract_url","map_site_urls","extract_site","audit_site","directory_workflow"],e),o("build",["create_editorial_reading_room","rank_tracker_workflow","portable HTML"],e),o("media",["youtube_harvest","youtube_transcribe","facebook_reels_inventory","facebook_ad_search","facebook_page_intel","facebook_ad_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","reddit_thread"],e),o("browser",["serp_identity_create","serp_identity_list","browser_open","browser_profile_connect","browser_profile_list","browser_close","browser_screenshot","browser_read","browser_locate","browser_replay_mark","browser_replay_annotate"],e),o("connect",["list_service_connections","describe_service_connection_tool","import_service_connection_to_memory","export_connected_service_data","renew_connected_data_download","read_service_connection","call_service_connection_action"],e),o("commons",["commons_search_entities","commons_get_entity_linkset","commons_prepare_entity","commons_submit_entity","commons_prepare_publication","commons_claim_publication","commons_publish_editorial","commons_get_publication"],e),o("account",["credits_info","reports","MCP resources"],e),o("memory",["memory-put","memory-get","memory-search","list-vaults","record-fact","list-scheduled-actions"],e),`${r("Workflows","cyan",e)} ${r("(MCP + CLI + API)","muted",e)}`,o("route",["workflow_list","workflow_suggest","workflow_run","workflow_step","workflow_status","workflow_artifact_read"],e),o("seo",["directory","agent-packet","competitive audit","map/serp comparison","PAA/AIO briefs","scheduled runs"],e),"",r("Usage tips:","amber",e),"Run mcp-scraper-install for this visible card. Run mcp-scraper-cli for setup utilities and subcommands.","Run mcp-scraper in a human terminal to print this card; MCP clients get the same command as a silent stdio server.","Explicit card command: npx -y -p mcp-scraper@latest mcp-scraper-install","Hosted browser sessions use direct/no-proxy egress by default.","Customer auth setup: run browser_profile_connect, send the watch_url, let the user sign in, then call browser_profile_list until AUTHENTICATED.","Connected account ranges: call export_connected_service_data once. It handles Gmail, Calendar, Zoom, and Resend pagination; do not loop read_service_connection over individual records.","Connected account RAG: call import_service_connection_to_memory for one bounded approved read. It writes a redacted, untrusted snapshot to a stable Memory path and embeds it for search.","Stack logins / reconnect: run browser_profile_connect again with the same profile name and another domain to add accounts or refresh a login.","Start with workflow_suggest for broad jobs like market analysis, ICP research, CRO audits, brand briefs, content gaps, and AI visibility.","For MCP clients, use mcp-scraper so one install can mix SERP, Maps, browser, reports, and saved MCP resources.","If you hit the concurrency limit, add 2 browsers for $5/month with mcp-scraper-cli billing concurrency checkout.","",`${r("Ready.","lime",e)} Install the combined MCP server with one command:`,"",r("Setup doctor","amber",e),"npx -y -p mcp-scraper@latest mcp-scraper-cli doctor","",r("Hosted profile setup","amber",e),'In your MCP client, call browser_profile_connect with email="seo@example.com" and domain="chatgpt.com".',"Give the returned watch_url to the user. After they sign in, call browser_profile_list, then browser_open with the returned profile. Add more logins by calling browser_profile_connect again with the same profile and a new domain.","",r("Claude Code one-command setup","amber",e),c,"Then fully exit Claude Code and open a new Claude terminal. Check with: claude mcp list","",r("Codex config","amber",e),i,"",r("Claude Desktop Extension","amber",e),"Download: https://mcpscraper.dev/downloads/mcp-scraper.mcpb","",r("Safety note:","muted",e),"mcp-scraper prints this card only when stdin/stdout are an interactive TTY. In MCP clients it writes only JSON-RPC to stdout.","Use --stdio or MCP_SCRAPER_FORCE_STDIO=1 to force server mode from a terminal.",""].join(`
|
|
16
16
|
`)}export{p as a};
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{a as Q,d as Se}from"./chunk-
|
|
1
|
+
import{a as Q,d as Se}from"./chunk-Z27SYWYD.js";import{createHash as pe,createHmac as Yt,randomBytes as P,randomUUID as y}from"crypto";import{Pool as Wt}from"pg";var ct=[[/perplexity/i,"llm","Perplexity"],[/chatgpt|chat\.openai|openai/i,"llm","ChatGPT"],[/claude|anthropic/i,"llm","Claude"],[/\bgrok\b|xai/i,"llm","Grok"],[/ai[ _-]?overview|google[ _-]?ai|\bsge\b|gemini/i,"llm","Google AI Overview"],[/instagram/i,"social","Instagram"],[/facebook|\bfb\b|meta/i,"social","Facebook"],[/linkedin/i,"social","LinkedIn"],[/tiktok/i,"social","TikTok"],[/youtube|youtu\.be/i,"social","YouTube"],[/twitter|x\.com/i,"social","X"],[/\bg2\b|g2\.com/i,"review","G2"],[/trustpilot/i,"review","Trustpilot"],[/\bbbb\b|better business bureau/i,"review","BBB"],[/capterra/i,"review","Capterra"],[/yelp/i,"review","Yelp"],[/google|bing|duckduckgo|yahoo/i,"search","Search"]];function q(e,t=240){return typeof e=="string"&&e.trim()?e.trim().slice(0,t):null}function Ce(e){try{return e?new URL(e).hostname.replace(/^www\./,""):""}catch{return""}}function Re(e){let t=e.properties??{},i=q(e.source,180)||Ce(e.referrer)||"(direct)",n=q(e.medium,180)||(i==="(direct)"?"(none)":"referral"),s=e.clickIds??{},a=`${i} ${n} ${Ce(e.referrer)}`,r=i==="(direct)"?"direct":/email|newsletter/i.test(n)?"email":"referral",o=i==="(direct)"?"Direct":i;for(let[d,_,u]of ct)if(d.test(a)){r=_,o=u;break}return s.fbclid?(r="social",o=/instagram/i.test(a)?"Instagram":"Facebook"):s.ttclid?(r="social",o="TikTok"):s.rdt_cid?(r="social",o="Reddit"):s.gclid||s.gbraid||s.wbraid?(r="search",o="Google"):s.msclkid&&(r="search",o="Microsoft Ads"),{channelFamily:r,platform:o,source:i,medium:n,campaign:q(e.campaign),campaignId:q(t.campaign_id??t.utm_id),adSetId:q(t.adset_id??t.ad_set_id??t.adgroup_id??t.ad_group_id),adId:q(t.ad_id),creativeId:q(t.creative_id),placement:q(t.placement),term:q(t.utm_term),content:q(t.utm_content)}}import{createHash as yt,randomUUID as ee}from"crypto";import{createCipheriv as _t,createDecipheriv as ut,createHmac as fe,randomBytes as Et,scryptSync as lt,timingSafeEqual as Nt}from"crypto";var Z="xray1";function mt(){return process.env.ANALYTICS_RESTRICTED_DATA_SECRET?.trim()||Q()}function _e(e){return lt(mt(),`xray-restricted:${e}:v1`,32)}function De(e,t){let i=t.trim().slice(0,80);if(!i)throw new Error("analytics encryption purpose required");let n=Et(12),s=_t("aes-256-gcm",_e(i),n);s.setAAD(Buffer.from(`${Z}:${i}`,"utf8"));let a=Buffer.concat([s.update(e,"utf8"),s.final()]),r=s.getAuthTag();return[Z,Buffer.from(i,"utf8").toString("base64url"),n.toString("base64url"),r.toString("base64url"),a.toString("base64url")].join(".")}function Ue(e,t){try{let[i,n,s,a,r]=e.split(".");if(i!==Z||!n||!s||!a||!r)return null;let o=Buffer.from(n,"base64url").toString("utf8");if(o!==t)return null;let d=ut("aes-256-gcm",_e(o),Buffer.from(s,"base64url"));return d.setAAD(Buffer.from(`${Z}:${o}`,"utf8")),d.setAuthTag(Buffer.from(a,"base64url")),Buffer.concat([d.update(Buffer.from(r,"base64url")),d.final()]).toString("utf8")}catch{return null}}function bi(e,t){return fe("sha256",_e(`identity:${e}`)).update(t.trim().toLowerCase()).digest("hex")}function hi(e,t,i,n){return fe("sha256",n).update(`${e}.${t}.${i}`).digest("hex")}function xi(e,t){if(!e||e.length!==t.length)return!1;try{return Nt(Buffer.from(e),Buffer.from(t))}catch{return!1}}var we="enhanced_matching",pt=90,Tt=10,H=class extends Error{constructor(i,n,s){super(n);this.code=i;this.status=s;this.name="AnalyticsIdentityProfileError"}code;status};function te(e,t){return e?.trim().replace(/\s+/g," ").slice(0,t)||null}function Lt(e){if(!e)return null;let[t,i]=e.split("@");return!t||!i?null:`${t.slice(0,1)}${"*".repeat(Math.min(6,Math.max(2,t.length-1)))}@${i}`}function At(e){if(!e)return null;let t=e.replace(/\D/g,"");return t.length<4?null:`***-***-${t.slice(-4)}`}function ue(e){return yt("sha256").update(e).digest("hex")}async function ie(e,t){if(!t.signals.length)return;let i=t.signals.map(a=>a.kind),n=t.signals.map(a=>a.valueHmac);if((await e.query(`SELECT 1 FROM analytics_identity_tombstones
|
|
2
2
|
WHERE site_id=$1 AND (kind,value_hmac) IN (
|
|
3
3
|
SELECT * FROM unnest($2::text[],$3::text[])
|
|
4
4
|
) LIMIT 1`,[t.siteId,i,n])).rowCount)throw new H("analytics_identity_deleted","This identity cannot be automatically recreated.",410)}async function ne(e,t){let i=te(t.displayName,160),n=te(t.email,320)?.toLowerCase()??null,s=te(t.phone,40);if(!i&&!n&&!s)return!1;let r=(await e.query(`SELECT p.public_ref FROM analytics_people p
|
|
@@ -481,7 +481,7 @@ import{a as Q,d as Se}from"./chunk-D6IQJP64.js";import{createHash as pe,createHm
|
|
|
481
481
|
plan_id text,
|
|
482
482
|
subscription_status text,
|
|
483
483
|
monthly_recurring_revenue_cents integer NOT NULL DEFAULT 0,
|
|
484
|
-
minimum_monthly_recurring_revenue_cents integer NOT NULL DEFAULT
|
|
484
|
+
minimum_monthly_recurring_revenue_cents integer NOT NULL DEFAULT 0,
|
|
485
485
|
verified_at timestamptz NOT NULL,
|
|
486
486
|
refresh_after timestamptz NOT NULL,
|
|
487
487
|
access_expires_at timestamptz NOT NULL,
|
|
@@ -490,7 +490,7 @@ import{a as Q,d as Se}from"./chunk-D6IQJP64.js";import{createHash as pe,createHm
|
|
|
490
490
|
last_refresh_error text,
|
|
491
491
|
updated_at timestamptz NOT NULL DEFAULT now()
|
|
492
492
|
)
|
|
493
|
-
`),await t.query("ALTER TABLE analytics_thorbit_entitlements ALTER COLUMN thorbit_org_id DROP NOT NULL"),await t.query("ALTER TABLE analytics_thorbit_entitlements ADD COLUMN IF NOT EXISTS trial_started_at timestamptz NOT NULL DEFAULT now()"),await t.query("ALTER TABLE analytics_thorbit_entitlements ADD COLUMN IF NOT EXISTS trial_ends_at timestamptz NOT NULL DEFAULT (now() + interval '30 days')"),await t.query("ALTER TABLE analytics_thorbit_entitlements DROP CONSTRAINT IF EXISTS analytics_thorbit_entitlements_connection_source_check"),await t.query("ALTER TABLE analytics_thorbit_entitlements ADD CONSTRAINT analytics_thorbit_entitlements_connection_source_check CHECK (connection_source IN ('trial','api_key','thorbit_session'))"),await t.query(`
|
|
493
|
+
`),await t.query("ALTER TABLE analytics_thorbit_entitlements ALTER COLUMN thorbit_org_id DROP NOT NULL"),await t.query("ALTER TABLE analytics_thorbit_entitlements ADD COLUMN IF NOT EXISTS trial_started_at timestamptz NOT NULL DEFAULT now()"),await t.query("ALTER TABLE analytics_thorbit_entitlements ADD COLUMN IF NOT EXISTS trial_ends_at timestamptz NOT NULL DEFAULT (now() + interval '30 days')"),await t.query("ALTER TABLE analytics_thorbit_entitlements DROP CONSTRAINT IF EXISTS analytics_thorbit_entitlements_connection_source_check"),await t.query("ALTER TABLE analytics_thorbit_entitlements ADD CONSTRAINT analytics_thorbit_entitlements_connection_source_check CHECK (connection_source IN ('trial','api_key','thorbit_session'))"),await t.query("ALTER TABLE analytics_thorbit_entitlements ALTER COLUMN minimum_monthly_recurring_revenue_cents SET DEFAULT 0"),await t.query(`
|
|
494
494
|
CREATE TABLE IF NOT EXISTS analytics_thorbit_connect_states (
|
|
495
495
|
state_hash text PRIMARY KEY,
|
|
496
496
|
user_id bigint NOT NULL,
|