mcp-scraper 0.90.7 → 0.91.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -1
- package/README.md +4 -4
- package/dist/{analytics-repository-FXS45ZTM.js → analytics-repository-IHOFBSUV.js} +1 -1
- package/dist/bin/api-server.js +1 -1
- package/dist/bin/mcp-scraper-cli.js +1 -1
- package/dist/bin/mcp-scraper-core.js +1 -1
- package/dist/bin/mcp-scraper-install.js +1 -1
- package/dist/bin/mcp-stdio-server.js +1 -1
- package/dist/bin/paa-harvest.js +1 -1
- package/dist/{chunk-BDLR4CCL.js → chunk-2SP57VCG.js} +6 -6
- package/dist/chunk-3JUD76B6.js +1 -0
- package/dist/chunk-6DTXIZY2.js +1 -0
- package/dist/{chunk-QUFR3KG2.js → chunk-6KASNPKR.js} +1 -1
- package/dist/chunk-72YO6PKN.js +1 -0
- package/dist/{chunk-YTEAWZEN.js → chunk-C3QWOZUH.js} +1 -1
- package/dist/chunk-EVQA7RMG.js +1 -0
- package/dist/{chunk-TBTA7VVZ.js → chunk-IK5BG7MO.js} +1 -1
- package/dist/{chunk-I7IS7NTC.js → chunk-KZV2FLGG.js} +1 -1
- package/dist/{chunk-UKFNSXP3.js → chunk-LI7WHOII.js} +1 -1
- package/dist/{chunk-B2C7JU4O.js → chunk-LSFDB5XE.js} +1 -1
- package/dist/chunk-M5TYS4EX.js +16 -0
- package/dist/{chunk-ZKQRA6NQ.js → chunk-MWUKEBQA.js} +175 -171
- package/dist/chunk-ORB4RHCK.js +63 -0
- package/dist/{chunk-LPEIHMYH.js → chunk-POTKJRLE.js} +1 -1
- package/dist/{chunk-IGJKAECE.js → chunk-QTDLZTQ7.js} +1 -1
- package/dist/chunk-RK2VCTZI.js +1 -0
- package/dist/{chunk-OT2HUS2G.js → chunk-VEP42LGR.js} +1 -1
- package/dist/{chunk-TOJBPATJ.js → chunk-WJ4XFLS4.js} +159 -115
- package/dist/{chunk-74HHYSAZ.js → chunk-X7GZJU5P.js} +1 -1
- package/dist/{chunk-5NQY2N7J.js → chunk-YQNMW4XI.js} +2 -2
- package/dist/{chunk-R7LCD65B.js → chunk-ZMNUI5LB.js} +7 -7
- package/dist/db-B5XJTOGN.js +1 -0
- package/dist/{extract-bundle-INT6JMJP.js → extract-bundle-TMS6LLIT.js} +10 -10
- package/dist/gmail-service-LXBBOEIQ.js +1 -0
- package/dist/index.cjs +18 -18
- package/dist/index.js +1 -1
- package/dist/{lead-list-enrichment-repository-B6Q4O4DG.js → lead-list-enrichment-repository-AWZZMY2H.js} +1 -1
- package/dist/{location-data-repository-CBI5K7MR.js → location-data-repository-6XAVI65Z.js} +1 -1
- package/dist/operation-metering-DB5F7SXN.js +1 -0
- package/dist/server-MPVSSCAH.js +4963 -0
- package/dist/{site-extract-repository-VPYOC3CP.js → site-extract-repository-OKXC3NFC.js} +1 -1
- package/dist/stripe-event-worker-KQOQOM22.js +1 -0
- package/dist/worker-NKLJTKDQ.js +1 -0
- package/package.json +1 -1
- package/dist/chunk-CTX3NMZQ.js +0 -1
- package/dist/chunk-DSK6RKBS.js +0 -16
- package/dist/chunk-KDDU6BW3.js +0 -48
- package/dist/chunk-WEPU72WA.js +0 -1
- package/dist/db-SVSD2AWJ.js +0 -1
- package/dist/gmail-service-YCPMGW3E.js +0 -1
- package/dist/server-LYWW6XYE.js +0 -4956
- package/dist/stripe-event-worker-AQVRF24K.js +0 -1
- package/dist/worker-U43KLOLN.js +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,43 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.91.0] - 2026-09-24
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Transcribe speech from a public TikTok video or short share URL through `tiktok_video_transcribe`, with full text, timed chunks, source details, and a speech-quality signal. When playable media is unavailable, matching public captions can provide a free fallback.
|
|
12
|
+
- Track TikTok browser and speech-model provider costs under the TikTok operation, with the same 200-credit base, 2-credit-per-minute speech charge, credit reconciliation, and failed-call refund used for Facebook video transcription.
|
|
13
|
+
|
|
14
|
+
## [0.90.8] - 2026-09-24
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
|
|
18
|
+
- Record observed provider product, billing tier, usage units, and applied rate snapshots on normalized cost receipts; preserve older receipts through the schema migration.
|
|
19
|
+
- Link direct SERP page requests, Maps browser rotations, extraction sessions, shared site-crawl batches, and FAL jobs to operation attempts before provider contact.
|
|
20
|
+
- Store provider account-window totals and explicit unattributed variance without distributing the remainder across customer operations.
|
|
21
|
+
- Link transcription, Instagram media, and Maps place debit receipts to their operation runs, including additional duration charges.
|
|
22
|
+
- Record local caption, media, plain-page, Wayback, Maps detail, and export phases without creating duplicate provider charges.
|
|
23
|
+
- Start synchronous and durable site cost runs before discovery, and identify browser discovery retries under the site operation.
|
|
24
|
+
- Persist receipt-creation coverage defects when the database remains writable, so missing provider-cost evidence appears in operator reports.
|
|
25
|
+
- Check scoped provider metering calls against the generated method registry and reject unregistered methods during local verification.
|
|
26
|
+
|
|
27
|
+
### Changed
|
|
28
|
+
|
|
29
|
+
- Price measured browser bandwidth at the applicable standard or premium-domain tier, and record the selected tier and list snapshot on the provider receipt.
|
|
30
|
+
- Report receipt identity and known-cost closure separately. CEO and CTO cost views distinguish actual, measured, estimated, pending, and unavailable amounts, with product, method, and retry breakdowns.
|
|
31
|
+
- Keep existing wall-time estimates visible as estimates while waiting for provider usage evidence and accepted account rates.
|
|
32
|
+
- Attribute Facebook video fallback to the browser session it actually uses, and keep optional rendered capture as a separate browser session cost.
|
|
33
|
+
- Keep Instagram browser acquisition and optional FAL transcription under one operation run, with separate provider receipts.
|
|
34
|
+
- Count Maps services fallback sessions separately, and stop browser wall-time estimates at session close when a route continues with other work.
|
|
35
|
+
|
|
36
|
+
### Fixed
|
|
37
|
+
|
|
38
|
+
- Recover completed single-page extraction results from their original paid job: hosted and installed MCP status expose owner-scoped artifact readback, same-key replay keeps the job receipt, and dashboard history opens the durable job instead of a duplicate activity row.
|
|
39
|
+
- Keep the provider-receipt table rewrite out of hosted startup; releases now apply that constraint change through the explicit database migration command, while application instances fail closed when the operator migration is missing.
|
|
40
|
+
- Keep the five-minute and hourly maintenance passes within the hosted function's time limit when they start on a cold instance, so hourly billing, retention, and daily cleanup complete and record their runs instead of timing out.
|
|
41
|
+
- Stop creating a provider-usage index during hosted startup; the provider-usage lookback now uses the existing timestamp index instead.
|
|
42
|
+
- Stop the hourly job-retention sweep from holding the database write lock for minutes: expired completed and failed jobs are now found by insertion order instead of by scanning every job's stored result, which had stalled schema changes, cold starts, and scheduled maintenance for several minutes after each run.
|
|
43
|
+
|
|
7
44
|
## [0.90.7] - 2026-09-23
|
|
8
45
|
|
|
9
46
|
### Changed
|
|
@@ -1986,7 +2023,9 @@ All notable changes to MCP Scraper are documented here. The format is based on [
|
|
|
1986
2023
|
- Write actions remain unavailable until the account owner explicitly enables them.
|
|
1987
2024
|
- Provider-specific connection data is normalized into one agent-facing contract.
|
|
1988
2025
|
|
|
1989
|
-
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.
|
|
2026
|
+
[Unreleased]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.91.0...HEAD
|
|
2027
|
+
[0.91.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.8...v0.91.0
|
|
2028
|
+
[0.90.8]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.7...v0.90.8
|
|
1990
2029
|
[0.90.7]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.90.6...v0.90.7
|
|
1991
2030
|
[0.88.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.87.0...v0.88.0
|
|
1992
2031
|
[0.87.0]: https://github.com/VilovietaSEO/mcp-scraper/compare/v0.86.5...v0.87.0
|
package/README.md
CHANGED
|
@@ -24,7 +24,7 @@ Use the MCPB Desktop Extension for the branded Claude Desktop install, or use th
|
|
|
24
24
|
|
|
25
25
|
MCP Scraper ships one stdio entrypoint plus human-facing helper CLIs:
|
|
26
26
|
|
|
27
|
-
- `mcp-scraper` — the single MCP server with every tool: live web intelligence (SERP, PAA, site extraction, YouTube, Facebook ads and organic video transcripts, Maps, directory, rank tracker blueprint, credits) plus agent-controlled direct/no-proxy hosted browser sessions (screenshots, clicks, typing, scrolling, watch URLs, replay links, MP4 replay download, and saved profile/login setup for authenticated AI visibility workflows). It is context-aware: in a human terminal it prints the branded ASCII install card; in an MCP client it runs as a protocol-clean stdio server. This is the entrypoint used by the MCPB Desktop Extension.
|
|
27
|
+
- `mcp-scraper` — the single MCP server with every tool: live web intelligence (SERP, PAA, site extraction, YouTube, Facebook ads and organic video transcripts, TikTok video transcripts, Maps, directory, rank tracker blueprint, credits) plus agent-controlled direct/no-proxy hosted browser sessions (screenshots, clicks, typing, scrolling, watch URLs, replay links, MP4 replay download, and saved profile/login setup for authenticated AI visibility workflows). It is context-aware: in a human terminal it prints the branded ASCII install card; in an MCP client it runs as a protocol-clean stdio server. This is the entrypoint used by the MCPB Desktop Extension.
|
|
28
28
|
- `mcp-scraper-install` — explicit alias for the human-facing terminal installer card with the branded ASCII intro and copyable install commands.
|
|
29
29
|
- `mcp-scraper-cli` — a human-facing CLI for setup checks, AI-agent config generation, workflow prompts, SEO workflow runs, and HTML reports. This command is safe to print because it is not an MCP stdio server.
|
|
30
30
|
|
|
@@ -175,7 +175,7 @@ Build the branded one-click bundle:
|
|
|
175
175
|
npm run build:mcpb
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
-
The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.
|
|
178
|
+
The generated bundle is written to `build/mcpb/mcp-scraper-<version>.mcpb` and copied to `public/downloads/` for the hosted download. The current public bundle is `https://mcpscraper.dev/downloads/mcp-scraper.mcpb` (`0.91.0`, SHA-256 `abf2fe0e0b8f218c401af223114fc04189a8a8e78cff17cf95070fe5c8cc4f91`). Install it by opening or dragging it into Claude Desktop. Claude displays the `MCP Scraper` install card, icon, API-key configuration field, and manually curated current-release message from the bundle manifest.
|
|
179
179
|
|
|
180
180
|
The MCPB install exposes every tool — web-intelligence plus all `browser_*` tools — through the one `mcp-scraper` server.
|
|
181
181
|
|
|
@@ -252,7 +252,7 @@ Check `pagination.requestedPages`, `pagination.capturedPages`, and `pagination.p
|
|
|
252
252
|
- `harvest_paa` — expand People Also Ask on the original first result page. Optional `pages: 2` captures a second organic result page first; the default is one page.
|
|
253
253
|
- `harvest_paa_start` — start the same harvest as a durable job, with the same optional `pages: 2`.
|
|
254
254
|
- `search_serp`
|
|
255
|
-
- `extract_url` — start a durable normal or Wayback-replayed page extraction and receive a job receipt immediately; poll `extract_url_status` with the returned job ID until it reaches a terminal state. Wayback results omit playback chrome and can include a timestamp-matched featured image. Set `preserveMedia:true` to union static and rendered/lazy media, collapse responsive variants, attach up to `maxInlineImages` AI-readable images, and receive an owner-scoped ZIP manifest readable with `archive_read`. Branding output ranks the site logo separately from evidence-bounded proof images such as certifications, awards, memberships, partner/customer marks, and press mentions.
|
|
255
|
+
- `extract_url` — start a durable normal or Wayback-replayed page extraction and receive a job receipt immediately; poll `extract_url_status` with the returned job ID until it reaches a terminal state. For artifact delivery, use the returned `artifact.artifactId` with `report_artifact_read` to retrieve the full saved page in windows. Polling and reading do not start or bill another extraction. Hosted readback lasts while the completed job is retained (up to 45 days); an installed MCP client mirrors the page into its local artifact store for 24 hours from each successful status read. Wayback results omit playback chrome and can include a timestamp-matched featured image. Set `preserveMedia:true` to union static and rendered/lazy media, collapse responsive variants, attach up to `maxInlineImages` AI-readable images, and receive an owner-scoped ZIP manifest readable with `archive_read`. Branding output ranks the site logo separately from evidence-bounded proof images such as certifications, awards, memberships, partner/customer marks, and press mentions.
|
|
256
256
|
- `map_site_urls`
|
|
257
257
|
- `map_wayback_snapshots` — count and inventory Wayback captures across an inclusive date range without downloading page bodies. Supports exact pages, prefixes, hosts, domains, or selected URLs; reports exact versus lower-bound counts, unique URLs/content digests, monthly coverage, missing months, and optional timestamp rows.
|
|
258
258
|
- `extract_site` — crawl a live site, batch one archived site snapshot from a Wayback replay URL, or pass a `wayback` plan for whole-site, single-page, or selected-page timelines across explicit months or a `from`/`to` range. Timeline ZIPs include month folders and a capture matrix.
|
|
@@ -351,7 +351,7 @@ The `mcp-scraper` server (and the MCPB bundle, which runs it) exposes both secti
|
|
|
351
351
|
|
|
352
352
|
All MCP tools return `structuredContent` with the IDs, URLs, CSV paths, transcripts, browser session handles, replay paths, artifacts, recipe fields, or blueprint fields needed by the next step, plus readable text content for compatibility. Runtime `tools/list` omits output schemas so strict clients can register the complete catalog; the generated developer manifest retains every canonical output schema for validation and typed SDK generation. All tools carry MCP annotations; file-writing tools such as replay downloads and annotations state their filesystem side effects.
|
|
353
353
|
|
|
354
|
-
The canonical tool inventory is generated at `docs/mcp-tool-manifest.generated.json`. The unified server exposes
|
|
354
|
+
The canonical tool inventory is generated at `docs/mcp-tool-manifest.generated.json`. The unified server exposes 362 tools: 236 scraper, browser, workflow, billing, and connected-service tools plus 126 durable-memory tools. Personal Assistant tools and routes are retired from active builds; their source history and persisted data remain recoverable from the archived pre-retirement branch. The scraper-side inventory includes complete Gmail selection, message, attachment, export, bulk-action, and Memory-import workflows; durable SERP, PAA, and single-page extraction starts and status; rendered site-content similarity; governed Local Sourcebook and Transparent Commons workflows; direct site-export reads; Editorial Reading Room and News Publisher templates; and production X-Ray setup, analytics, seven-model attribution, structured post-purchase surveys, reported impact, truthful view-evidence status, CRM policy and receipt, campaign, export, and scheduled-report tools. Provider setup remains absent until its authorization, ingestion, reconciliation, privacy, canary, cleanup, and deployment receipts are complete. Successful evidence-compiled Local Sourcebook revisions publish automatically to their canonical `localsourcebook.com` category profile and review URLs; administrator controls handle exceptional rejection or unpublishing. Release verification compares the exact local and hosted tool-name sets, not only the count.
|
|
355
355
|
|
|
356
356
|
For contract parity, stdio and MCPB memory calls invoke the matching public tool on the hosted MCP Scraper `/mcp` endpoint. The hosted aggregate runtime owns MCP Scraper-specific billing, scheduling, credential, and in-process cutover policy; its internal `/memory/mcp-call` bridge is a fallback to the standalone Memory service, not a second customer setup path. Existing direct Memory credentials remain compatible for one release, but all new customer setup uses the root endpoint and `MCP_SCRAPER_API_KEY`.
|
|
357
357
|
|
|
@@ -1 +1 @@
|
|
|
1
|
-
import{$ as L,$a as La,A as k,Aa as ka,B as l,Ba as la,C as m,Ca as ma,D as n,Da as na,E as o,Ea as oa,F as p,Fa as pa,G as q,Ga as qa,H as r,Ha as ra,I as s,Ia as sa,J as t,Ja as ta,K as u,Ka as ua,L as v,La as va,M as w,Ma as wa,N as x,Na as xa,O as y,Oa as ya,P as z,Pa as za,Q as A,Qa as Aa,R as B,Ra as Ba,S as C,Sa as Ca,T as D,Ta as Da,U as E,Ua as Ea,V as F,Va as Fa,W as G,Wa as Ga,X as H,Xa as Ha,Y as I,Ya as Ia,Z as J,Za as Ja,_ as K,_a as Ka,aa as M,ab as Ma,ba as N,bb as Na,ca as O,da as P,ea as Q,fa as R,ga as S,ha as T,ia as U,ja as V,ka as W,la as X,ma as Y,na as Z,oa as _,pa as $,q as a,qa as aa,r as b,ra as ba,s as c,sa as ca,t as d,ta as da,u as e,ua as ea,v as f,va as fa,w as g,wa as ga,x as h,xa as ha,y as i,ya as ia,z as j,za as ja}from"./chunk-
|
|
1
|
+
import{$ as L,$a as La,A as k,Aa as ka,B as l,Ba as la,C as m,Ca as ma,D as n,Da as na,E as o,Ea as oa,F as p,Fa as pa,G as q,Ga as qa,H as r,Ha as ra,I as s,Ia as sa,J as t,Ja as ta,K as u,Ka as ua,L as v,La as va,M as w,Ma as wa,N as x,Na as xa,O as y,Oa as ya,P as z,Pa as za,Q as A,Qa as Aa,R as B,Ra as Ba,S as C,Sa as Ca,T as D,Ta as Da,U as E,Ua as Ea,V as F,Va as Fa,W as G,Wa as Ga,X as H,Xa as Ha,Y as I,Ya as Ia,Z as J,Za as Ja,_ as K,_a as Ka,aa as M,ab as Ma,ba as N,bb as Na,ca as O,da as P,ea as Q,fa as R,ga as S,ha as T,ia as U,ja as V,ka as W,la as X,ma as Y,na as Z,oa as _,pa as $,q as a,qa as aa,r as b,ra as ba,s as c,sa as ca,t as d,ta as da,u as e,ua as ea,v as f,va as fa,w as g,wa as ga,x as h,xa as ha,y as i,ya as ia,z as j,za as ja}from"./chunk-LI7WHOII.js";import"./chunk-QTDLZTQ7.js";import"./chunk-WJ4XFLS4.js";export{P as ANALYTICS_CONTENT_SORTS,a as AnalyticsRepositoryError,B as ENGAGED_SESSION_MS,A as MAX_ENGAGED_MS,N as analyticsAcquisition,l as analyticsBusinessMetrics,O as analyticsChannelBreakdown,R as analyticsContent,T as analyticsConversions,Ma as analyticsCsvCell,V as analyticsDimensions,S as analyticsEventCounts,m as analyticsForecast,Ja as analyticsHealth,x as analyticsIdentityPromotionAllowed,w as analyticsIdentityResolutionAllowed,L as analyticsOverview,U as analyticsPaths,M as analyticsTimeseries,H as appendAnalyticsAuthoritativeOutcomeVersion,ta as archiveAnalyticsActivationDestination,Y as archiveAnalyticsCampaignLink,ga as assignAnalyticsIdentityNode,fa as backfillAnalyticsConfirmedHistory,oa as claimAnalyticsCrmImportRows,Ea as claimAnalyticsFormDeliveryJobs,c as closeAnalyticsPool,pa as completeAnalyticsCrmImportRow,Ga as completeAnalyticsFormBridgeDelivery,Fa as completeAnalyticsFormDelivery,J as consumeAnalyticsSurveyInvite,ra as createAnalyticsActivationDestination,W as createAnalyticsCampaignLink,K as createAnalyticsConversion,ma as createAnalyticsCrmImport,Na as createAnalyticsExport,_ as createAnalyticsForm,o as createAnalyticsPixel,h as createAnalyticsSite,qa as deferAnalyticsCrmImportRow,Ha as deferAnalyticsFormDelivery,n as deleteAnalyticsSite,ea as deterministicAnalyticsCrmEntityId,ja as enrichAnalyticsExistingCrmIdentity,ua as getAnalyticsActivationDestinationConnectionRef,la as getAnalyticsPersonJourney,b as getAnalyticsPool,aa as getPublicAnalyticsForm,da as identityHmac,F as ingestAnalyticsEvents,G as insertAnalyticsRevenueSetupRevision,Ia as isAnalyticsFormPlacementApproved,ia as linkAnalyticsFormIdentity,ha as linkAnalyticsIdentityInTransaction,sa as listAnalyticsActivationDestinations,xa as listAnalyticsActivationReceipts,X as listAnalyticsCampaignLinks,na as listAnalyticsCrmImports,$ as listAnalyticsForms,t as listAnalyticsHostGroups,ka as listAnalyticsPeople,p as listAnalyticsPixels,i as listAnalyticsSites,d as migrateAnalytics,C as normalizeAnalyticsPath,D as normalizeAnalyticsUrl,Q as normalizeContentOptions,e as normalizeObservedHostname,za as pollAnalyticsActivationDiagnostics,u as prepareAnalyticsLinkerIssue,I as projectAnalyticsAuthoritativeConversion,y as projectAnalyticsPixelEventConsent,Aa as queueAnalyticsActivation,Ca as queueAnalyticsFormBridgeTransaction,Ba as queueAnalyticsFormDelivery,ba as recordAnalyticsFormSubmission,v as redeemAnalyticsLinkerRecord,Ka as refreshAnalyticsDailyRollups,La as refreshAnalyticsDailyRollupsIfDue,f as requireAnalyticsAccess,g as requireAnalyticsEditor,Z as resolveAnalyticsCampaignLink,z as resolveAnalyticsConfirmedActivationIdentity,ya as retryAnalyticsActivationJob,E as sanitizeAnalyticsProperties,ca as sanitizeClickIds,va as setAnalyticsActivationReadiness,r as setAnalyticsPixelDomainState,Da as sweepAnalyticsRestrictedRetention,wa as testAnalyticsActivationDestination,j as updateAnalyticsBusinessModel,q as updateAnalyticsPixel,k as upsertAnalyticsAdSpend,s as upsertAnalyticsHostGroup};
|
package/dist/bin/api-server.js
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import{readFileSync as s}from"fs";function c(){try{for(let r of s(".env","utf8").split(`
|
|
3
|
-
`)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-
|
|
3
|
+
`)){let o=r.indexOf("=");if(o<1||r.trimStart().startsWith("#"))continue;let e=r.slice(0,o).trim();process.env[e]||(process.env[e]=r.slice(o+1).trim())}}catch{}}c();async function a(){let[{serve:r},{app:o},{startWorker:e},{migrate:i}]=await Promise.all([import("@hono/node-server"),import("../server-MPVSSCAH.js"),import("../worker-NKLJTKDQ.js"),import("../db-B5XJTOGN.js")]),n=parseInt(process.env.PORT??"3001");try{if(await i(),process.env.ANALYTICS_DATABASE_URL){let{migrateAnalytics:t}=await import("../analytics-repository-IHOFBSUV.js");await t()}e(),r({fetch:o.fetch,port:n},t=>{console.log(`[server] http://localhost:${t.port}`),console.log(`[server] admin auth: ${process.env.ADMIN_KEY?"configured":"not configured"}`)})}catch(t){console.error("[startup] server preflight failed",t instanceof Error?t.name:"unknown_error"),process.exit(1)}}a();
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-JHMI6HEO.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-2TZBO52D.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import{a as P}from"../chunk-
|
|
2
|
+
import{b as E,c as N,d as M,e as L,f as T,j as U}from"../chunk-JHMI6HEO.js";import"../chunk-KJQXUZ4Y.js";import"../chunk-2TZBO52D.js";import{g as v,j as b,k as D,l as K,m as H}from"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import{a as P}from"../chunk-EVQA7RMG.js";import{Command as he}from"commander";import{spawn as ne}from"child_process";import{mkdir as ke,writeFile as Pe}from"fs/promises";import{basename as Ce,join as Z}from"path";function se(e){return e.apiKey?.trim()||"sk_live_your_key"}function ce(e){return e.packageSpec?.trim()||"mcp-scraper@latest"}function A(e={}){return["-y","--package",ce(e),"mcp-scraper"]}function pe(e){let n={MCP_SCRAPER_API_KEY:se(e)},c=e.browserProfileName?.trim();return c&&(n.BROWSER_AGENT_PROFILE_NAME=c),e.browserProfileSaveChanges===!0&&(n.BROWSER_AGENT_PROFILE_SAVE_CHANGES="true"),n}function q(){return["mcp","remove","mcp-scraper","-s","user"]}function J(){return["mcp","get","mcp-scraper"]}function B(e){let n=e.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];if(!n)return null;let c=e.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1]??"",i=c.length?c.split(/\s+/):[],p={},u=e.split(/^\s*Environment:\s*$/m)[1];if(u)for(let a of u.split(`
|
|
3
3
|
`)){let l=a.match(/^\s{2,}([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);if(!l){if(a.trim().length&&!/^\s{2,}/.test(a))break;continue}p[l[1]]=l[2]}return{command:n,args:i,env:p}}function j(e){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(e.env))n.push("--env",`${c}=${i}`);return n.push("--",e.command,...e.args),n}function G(e={}){let n=["mcp","add","mcp-scraper","--scope","user"];for(let[c,i]of Object.entries(pe(e)))n.push("--env",`${c}=${i}`);return n.push("--","npx",...A(e)),n}function O(e){if(e==="claude-code")return"claude";if(e==="claude"||D.hosts.some(n=>n.id===e))return e;throw new Error('Unknown host "'+e+'". Use: codex, claude, claude-code, claude-desktop, cursor, windsurf, cline, or user-action-only')}function ue(e){return K(e==="claude"?"claude-code":e)}function W(e,n={}){let c=O(e),i=ue(c),p="Restart the MCP client so it starts a fresh npx process.",u='MCP_SCRAPER_API_KEY="$MCP_SCRAPER_API_KEY" npx -y -p mcp-scraper@latest mcp-scraper-cli agent install claude --apply',a=`X-Ray install protocol: ${v} (${b})`;return c==="codex"?["# Codex MCP config",a,i.exactConfig,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
|
|
4
4
|
`):c==="claude"?["# Claude Code command",a,i.exactConfig,"","# One-command Claude Code setup",u,"",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,"",p].join(`
|
|
5
5
|
`):c==="claude-desktop"?["# Claude Desktop config",a,i.exactConfig,"","Desktop Extension: https://mcpscraper.dev/downloads/mcp-scraper.mcpb",`Continuation: ${i.continuation}`,`Rollback: ${i.rollback}`,p].join(`
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as e}from"../chunk-
|
|
2
|
+
import{a as e}from"../chunk-6KASNPKR.js";import"../chunk-MWUKEBQA.js";import"../chunk-KZV2FLGG.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-72YO6PKN.js";import"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-M5TYS4EX.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import"../chunk-EVQA7RMG.js";import"../chunk-WJ4XFLS4.js";var _=["harvest_paa","search_serp","extract_url","diff_page","map_site_urls","map_wayback_snapshots","extract_site","analyze_site_similarity","audit_site","check_site_export","site_export_read","site_export_image","archive_read","youtube_harvest","youtube_transcribe","facebook_page_intel","facebook_ad_search","reddit_thread","reddit_trending","video_frame_analysis","video_frame_analysis_status","facebook_ad_transcribe","google_ads_search","google_ads_page_intel","google_ads_transcribe","facebook_video_transcribe","instagram_profile_content","instagram_media_download","maps_place_intel","maps_search","trustpilot_reviews","g2_reviews","capture_serp_snapshot","capture_serp_page_snapshots"];e({toolsets:new Set(["paa","serp"]),allowedToolNames:_});
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as s}from"../chunk-
|
|
2
|
+
import{a as s}from"../chunk-M5TYS4EX.js";import{a as e}from"../chunk-EVQA7RMG.js";var r=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY,n=process.argv.includes("--help")||process.argv.includes("-h");n&&(process.stdout.write(["Usage: mcp-scraper-install [--no-color]","","Prints the branded MCP Scraper terminal install card and copyable install commands.","mcp-scraper prints the same card in a human terminal and runs as the MCP stdio server in clients.",""].join(`
|
|
3
3
|
`)),process.exit(0));process.stdout.write(s({version:e,color:!r,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()}));
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{a as r}from"../chunk-
|
|
2
|
+
import{a as r}from"../chunk-6KASNPKR.js";import"../chunk-MWUKEBQA.js";import"../chunk-KZV2FLGG.js";import"../chunk-W2BVJ7S2.js";import"../chunk-RK2VCTZI.js";import"../chunk-ORB4RHCK.js";import"../chunk-TMB56NCA.js";import"../chunk-HUV2WTRW.js";import"../chunk-YGBTTW5D.js";import"../chunk-72YO6PKN.js";import"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-M5TYS4EX.js";import"../chunk-WO3N5FH2.js";import"../chunk-HE45FFBU.js";import"../chunk-2PXP7TMY.js";import"../chunk-EVQA7RMG.js";import"../chunk-WJ4XFLS4.js";r();
|
package/dist/bin/paa-harvest.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import{w as t}from"../chunk-
|
|
2
|
+
import{w as t}from"../chunk-ZMNUI5LB.js";import"../chunk-M5QHXNFZ.js";import{a as r}from"../chunk-4FROKQJN.js";import"../chunk-2SP57VCG.js";import"../chunk-6DTXIZY2.js";import"../chunk-2TZBO52D.js";import"../chunk-2PXP7TMY.js";import"../chunk-WJ4XFLS4.js";import{Command as s,Option as a}from"commander";var i=new s;i.name("paa-harvest").description("Recursively extract Google People Also Ask questions").requiredOption("-q, --query <query>","Seed query").option("-l, --location <location>",'Location name (e.g. "austin" or "Austin,Texas,United States")').option("--gl <gl>","Google country code","us").option("--hl <hl>","Google language code","en").option("-d, --depth <depth>","BFS depth (1-30)","3").option("-m, --max-questions <n>","Max questions to harvest","100").option("-o, --output <dir>","Output directory","./paa-output").option("-f, --format <format>","Output format: json, csv, or both","both").option("--headless","Run browser in headless mode",!1).option("--profile <dir>","Persistent browser profile directory").option("--proxy <url>","Proxy server URL").option("--browser-api-key <key>","Browser service API key (or set BROWSER_SERVICE_API_KEY env var)").addOption(new a("--\u006b\u0065\u0072\u006e\u0065\u006c-api-key <key>").hideHelp()).action(async e=>{try{let o=await t({query:e.query,location:e.location,gl:e.gl,hl:e.hl,depth:parseInt(e.depth,10),maxQuestions:parseInt(e.maxQuestions,10),outputDir:e.output,format:e.format,headless:e.headless,profileDir:e.profile,proxy:e.proxy,\u006b\u0065\u0072\u006e\u0065\u006cApiKey:e.browserApiKey??e.\u006b\u0065\u0072\u006e\u0065\u006cApiKey??r()});console.log(JSON.stringify({totalQuestions:o.totalQuestions,outputDir:o.stats.seed}))}catch(o){console.error(o instanceof Error?o.message:String(o)),process.exit(1)}});async function n(){await i.parseAsync()}n();
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{
|
|
1
|
+
import{c as l}from"./chunk-6DTXIZY2.js";import{t as _}from"./chunk-WJ4XFLS4.js";var c=166667e-10,N=.0001333336,I=new Set(["serp","fb_search","fb_ad","instagram"]),p=.00111,L=4,R=4e-4;function o(e,r){let t=process.env[e]?.trim();if(!t)return r;let s=Number(t);return Number.isFinite(s)&&s>=0?s:r}var A=o("NANGO_USD_PER_CONNECTION_MONTH",1),U=o("NANGO_USD_PER_FUNCTION_RUN",1e-4),m=o("NANGO_USD_PER_PROXY_REQUEST",1e-4),g=o("NANGO_USD_PER_COMPUTE_SEC",2e-4),O=o("\u0042\u0052\u0049\u0047\u0048\u0054\u0044\u0041\u0054\u0041_BROWSER_USD_PER_GB",8),D=o("\u0042\u0052\u0049\u0047\u0048\u0054\u0044\u0041\u0054\u0041_SERP_USD_PER_REQUEST",.002836168);function u(e,r){return Math.max(0,e)/1e3*(r?N:c)}function d(e,r){return e==="fal_wizper"?Math.max(0,r)/L*p:e==="deepinfra_qwen"?Math.max(0,r)/1e3*R:e==="openrouter"||e==="mcp_memory_video"||e==="mcp_memory_ai"?Math.max(0,r):e==="nango_connection"?Math.max(0,r)*A:e==="nango_function_run"?Math.max(0,r)*U:e==="nango_proxy_request"?Math.max(0,r)*m:e==="nango_compute"?Math.max(0,r)*g:e==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061_browser_api"?Math.max(0,r)/1e9*O:0}import{randomUUID as i}from"crypto";var E=!1,n=null;async function T(){if(!E)return n||(n=S().finally(()=>{n=null}),n)}async function S(){let e=_(),r=await e.execute(`
|
|
2
2
|
SELECT
|
|
3
3
|
(SELECT COUNT(*) FROM sqlite_master WHERE type = 'table' AND name IN ('\u006b\u0065\u0072\u006e\u0065\u006c_session_log', 'vendor_usage_log', 'cost_probe_runs')) = 3
|
|
4
4
|
AND (SELECT COUNT(*) FROM pragma_table_info('\u006b\u0065\u0072\u006e\u0065\u006c_session_log') WHERE name IN ('proxy_source', 'proxy_type', 'method')) = 3
|
|
@@ -6,7 +6,7 @@ import{t as _}from"./chunk-TOJBPATJ.js";import{AsyncLocalStorage as N}from"async
|
|
|
6
6
|
AND (SELECT COUNT(*) FROM sqlite_master WHERE type = 'index' AND name = 'vendor_usage_log_vendor_source_key') = 1
|
|
7
7
|
AND (SELECT COUNT(*) FROM pragma_table_info('cost_probe_runs') WHERE name IN ('units', 'unit_type', 'mode')) = 3
|
|
8
8
|
AS ready
|
|
9
|
-
`);if(Number(
|
|
9
|
+
`);if(Number(r.rows[0]?.ready??0)===1){E=!0;return}await e.execute(`
|
|
10
10
|
CREATE TABLE IF NOT EXISTS \u006b\u0065\u0072\u006e\u0065\u006c_session_log (
|
|
11
11
|
id TEXT PRIMARY KEY,
|
|
12
12
|
\u006b\u0065\u0072\u006e\u0065\u006c_session_id TEXT,
|
|
@@ -44,7 +44,7 @@ import{t as _}from"./chunk-TOJBPATJ.js";import{AsyncLocalStorage as N}from"async
|
|
|
44
44
|
provider_status TEXT,
|
|
45
45
|
created_at TEXT NOT NULL DEFAULT (datetime('now'))
|
|
46
46
|
)
|
|
47
|
-
`),await e.execute("CREATE INDEX IF NOT EXISTS vendor_usage_log_op ON vendor_usage_log(op)"),await e.execute("CREATE INDEX IF NOT EXISTS vendor_usage_log_probe ON vendor_usage_log(probe_run_id)"),await e.execute("CREATE INDEX IF NOT EXISTS vendor_usage_log_created ON vendor_usage_log(created_at)");try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN method TEXT")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN source_key TEXT")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN provider_duration_ms INTEGER")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN provider_captcha INTEGER")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN provider_status TEXT")}catch{}await e.execute("CREATE UNIQUE INDEX IF NOT EXISTS vendor_usage_log_vendor_source_key ON vendor_usage_log(vendor, source_key) WHERE source_key IS NOT NULL"),await e.execute(
|
|
47
|
+
`),await e.execute("CREATE INDEX IF NOT EXISTS vendor_usage_log_op ON vendor_usage_log(op)"),await e.execute("CREATE INDEX IF NOT EXISTS vendor_usage_log_probe ON vendor_usage_log(probe_run_id)"),await e.execute("CREATE INDEX IF NOT EXISTS vendor_usage_log_created ON vendor_usage_log(created_at)");try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN method TEXT")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN source_key TEXT")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN provider_duration_ms INTEGER")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN provider_captcha INTEGER")}catch{}try{await e.execute("ALTER TABLE vendor_usage_log ADD COLUMN provider_status TEXT")}catch{}await e.execute("CREATE UNIQUE INDEX IF NOT EXISTS vendor_usage_log_vendor_source_key ON vendor_usage_log(vendor, source_key) WHERE source_key IS NOT NULL"),await e.execute(`
|
|
48
48
|
CREATE TABLE IF NOT EXISTS cost_probe_runs (
|
|
49
49
|
id TEXT PRIMARY KEY,
|
|
50
50
|
tool TEXT NOT NULL,
|
|
@@ -66,8 +66,8 @@ import{t as _}from"./chunk-TOJBPATJ.js";import{AsyncLocalStorage as N}from"async
|
|
|
66
66
|
margin_usd_headful_starter REAL NOT NULL DEFAULT 0,
|
|
67
67
|
notes TEXT
|
|
68
68
|
)
|
|
69
|
-
`),await e.execute("CREATE INDEX IF NOT EXISTS cost_probe_runs_tool ON cost_probe_runs(tool)");try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN units REAL")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN unit_type TEXT")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN mode TEXT")}catch{}E=!0}async function
|
|
69
|
+
`),await e.execute("CREATE INDEX IF NOT EXISTS cost_probe_runs_tool ON cost_probe_runs(tool)");try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN units REAL")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN unit_type TEXT")}catch{}try{await e.execute("ALTER TABLE cost_probe_runs ADD COLUMN mode TEXT")}catch{}E=!0}async function f(e){try{await T();let r=l(),t=Math.max(0,e.closedAtMs-e.openedAtMs);await _().execute({sql:`INSERT INTO \u006b\u0065\u0072\u006e\u0065\u006c_session_log
|
|
70
70
|
(id, \u006b\u0065\u0072\u006e\u0065\u006c_session_id, op, source, probe_run_id, user_id, stealth, headless_sent, proxy_used, proxy_source, proxy_type, fallback, opened_at, closed_at, duration_ms, est_cost_usd_headless, est_cost_usd_headful, error, method)
|
|
71
|
-
VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[
|
|
71
|
+
VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[i(),e.\u006b\u0065\u0072\u006e\u0065\u006cSessionId??null,r?.op??null,e.source,r?.probeRunId??null,r?.userId??null,a(e.stealth),a(e.headlessSent),a(e.proxyUsed),e.proxySource??null,e.proxyType??null,e.fallback?1:0,new Date(e.openedAtMs).toISOString(),new Date(e.closedAtMs).toISOString(),t,u(t,!1),u(t,!0),e.error??null,r?.subOp??null]})}catch(r){console.warn("[cost-telemetry] record\u004b\u0065\u0072\u006e\u0065\u006cSession failed:",r instanceof Error?r.message:String(r))}}async function v(e){try{await T();let r=l();return(await _().execute({sql:`INSERT OR IGNORE INTO vendor_usage_log
|
|
72
72
|
(id, op, probe_run_id, user_id, vendor, model, units, unit_type, est_cost_usd, error, method, source_key, provider_duration_ms, provider_captcha, provider_status)
|
|
73
|
-
VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[
|
|
73
|
+
VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)`,args:[i(),e.op??r?.op??null,e.probeRunId??r?.probeRunId??null,e.userId??r?.userId??null,e.vendor,e.model??null,e.units,e.unitType,d(e.vendor,e.units),e.error??null,e.method??r?.subOp??null,e.sourceKey??null,e.providerDurationMs??null,a(e.providerCaptcha),e.providerStatus??null]})).rowsAffected===1}catch(r){return console.warn("[cost-telemetry] recordVendorUsage failed:",r instanceof Error?r.message:String(r)),!1}}function a(e){return e==null?null:e?1:0}export{c as a,N as b,I as c,u as d,d as e,T as f,f as g,v as h};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import{a as oe}from"./chunk-XG6GCEUE.js";import{w as V}from"./chunk-ZMNUI5LB.js";import{b as E,c as te}from"./chunk-KZV2FLGG.js";import{b as W,c as X,d as Y,f as Z,g as w,h as j,k as S,n as ee,q as re}from"./chunk-ORB4RHCK.js";import{a as k,b as A}from"./chunk-72YO6PKN.js";import{a as D}from"./chunk-4FROKQJN.js";import{a as F,b as Q,d as G}from"./chunk-2SP57VCG.js";import{b as J}from"./chunk-6DTXIZY2.js";import{Nb as d,Rb as y,Tb as v,Va as g,b as f,g as c,h as T,hb as B,ia as q,j as L,ja as K,jb as h,la as $,m as z,o as U,oa as _}from"./chunk-WJ4XFLS4.js";async function P(e,t,a=3){for(let i=1;i<=a;i++){try{let n=await fetch(e,{method:"POST",headers:{"content-type":"application/json"},body:JSON.stringify(t),signal:AbortSignal.timeout(1e4)});if(n.ok)return;console.warn(`[webhook] attempt ${i} \u2192 ${n.status} from ${e}`)}catch(n){console.warn(`[webhook] attempt ${i} failed:`,n instanceof Error?n.message:n)}i<a&&await new Promise(n=>setTimeout(n,1e3*i*2))}console.error(`[webhook] gave up after ${a} attempts for ${e}`)}function de(e){return e instanceof Error?e.message:String(e)}function ue(e,t){return e instanceof DOMException&&(e.name==="TimeoutError"||e.name==="AbortError")?!0:/timeout|timed out|Timeout \d+ms exceeded|deadline/i.test(t)}function pe(e){return/captcha|recaptcha|unusual traffic|google\.com\/sorry|blocked/i.test(e)}function me(e){return/ERR_TUNNEL_CONNECTION_FAILED|ERR_PROXY_CONNECTION_FAILED|ERR_SOCKS_CONNECTION_FAILED|tunnel connection failed|proxy connection failed|transport error: proxy/i.test(e)}function be(e){return/proxy unavailable|proxy_unavailable|connection_test_failed|did not return a proxy id|configured fallback/i.test(e)}function fe(e){return/(?:spending|billing|organization|account).{0,80}(?:cap|limit|blocked|disabled)|(?:quota|capacity).{0,40}(?:reached|exceeded)|https?:\/\/\S*dashboard/i.test(e)}function _e(e){return/browser (?:has been )?closed|context (?:has been )?closed|target page, context or browser has been closed|session closed|page has been closed/i.test(e)}function M(e){let t=de(e);return e instanceof z?{error_code:"request_aborted",error_type:"request_aborted",message:c("request_aborted"),retryable:!0,httpStatus:408,terminalStatus:"cancelled"}:e instanceof L||pe(t)?{error_code:"captcha_exhausted",error_type:"captcha",message:c("captcha_exhausted"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:oe(e,t)?{error_code:"vendor_unavailable",error_type:"service_unavailable",message:c("vendor_unavailable"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:e instanceof U?{error_code:"location_mismatch",error_type:"location_mismatch",message:c("location_mismatch"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:me(t)?{error_code:"proxy_tunnel_failed",error_type:"connection",message:c("proxy_tunnel_failed"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:be(t)?{error_code:"proxy_unavailable",error_type:"connection",message:c("proxy_unavailable"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:_e(t)?{error_code:"browser_session_interrupted",error_type:"extraction",message:c("browser_session_interrupted"),retryable:!0,httpStatus:503,terminalStatus:"failed"}:ue(e,t)?{error_code:"harvest_timeout",error_type:"timeout",message:c("harvest_timeout"),retryable:!0,httpStatus:504,terminalStatus:"failed"}:fe(t)?{error_code:"service_unavailable",error_type:"service_unavailable",message:f,retryable:!0,httpStatus:503,terminalStatus:"failed"}:{error_code:"extraction_failed",error_type:"extraction",message:f,retryable:!1,httpStatus:500,terminalStatus:"failed"}}function x(e){return JSON.stringify({error_code:e.error_code,error_type:e.error_type,message:e.message,retryable:e.retryable})}function u(e,t={}){let a=T({errorCode:e.error_code,errorType:e.error_type,retryable:e.retryable,retryAfterSeconds:t.retryAfterSeconds,chargeStatus:t.chargeStatus,details:t.details});return{error:a.message,...a}}function O(e,t={}){let{error:a,...i}=u(e,t);return i}function xe(e,t={}){let a=M(e);return{body:u(a,t),status:a.httpStatus}}import{createHash as ye}from"crypto";var p={source:"https://\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061.com/users/zone/premium_domains",retrievedAt:"2026-09-23T20:14:30.483Z",includedStatuses:["premium","remove_candidate"],domains:["advanceautoparts.com","affitto.it","agoda.cn","albertsons.com","allpeople.com","autozone.com","bestbuy.com","bestwestern.com","billiger.de","bottlerover.com","carousell.com","carousell.com.hk","carousell.com.my","carousell.ph","carousell.sg","carsales.com.au","cdiscount.com","chewy.com","costco.com","cvs.com","despegar.com.mx","dickssportinggoods.com","dynos.es","emaxme.com","familytreenow.com","feuvert.fr","flooranddecor.com","foodlion.com","footlocker.co.uk","footlocker.com","giantfoodstores.com","gopuff.com","gplay.bg","hermes.com","hyatt.com","idealo.de","immobilienscout24.de","ingatlan.com","instacart.com","intersport.fr","joann.com","kroger.com","lazada.co.id","lazada.co.th","lazada.com.my","lazada.com.ph","lazada.sg","lazada.vn","lowes.ca","lowes.com","mcmaster.com","mediamarkt.de","mediamarkt.es","medline.com","mscdirect.com","napaonline.com","nofrills.ca","peoplefinders.com","platt.com","publicdatausa.com","realcanadiansuperstore.ca","realestate.com.au","restaurantguru.com","searchpeoplefree.com","shopee.cl","shopee.co.id","shopee.co.th","shopee.com.br","shopee.com.co","shopee.com.mx","shopee.com.my","shopee.ph","shopee.sg","shopee.tw","shopee.vn","similarweb.com","skyscanner.co.kr","skyscanner.net","stopandshop.com","target.com","temu.com","ticketmaster.com","totalwine.com","tractorsupply.com","walmart.com.mx","wayfair.com","weismarkets.com","wizzair.com","worten.pt"]};var He=p.source,Ce=p.retrievedAt,he=new Set(p.domains);function I(e){let t=new URL(e).hostname.toLowerCase().replace(/\.$/,"");for(let a of he)if(t===a||t.endsWith(`.${a}`))return"premium";return"standard"}var ve=1e9;function we(e){return ye("sha256").update(JSON.stringify(e)).digest("hex")}function m(e){return`harvest:${e}`}async function Ue(e){let t=m(e.jobId);try{return(await W({id:t,userId:e.userId,ownerScope:`user:${e.userId}`,tool:e.tool,normalizedFlags:e.normalizedFlags,idempotencyKey:`job:${e.jobId}`,requestFingerprint:we(e.normalizedFlags),jobId:e.jobId,requestId:e.requestId??null,deploymentCommit:process.env.VERCEL_GIT_COMMIT_SHA?.trim()||process.env.GIT_COMMIT_SHA?.trim()||null})).id}catch(a){return console.warn(JSON.stringify({event:"operation_run_start_failed",job_id:e.jobId,tool:e.tool,message:a instanceof Error?a.message:String(a)})),null}}async function H(e,t,a){if(e)try{await X({runId:e,status:t,failureClass:a})}catch(i){console.warn(JSON.stringify({event:"operation_run_finish_failed",operation_run_id:e,message:i instanceof Error?i.message:String(i)}))}}async function qe(e,t,a,i){if(!(!e||!t))try{await ee({runId:e,billingTable:"billing_debits",billingKey:t,relation:a,amountMc:i})}catch(n){console.warn(JSON.stringify({event:"operation_billing_link_failed",operation_run_id:e,relation:a,message:n instanceof Error?n.message:String(n)}))}}function R(e){return e==="timeout"?"timed_out":e==="request_aborted"?"cancelled":e==="browser_session_interrupted"?"interrupted":e==="captcha"||e==="proxy_tunnel_failed"||e==="proxy_unavailable"||e==="location_mismatch"||e==="error"?"failed":"succeeded"}function Se(e){return e==="\u0062\u0072\u0069\u0067\u0068\u0074_\u0064\u0061\u0074\u0061_browser_api"?"\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061":"\u006b\u0065\u0072\u006e\u0065\u006c"}function ae(e,t,a=0){let i=new Map,n=null;return async o=>{if(e)try{if(o.type==="started"){let s=a+o.attemptNumber;if(s>1&&!n){let N=(await re(e))?.attempts.at(-1);n=N?.id==null?null:String(N.id)}let l=await Y({runId:e,attemptNumber:s,method:o.method,provider:o.provider,parentAttemptId:s===1?null:n,edgeKind:s===1?null:a>0&&o.attemptNumber===1?"durable_resume":o.retryKind==="provider_fallback"?"provider_fallback":"same_method_retry",retryKind:s===1?null:a>0&&o.attemptNumber===1?"durable_resume":o.retryKind,retryReason:s===1?null:a>0&&o.attemptNumber===1?"durable_worker_resume":o.retryReason,retryOwner:s===1?null:a>0&&o.attemptNumber===1?"durable_reconciler":"harvest",normalizedFlags:{maxAttempts:o.maxAttempts,maxQuestions:o.maxQuestions,hasLocation:!!o.location,plannedProvider:o.provider,planReason:o.planReason??null,planTimeoutMs:o.planTimeoutMs??null},startedAt:o.startedAt}),b=await w({attemptId:l,provider:o.provider,providerProduct:o.provider==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061"?"\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061_browser_api":"\u006b\u0065\u0072\u006e\u0065\u006c_browser",billingTier:o.provider==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061"?I("https://www.google.com/search"):null,accountScope:t,method:o.method,providerStatus:"started"});i.set(o.attemptNumber,{attemptId:l,receiptId:b.id,plannedProvider:o.provider,observedProvider:null,providerMismatch:!1,method:o.method,startedAtMs:Date.parse(o.startedAt),billingMode:null}),n=l;return}let r=i.get(o.attemptNumber);if(!r)throw new Error(`Missing operation attempt state for attempt ${o.attemptNumber}`);if(o.type==="provider_session_started"){let s=Se(o.provider);if(r.observedProvider=s,r.billingMode=o.billingMode??null,s!==r.plannedProvider){r.providerMismatch=!0,await S({receiptId:r.receiptId,costState:"unavailable",providerStatus:"provider_mismatch",errorCode:"provider_mismatch",errorMessage:`Planned ${r.plannedProvider}; observed ${s}`,finalizedAt:o.observedAt});let l=s==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061"?"\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061_browser":"\u006b\u0065\u0072\u006e\u0065\u006c_browser",b=await w({attemptId:r.attemptId,provider:s,providerProduct:s==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061"?"\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061_browser_api":"\u006b\u0065\u0072\u006e\u0065\u006c_browser",billingTier:s==="\u0062\u0072\u0069\u0067\u0068\u0074\u0064\u0061\u0074\u0061"?I("https://www.google.com/search"):null,accountScope:t,method:l,providerStatus:"observed_provider_mismatch"});r.receiptId=b.id,r.method=l}await j(r.receiptId,o.providerSessionId);return}if(await Z({attemptId:r.attemptId,status:r.providerMismatch?"failed":R(o.outcome),errorCode:r.providerMismatch?"provider_mismatch":R(o.outcome)==="succeeded"?null:o.outcome,errorMessage:o.error,durationMs:o.durationMs,endedAt:o.completedAt}),(r.observedProvider??r.plannedProvider)==="\u006b\u0065\u0072\u006e\u0065\u006c"){let s=r.billingMode==="headful",l=r.billingMode==="headless"||r.billingMode==="headful"?G(o.durationMs,s):null;await S({receiptId:r.receiptId,costState:"estimated",units:o.durationMs/1e3,unitType:"wall_open_second",amountUsdNanos:l==null?null:Math.round(l*ve),providerProduct:"\u006b\u0065\u0072\u006e\u0065\u006c_browser",billingTier:r.billingMode,billableEvent:"provider_active_gb_second",measurementSource:"wall_open_time_proxy",rateSource:l!=null?`\u006b\u0065\u0072\u006e\u0065\u006c_${r.billingMode}_usd_per_second:${s?Q:F}`:null,providerStatus:r.providerMismatch?"provider_mismatch":o.outcome,providerDurationMs:o.durationMs,errorCode:r.providerMismatch?"provider_mismatch":R(o.outcome)==="succeeded"?null:o.outcome,errorMessage:o.error,finalizedAt:o.completedAt})}}catch(r){console.warn(JSON.stringify({event:"operation_attempt_telemetry_failed",operation_run_id:e,attempt_number:o.attemptNumber,message:r instanceof Error?r.message:String(r)}))}}}function ie(e,t,a=null,i="production-default",n=0){let o=ae(a,i,n);return async r=>{if(await o(r),r.type==="started"){await q({jobId:e,userId:t,attemptNumber:r.attemptNumber,maxAttempts:r.maxAttempts,query:r.query,location:r.location,maxQuestions:r.maxQuestions,startedAt:r.startedAt});return}if(r.type==="provider_session_started"){await K({jobId:e,attemptNumber:r.attemptNumber,browserProvider:r.provider,providerSessionId:r.providerSessionId,observedAt:r.observedAt});return}await $({jobId:e,attemptNumber:r.attemptNumber,outcome:r.outcome,\u006b\u0065\u0072\u006e\u0065\u006cSessionId:r.\u006b\u0065\u0072\u006e\u0065\u006cSessionId,questionCount:r.questionCount,durationMs:r.durationMs,error:r.error,willRetry:r.willRetry,\u006b\u0065\u0072\u006e\u0065\u006cDeleteStarted:r.cleanup.\u006b\u0065\u0072\u006e\u0065\u006cDeleteStarted,\u006b\u0065\u0072\u006e\u0065\u006cDeleteSucceeded:r.cleanup.\u006b\u0065\u0072\u006e\u0065\u006cDeleteSucceeded,\u006b\u0065\u0072\u006e\u0065\u006cDeleteError:r.cleanup.\u006b\u0065\u0072\u006e\u0065\u006cDeleteError,browserCloseSucceeded:r.cleanup.browserCloseSucceeded,browserCloseError:r.cleanup.browserCloseError,browserProvider:r.cleanup.provider??null,providerSessionId:r.cleanup.providerSessionId??null,providerLookupStatus:r.cleanup.providerSessionId?"pending":"not_applicable",providerErrorCode:r.cleanup.providerDisconnectObserved?"browser_session_interrupted":null,providerErrorMessage:r.cleanup.providerDisconnectMessage??null,providerReasonSource:r.cleanup.providerDisconnectObserved?"client_runtime":null,providerDisconnectObserved:r.cleanup.providerDisconnectObserved??!1,providerDisconnectMessage:r.cleanup.providerDisconnectMessage??null,debug:r.debug,completedAt:r.completedAt})}}var se=2,C=0;function ne(e){if(!e||typeof e!="object")return 0;let t=e;return typeof t.totalQuestions=="number"?t.totalQuestions:Array.isArray(t.flat)?t.flat.length:0}function le(e){return k.paa_base+Math.max(1,e)*k.paa}async function ce(e){C++;try{let t=typeof e.options=="string"?JSON.parse(e.options):e.options,a={value:null},i=m(e.id),n=await J({op:t.serpOnly?"serp":"paa",userId:Number(e.user_id),headlessSentOut:a,operationRunId:i,operationAccountScope:"production-default"},()=>V({...t,\u006b\u0065\u0072\u006e\u0065\u006cApiKey:D(),headless:!0,format:"json",outputDir:"/tmp/paa-output-api",onAttemptEvent:ie(e.id,e.user_id,i)}));await B(e.id,n),await H(i,"succeeded");let o=await _(e.id,e.user_id);if(typeof t.billingHoldMc=="number"&&t.billingDebitKey){let r=t.serpOnly?A(a.value):le(ne(n));await v(Number(e.user_id),t.billingDebitKey,r,t.serpOnly?"serp_refund":"paa_refund",t.serpOnly?"SERP search settlement":"PAA harvest settlement",t.serpOnly?"serp_search":"paa_harvest")}else if(!t.serpOnly&&typeof t.billingHoldMc=="number"){let r=le(ne(n)),s=t.billingHoldMc-r;s>0?await d(e.user_id,s,"paa_refund","overestimate refund"):s<0&&await y(e.user_id,-s,"paa",t.query??e.query)}else if(t.serpOnly&&typeof t.billingHoldMc=="number"){let r=A(a.value),s=t.billingHoldMc-r;s>0?await d(e.user_id,s,"serp_refund","headless-mode pricing settle"):s<0&&await y(e.user_id,-s,"serp",t.query??e.query)}e.callback_url&&await P(e.callback_url,{job_id:e.id,status:"done",result:te(n),attempts:E(o)})}catch(t){console.error("[harvest/worker] failed",{jobId:e.id,error:t instanceof Error?t.message:String(t)});let a=M(t);await H(m(e.id),a.error_code==="harvest_timeout"?"timed_out":a.terminalStatus,a.error_code);let i=typeof e.options=="string"?JSON.parse(e.options):e.options,n=typeof i.billingHoldMc=="number"&&i.billingHoldMc>0;await h(e.id,x(a),O(a,{chargeStatus:n?"refund_pending":"not_charged"}));let o=await _(e.id,e.user_id);try{n&&(i.billingDebitKey?await v(Number(e.user_id),i.billingDebitKey,0,i.serpOnly?"serp_refund":"paa_refund","failed call",i.serpOnly?"serp_search":"paa_harvest"):await d(e.user_id,i.billingHoldMc,"refund","failed call"),await h(e.id,x(a),O(a,{chargeStatus:"refunded"})))}catch{}e.callback_url&&await P(e.callback_url,{job_id:e.id,status:"failed",...u(a),attempts:E(o)})}finally{C--}}async function Ee(){let e=await g();if(!e)return{claimed:!1};let t=Date.now();return await ce(e),{claimed:!0,jobId:e.id,completed:!0,durationMs:Date.now()-t}}async function rr(e){let t=[];for(let a=0;a<e.maxJobs&&!(Date.now()>=e.deadlineMs);a++){let i=await Ee();if(t.push(i),!i.claimed)break}return t}function tr(){setInterval(async()=>{if(C>=se)return;let e=await g();e&&ce(e)},2e3),console.log(`[worker] started \u2014 polling every 2s, max ${se} concurrent`)}export{Ce as a,I as b,M as c,x as d,u as e,O as f,xe as g,m as h,Ue as i,H as j,qe as k,ie as l,P as m,Ee as n,rr as o,tr as p};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import{AsyncLocalStorage as n}from"async_hooks";var s="MCP_SCRAPER_ALLOW_HEADLESS_HARVEST_TEST",t=new n;function u(e,o){return t.run(e,o)}function l(){return t.getStore()}export{s as a,u as b,l as c};
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{Ca as R,Da as E,Fa as v,G as u,H as f,Na as T,Pa as h,Qa as w,Ra as _,Sa as y,Ta as C,na as g,pa as S}from"./chunk-
|
|
1
|
+
import{Ca as R,Da as E,Fa as v,G as u,H as f,Na as T,Pa as h,Qa as w,Ra as _,Sa as y,Ta as C,na as g,pa as S}from"./chunk-MWUKEBQA.js";import{a as P}from"./chunk-M5TYS4EX.js";import{a as m}from"./chunk-EVQA7RMG.js";import{readFileSync as N}from"fs";import{homedir as I}from"os";import{join as A}from"path";import{serveStdio as b}from"@modelcontextprotocol/server/stdio";import{McpServer as k}from"@modelcontextprotocol/server";var x=new Map(w.map(r=>[r.upstreamName,r.id]));function O(r){let s=r.split(`
|
|
2
2
|
`).filter(o=>o.startsWith("data:")).map(o=>o.slice(5).trim()).filter(Boolean);if(!s.length)return JSON.parse(r);for(let o of s){let e=JSON.parse(o);if(e.result||e.error)return e}throw new Error("hosted MCP returned no JSON-RPC result")}function a(r){return{content:[{type:"text",text:JSON.stringify({ok:!1,error:r})}],isError:!0}}var c=class{baseUrl;apiKey;constructor(s,o){this.baseUrl=s.replace(/\/$/,""),this.apiKey=o}async callMemoryTool(s,o){let e=x.get(s);if(!e)return a(`unknown memory tool: ${s}`);try{let n=await fetch(`${this.baseUrl}/mcp`,{method:"POST",headers:{"content-type":"application/json",accept:"application/json, text/event-stream","x-api-key":this.apiKey},body:JSON.stringify({jsonrpc:"2.0",id:`memory:${e}`,method:"tools/call",params:{name:e,arguments:o}})}),l=await n.text();if(!n.ok)return a(`hosted memory ${e} failed (HTTP ${n.status})`);let t=O(l);return t.error?a(t.error.message??`hosted memory ${e} failed`):t.result??a(`hosted memory ${e} returned no result`)}catch(n){return a(n instanceof Error?n.message:`hosted memory ${e} call failed`)}}};function M(r,s){if(s.length===0)throw new Error("Restricted MCP tool allowlist must contain at least one tool name");let o=new Set;for(let t of s){if(typeof t!="string"||t.trim()!==t||t.length===0)throw new Error("Restricted MCP tool allowlist contains a malformed tool name");if(o.has(t))throw new Error(`Restricted MCP tool allowlist contains duplicate tool name: ${t}`);o.add(t)}let e=new Set,n=r,l=n.registerTool.bind(r);return n.registerTool=(t,i,p)=>{if(o.has(t))return e.add(t),l(t,i,p)},{assertComplete(){let t=[...o].filter(i=>!e.has(i));if(t.length>0)throw new Error(`Restricted MCP tool allowlist contains unknown or unavailable tool names: ${t.join(", ")}`)},registeredToolNames(){return[...e]}}}var L=new Set(["paa","serp","browser-agent","scheduled-results","memory"]);function $(){let s=[process.env.MCP_SCRAPER_KEY_PATH?.trim(),A(I(),".mcp-scraper-key")].filter(Boolean);for(let o of s)try{let e=N(o,"utf8").trim();if(e)return e}catch{}}function Y(r,s={}){let o=s.toolsets??L,e=process.env.MCP_SCRAPER_BASE_URL?.trim()||process.env.MCP_BASE_URL?.trim()||"https://mcpscraper.dev",n=g(),l=S({deploymentProfile:n,transportProfile:"stdio",baseUrl:e,explicitlyEnabled:process.env.MCP_SCRAPER_ALLOW_PRIVATE_NETWORK==="1"}),t=process.env.BROWSER_AGENT_CONSOLE_URL?.trim()||e,i=new k({name:"mcp-scraper",version:m},{instructions:u,cacheHints:{"server/discover":{ttlMs:3e5,cacheScope:"private"},"tools/list":{ttlMs:3e5,cacheScope:"private"},"prompts/list":{ttlMs:3e5,cacheScope:"private"},"resources/list":{ttlMs:3e5,cacheScope:"private"},"resources/templates/list":{ttlMs:3e5,cacheScope:"private"},"resources/read":{ttlMs:6e4,cacheScope:"private"}}});f(i);let p=s.allowedToolNames?M(i,s.allowedToolNames):void 0,d=s.httpExecutor??new T(e,r,{localNetworkAccess:l});return o.has("paa")&&v(i,d,{ownerId:R(r),deploymentProfile:n,transportProfile:"stdio",baseUrl:e,localNetworkAccess:l,taskHandleSecret:r}),o.has("serp")&&E(i,d,{exposeDevelopmentDiagnostics:n==="development"||n==="test"}),o.has("browser-agent")&&h(i,{baseUrl:e,apiKey:r,consoleBaseUrl:t}),o.has("scheduled-results")&&C(i,new y(e,r)),o.has("memory")&&_(i,new c(e,r)),p?.assertComplete(),i}function se(r={}){let s=process.argv.includes("--stdio")||process.env.MCP_SCRAPER_FORCE_STDIO==="1",o=!!(process.stdin.isTTY&&process.stdout.isTTY),e=process.argv.includes("--help")||process.argv.includes("-h");if(!s&&(o||e)){let l=process.argv.includes("--no-color")||process.env.NO_COLOR!==void 0||process.env.FORCE_COLOR==="0"||!process.stdout.isTTY;process.stdout.write(P({version:m,color:!l,apiKeyConfigured:!!process.env.MCP_SCRAPER_API_KEY?.trim()})),process.exit(0)}let n=(process.env.MCP_SCRAPER_API_KEY??process.env.MCP_SCRAPER_KEY??process.env.MCP_API_KEY??$())?.trim();n||(process.stderr.write(`MCP_SCRAPER_API_KEY env var or ~/.mcp-scraper-key is required
|
|
3
3
|
`),process.exit(1)),b(()=>Y(n,r),{legacy:"serve",onerror(l){process.stderr.write(`${l.message}
|
|
4
4
|
`)}})}export{se as a};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
var p={schemaVersion:1,ratePolicy:{version:"serp-paa-2026-08-11",effectiveDate:"2026-08-11T00:00:00.000Z"},costEvidence:{id:"serp-paa-rate-basis-2026-08",status:"provisional",minimumRevenueCostRatio:2.5,observedRevenueCostRatio:2.5160607750756583},operationCosts:{serp:{credits:60,unit:"per search",label:"SERP search",notes:"Returns organic results and Google entity IDs by default; optional same-page local pack, forums, videos, AI surfaces, and What People Are Saying are enabled with flags at the same product price. search_serp and the base capture_serp_snapshot search are billed 60 credits per search. Optional SERP Intelligence page snapshots add 1 credit per attempted URL."},paa:{credits:10,unit:"per question (+400 credit base)",label:"PAA harvest",notes:"Returns original-query PAA questions with answers and sources plus organic results and Google entity IDs. Optional same-page SERP modules are flag-controlled. Billed 400 credit base + 10 per question actually returned (unused estimate refunded)."},page_scrape:{credits:1,unit:"per page",label:"Page crawl / extract",notes:"Applies to both single-URL extraction and per-page site crawls."},listing_create:{credits:10,unit:"per new listing",label:"Directory or wiki listing creation",notes:"Fixed total price for a new Local Sourcebook directory listing or Transparent Commons wiki entity. Sourcebook acquisition is included."},listing_edit:{credits:2,unit:"per existing listing revision",label:"Directory or wiki listing edit",notes:"Fixed total price for refreshing an existing Local Sourcebook listing or revising an existing Transparent Commons entity."},url_map:{credits:5,unit:"per mapping operation",label:"Site URL mapping",notes:"Flat rate for the full /map-urls call regardless of URL count discovered."},yt_channel:{credits:2,unit:"per call",label:"YouTube search / channel harvest"},yt_transcription:{credits:200,unit:"base + 2/min of video",label:"YouTube transcription",notes:"Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},fb_ad:{credits:7,unit:"per call",label:"Facebook search / ad lookup"},maps_search:{credits:5,unit:"per search",label:"Maps business search",notes:"Returns up to 50 Google Maps business/profile candidates. Use maps_place_intel to hydrate selected businesses."},maps_place:{credits:60,unit:"per business",label:"Maps business lookup",notes:"Base lookup. Reviews billed separately per card at maps_review rate."},maps_review:{credits:1,unit:"per review card",label:"Maps review",notes:"Charged after extraction when includeReviews is true."},fb_search:{credits:8,unit:"per search",label:"Facebook ad library search",notes:"Browser automation to search Facebook Ads Library by keyword."},google_ads_search:{credits:6,unit:"per search",label:"Google Ads Transparency search",notes:"Browser automation to find advertisers in Google Ads Transparency Center by domain or name."},google_ads_intel:{credits:2,unit:"per call",label:"Google Ads Transparency advertiser intel",notes:"Lists and hydrates an advertiser's creatives with image URLs and video references."},google_ads_transcribe:{credits:200,unit:"base + 2/min of video",label:"Google ad video transcription",notes:"Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},fb_transcribe:{credits:200,unit:"base + 2/min of video",label:"Facebook video / ad transcription",notes:"Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},fb_reels_inventory:{credits:4,unit:"per profile scan",label:"Facebook Reel URL inventory",notes:"Collects canonical Reel URLs from a Facebook profile. Logged-out scans are capped at 60 URLs; saved browser profiles can request larger inventories."},tiktok_transcribe:{credits:200,unit:"base + 2/min of video",label:"TikTok video transcription",notes:"Transcribes a public TikTok video from its playback media. Flat 200-credit base plus 2 credits per minute, reconciled after transcription."},instagram_profile:{credits:4,unit:"per profile scan",label:"Instagram profile content discovery",notes:"Browser extraction of public Instagram profile grid links. Complete history may require a logged-in profile."},instagram_media:{credits:4,unit:"per post or reel",label:"Instagram media download",notes:"Extracts post text, image URLs, and reel audio/video tracks, with local downloads when the server can write files."},instagram_transcribe:{credits:200,unit:"base + 2/min of video",label:"Instagram media transcription",notes:"Whisper transcription of selected Instagram audio/video media. Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},browser_minute:{credits:120,unit:"per minute of use",label:"Interactive browser session",notes:"Metered per minute of use for the whole time the browser session is open. Close the session to stop the meter; abandoned sessions are auto-closed after a short idle window."},reddit_thread:{credits:30,unit:"per thread",label:"Reddit thread base lookup",notes:"Base lookup for the post itself. Comments billed separately per comment at reddit_comment rate. Refunded if the thread cannot be retrieved."},reddit_comment:{credits:2,unit:"per comment",label:"Reddit comment",notes:"Charged per comment actually extracted, billed down automatically if the balance runs out mid-thread."},video_analysis:{credits:6667,unit:"per 120 frames (max 480)",label:"Video breakdown (frame-by-frame + transcript)",notes:"Full multi-lens video breakdown: samples frames across a video (up to 30 minutes), analyzes each with vision AI, transcribes the audio, then produces summary, pacing/energy, words-per-minute, topic outline, key points, hook analysis, visual style, and a how-to-replicate recipe with a quality-control pass. $1 per 120 frames requested (max 480 = $4); billed down automatically if the video cannot use the requested frames, and refunded fully if the run fails."},trustpilot_reviews:{credits:5,unit:"per call",label:"Trustpilot review harvest",notes:"Base lookup. Reviews billed separately per card at trustpilot_review rate. Refunded if no reviews are found."},trustpilot_review:{credits:1,unit:"per review card",label:"Trustpilot review card",notes:"Charged per review actually extracted, billed down automatically if the balance runs out mid-page."},g2_reviews:{credits:5,unit:"per call",label:"G2 review harvest",notes:"Base lookup. Reviews billed separately per card at g2_review rate. Refunded if no reviews are found."},g2_review:{credits:2,unit:"per review card",label:"G2 review card",notes:"Charged per review actually extracted (each card carries up to 3 Q&A sections), billed down automatically if the balance runs out mid-page."},diff_page:{credits:1,unit:"per check",label:"Page change check",notes:"Same cost as a single page extract \u2014 one scrape is performed per check, then compared against your last stored snapshot for that URL."}},publicTiers:[{tier:"starter",label:"Starter",monthlyUsd:12,credits:8e4,concurrency:3},{tier:"growth",label:"Growth",monthlyUsd:40,credits:266667,concurrency:10},{tier:"scale",label:"Scale",monthlyUsd:100,credits:8e5,concurrency:20}],credits:{mcPerCredit:100,freeSignupCredits:0},inventory:{totalTools:362,scraperTools:236,memoryTools:126},concurrencyPack:{ratePolicyVersion:"2026-08-14.concurrency-pack-v1",monthlyUsd:5,currency:"usd",interval:"month",billingUnit:"pack",slotsPerPack:2},scaleCreditPack:{ratePolicyVersion:"2026-09-21.scale-credit-pack-v1",eligibleTier:"scale",incrementUsd:10,creditsPerIncrement:8e4,maxQuantity:100,currency:"usd",expiryMonths:3},addons:{memory:{monthlyUsd:19,quotaGb:5,freeQuotaGb:.01},scheduling:{billingMode:"credits",entitlement:"paid_plan_included",separateSubscriptionRequired:!1,chargeEvent:"started_occurrence",runBaseCredits:75,runBaseMc:7500,llmCostMultiplier:1.5,llmCostMultiplierBps:15e3,modes:{connection_sync:{runBaseCredits:75,modelCostBilling:"none",modelCostMultiplier:0},agent:{runBaseCredits:75,modelCostBilling:"actual_reported_cost",modelCostMultiplier:1.5,modelCostMultiplierBps:15e3}}},connectedAccounts:{schemaVersion:2,ratePolicyVersion:"2026-08-03.connected-usage-v3",effectiveAt:"2026-08-03T00:00:00.000Z",entitlement:"paid_plan_add_on",connection:{billingMode:"flat_recurring_credits",amountCredits:15e3,interval:"month",unit:"active_connection",proration:"none"},usage:{billingMode:"credits",settlementMode:"authoritative_receipt",rounding:"aggregate_then_ceil_to_mc",functionRun:{credits:2,mc:200,unit:"per_function_run"},proxyRequest:{credits:2,mc:200,unit:"per_proxy_request"},compute:{creditsPerSecond:5,mcPerSecond:500,sourceUnit:"millisecond",unit:"per_compute_second",formula:"ceil(computeMilliseconds * mcPerSecond / 1000)"}}}}};var t={serp:6e3,serp_headless:6e3,serp_headful:6e3,paa:1e3,paa_base:4e4,page_scrape:100,url_map:500,yt_channel:200,yt_transcription:200,fb_ad:700,maps_search:500,maps_place:6e3,maps_review:100,fb_search:800,fb_search_result:100,fb_page_intel:700,fb_page_intel_ad:50,fb_reels_inventory:400,fb_transcribe:33300,tiktok_transcribe:33300,google_ads_search:600,google_ads_intel:200,google_ads_transcribe:33300,instagram_profile:400,instagram_media:400,instagram_transcribe:33300,browser_minute:12e3,reddit_thread:3e3,reddit_comment:200,video_analysis:666700,trustpilot_reviews:500,trustpilot_review:100,g2_reviews:500,g2_review:200,diff_page:100,listing_create:1e3,listing_edit:200};function ne(e){return e===!0?t.serp_headless:t.serp_headful}var G=t.browser_minute/6e4;function ae(e){return Math.round(e*G)}var oe=1e3,a=100,ce=t.serp_headful/a,le=t.paa_base/a,_e=t.paa/a,de=t.page_scrape/a,Y={billingMode:"credits",chargeEvent:"accepted_write",validationCredits:0,create:{credits:t.listing_create/a,mc:t.listing_create},edit:{credits:t.listing_edit/a,mc:t.listing_edit},surfaces:["local_sourcebook","transparent_commons"],sourcebookAcquisitionIncluded:!0},q=2e6,V=3,T=1e4,C=1e9,H=75,ue=H*a,K=15e3,pe=K/T,me=5e3,W="2026-08-03.connected-usage-v3",y=15e3,be=y*a,v=2,A=2,S=5,N=v*a,I=A*a,w=S*a,ge={schemaVersion:2,ratePolicyVersion:W,effectiveAt:"2026-08-03T00:00:00.000Z",provider:"nango",entitlement:"paid_plan_add_on",connection:{billingMode:"flat_recurring_credits",amountCredits:y,interval:"month",unit:"active_connection",proration:"none"},usage:{billingMode:"credits",settlementMode:"authoritative_receipt",rounding:"aggregate_then_ceil_to_mc",functionRun:{credits:v,mc:N,unit:"per_function_run"},proxyRequest:{credits:A,mc:I,unit:"per_proxy_request"},compute:{creditsPerSecond:S,mcPerSecond:w,sourceUnit:"millisecond",unit:"per_compute_second",formula:"ceil(computeMilliseconds * mcPerSecond / 1000)"}}};function m(e,r){if(!Number.isSafeInteger(e)||e<0)throw new Error(`${r} must be a non-negative safe integer`);return BigInt(e)}function _(e){if(e>BigInt(Number.MAX_SAFE_INTEGER))throw new Error("connected usage charge exceeds the safe integer range");return Number(e)}function fe(e){let r=m(e.functionRuns,"functionRuns"),o=m(e.proxyRequests,"proxyRequests"),n=m(e.computeMilliseconds,"computeMilliseconds"),s=r*BigInt(N),c=o*BigInt(I),l=n*BigInt(w),u=l===0n?0n:(l+999n)/1000n,E=s+c+u;return{functionRunMc:_(s),proxyRequestMc:_(c),computeMc:_(u),amountMc:_(E),credits:_(E)/a}}function Re(e,r){if(!Number.isFinite(e)||e<0)throw new Error("costUsd must be a finite non-negative number");if(!Number.isInteger(r)||r<0)throw new Error("markupBps must be a non-negative integer");let n=BigInt(Math.round(e*C))*BigInt(r)*BigInt(q),s=BigInt(C)*BigInt(T)*BigInt(V),c=n===0n?0n:(n+s-1n)/s;if(c>BigInt(Number.MAX_SAFE_INTEGER))throw new Error("calculated mc exceeds the safe integer range");return Number(c)}var $=[{key:"serp",label:"SERP search",aliases:["search_serp","serp","google search","organic results"],credits:i(t.serp_headful),unit:"per search",notes:`Returns organic results and Google entity IDs by default; optional same-page local pack, forums, videos, AI surfaces, and What People Are Saying are enabled with flags at the same product price. search_serp and the base capture_serp_snapshot search are billed ${i(t.serp_headful)} credits per search. Optional SERP Intelligence page snapshots add ${i(t.page_scrape)} credit per attempted URL.`},{key:"paa",label:"PAA harvest",aliases:["harvest_paa","paa","people also ask","questions"],credits:i(t.paa),unit:`per question (+${i(t.paa_base)} credit base)`,notes:`Returns original-query PAA questions with answers and sources plus organic results and Google entity IDs. Optional same-page SERP modules are flag-controlled. Billed ${i(t.paa_base)} credit base + ${i(t.paa)} per question actually returned (unused estimate refunded).`},{key:"page_scrape",label:"Page crawl / extract",aliases:["extract_url","extract_site","page scrape","url scrape","single page","site crawl"],credits:i(t.page_scrape),unit:"per page",notes:"Applies to both single-URL extraction and per-page site crawls."},{key:"listing_create",label:"Directory or wiki listing creation",aliases:["local-sourcebook-capture","commons_submit_entity","directory listing","wiki listing"],credits:i(t.listing_create),unit:"per new listing",notes:"Fixed total price for a new Local Sourcebook directory listing or Transparent Commons wiki entity. Sourcebook acquisition is included."},{key:"listing_edit",label:"Directory or wiki listing edit",aliases:["local_sourcebook_refresh","commons_submit_entity","listing refresh","wiki edit"],credits:i(t.listing_edit),unit:"per existing listing revision",notes:"Fixed total price for refreshing an existing Local Sourcebook listing or revising an existing Transparent Commons entity."},{key:"url_map",label:"Site URL mapping",aliases:["map_site_urls","url map","site map","crawl urls"],credits:i(t.url_map),unit:"per mapping operation",notes:"Flat rate for the full /map-urls call regardless of URL count discovered."},{key:"yt_channel",label:"YouTube search / channel harvest",aliases:["youtube_harvest","youtube search","youtube channel","yt_channel"],credits:i(t.yt_channel),unit:"per call"},{key:"yt_transcription",label:"YouTube transcription",aliases:["youtube_transcribe","youtube transcript","transcription","yt_transcription"],credits:i(2e4),unit:"base + 2/min of video",notes:"Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},{key:"fb_ad",label:"Facebook search / ad lookup",aliases:["facebook_page_intel","facebook_ad_search","facebook_ad","facebook ads","fb ads"],credits:i(t.fb_ad),unit:"per call"},{key:"maps_search",label:"Maps business search",aliases:["maps_search","google maps search","gmb search","gbp search","business profiles"],credits:i(t.maps_search),unit:"per search",notes:"Returns up to 50 Google Maps business/profile candidates. Use maps_place_intel to hydrate selected businesses."},{key:"maps_place",label:"Maps business lookup",aliases:["maps_place_intel","google maps","maps place","place intel"],credits:i(t.maps_place),unit:"per business",notes:"Base lookup. Reviews billed separately per card at maps_review rate."},{key:"maps_review",label:"Maps review",aliases:["maps_reviews","google reviews","review cards","reviews"],credits:i(t.maps_review),unit:"per review card",notes:"Charged after extraction when includeReviews is true."},{key:"fb_search",label:"Facebook ad library search",aliases:["facebook_search","fb_search","fb ad search"],credits:i(t.fb_search),unit:"per search",notes:"Browser automation to search Facebook Ads Library by keyword."},{key:"google_ads_search",label:"Google Ads Transparency search",aliases:["google_ads_search","google ads search","ads transparency search"],credits:i(t.google_ads_search),unit:"per search",notes:"Browser automation to find advertisers in Google Ads Transparency Center by domain or name."},{key:"google_ads_intel",label:"Google Ads Transparency advertiser intel",aliases:["google_ads_page_intel","google ads intel","ads transparency intel"],credits:i(t.google_ads_intel),unit:"per call",notes:"Lists and hydrates an advertiser's creatives with image URLs and video references."},{key:"google_ads_transcribe",label:"Google ad video transcription",aliases:["google_ads_transcribe","google ad transcript"],credits:i(2e4),unit:"base + 2/min of video",notes:"Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},{key:"fb_transcribe",label:"Facebook video / ad transcription",aliases:["facebook_transcribe","facebook_video_transcribe","fb_transcribe","fb ad transcript"],credits:i(2e4),unit:"base + 2/min of video",notes:"Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},{key:"fb_reels_inventory",label:"Facebook Reel URL inventory",aliases:["facebook_reels_inventory","facebook reels","facebook reel urls","facebook profile reels"],credits:i(t.fb_reels_inventory),unit:"per profile scan",notes:"Collects canonical Reel URLs from a Facebook profile. Logged-out scans are capped at 60 URLs; saved browser profiles can request larger inventories."},{key:"tiktok_transcribe",label:"TikTok video transcription",aliases:["tiktok_video_transcribe","tiktok transcript"],credits:i(2e4),unit:"base + 2/min of video",notes:"Transcribes a public TikTok video from its playback media. Flat 200-credit base plus 2 credits per minute, reconciled after transcription."},{key:"instagram_profile",label:"Instagram profile content discovery",aliases:["instagram_profile_content","instagram profile","instagram content list","ig profile"],credits:i(t.instagram_profile),unit:"per profile scan",notes:"Browser extraction of public Instagram profile grid links. Complete history may require a logged-in profile."},{key:"instagram_media",label:"Instagram media download",aliases:["instagram_media_download","instagram reel download","instagram post download","ig media"],credits:i(t.instagram_media),unit:"per post or reel",notes:"Extracts post text, image URLs, and reel audio/video tracks, with local downloads when the server can write files."},{key:"instagram_transcribe",label:"Instagram media transcription",aliases:["instagram transcript","instagram reel transcript","ig transcribe"],credits:i(2e4),unit:"base + 2/min of video",notes:"Whisper transcription of selected Instagram audio/video media. Flat 200-credit base plus 2 credits per minute of video, reconciled to the real video length after transcription (e.g. a 10-minute video is about 220 credits)."},{key:"browser_minute",label:"Interactive browser session",aliases:["browser_open","browser agent","browser_agent","live browser","browse","browser control","interactive browser"],credits:i(t.browser_minute),unit:"per minute of use",notes:"Metered per minute of use for the whole time the browser session is open. Close the session to stop the meter; abandoned sessions are auto-closed after a short idle window."},{key:"reddit_thread",label:"Reddit thread base lookup",aliases:["reddit","reddit_thread","reddit comments","subreddit","reddit post"],credits:i(t.reddit_thread),unit:"per thread",notes:"Base lookup for the post itself. Comments billed separately per comment at reddit_comment rate. Refunded if the thread cannot be retrieved."},{key:"reddit_comment",label:"Reddit comment",aliases:["reddit_comment","reddit comments"],credits:i(t.reddit_comment),unit:"per comment",notes:"Charged per comment actually extracted, billed down automatically if the balance runs out mid-thread."},{key:"video_analysis",label:"Video breakdown (frame-by-frame + transcript)",aliases:["video_frame_analysis","video breakdown","video analysis","analyze video"],credits:i(t.video_analysis),unit:"per 120 frames (max 480)",notes:"Full multi-lens video breakdown: samples frames across a video (up to 30 minutes), analyzes each with vision AI, transcribes the audio, then produces summary, pacing/energy, words-per-minute, topic outline, key points, hook analysis, visual style, and a how-to-replicate recipe with a quality-control pass. $1 per 120 frames requested (max 480 = $4); billed down automatically if the video cannot use the requested frames, and refunded fully if the run fails."},{key:"trustpilot_reviews",label:"Trustpilot review harvest",aliases:["trustpilot_reviews","trustpilot","trustpilot reviews"],credits:i(t.trustpilot_reviews),unit:"per call",notes:"Base lookup. Reviews billed separately per card at trustpilot_review rate. Refunded if no reviews are found."},{key:"trustpilot_review",label:"Trustpilot review card",aliases:["trustpilot_review","trustpilot card"],credits:i(t.trustpilot_review),unit:"per review card",notes:"Charged per review actually extracted, billed down automatically if the balance runs out mid-page."},{key:"g2_reviews",label:"G2 review harvest",aliases:["g2_reviews","g2","g2 reviews"],credits:i(t.g2_reviews),unit:"per call",notes:"Base lookup. Reviews billed separately per card at g2_review rate. Refunded if no reviews are found."},{key:"g2_review",label:"G2 review card",aliases:["g2_review","g2 card"],credits:i(t.g2_review),unit:"per review card",notes:"Charged per review actually extracted (each card carries up to 3 Q&A sections), billed down automatically if the balance runs out mid-page."},{key:"diff_page",label:"Page change check",aliases:["diff_page","page diff","change detection"],credits:i(t.diff_page),unit:"per check",notes:"Same cost as a single page extract \u2014 one scrape is performed per check, then compared against your last stored snapshot for that URL."}],Ee="price_1Ta1NRS8aAcsk3TGwsRnYbix";function k(e){return e.billing_interval??"month"}function O(e){return e.billed_usd??e.monthly_usd}function M(e){return e.credits_mc/a}var d={price_1TmiHRS8aAcsk3TGwmSNfNIa:{tier:"starter",label:"Starter",price_id:"price_1TmiHRS8aAcsk3TGwmSNfNIa",monthly_usd:12,credits_mc:8e6,concurrency:3,intro_coupon:"mcp-starter-1dollar-intro-12"},price_1TrgihS8aAcsk3TG5cglrq4D:{tier:"growth",label:"Growth",price_id:"price_1TrgihS8aAcsk3TG5cglrq4D",monthly_usd:40,credits_mc:26666700,concurrency:10,intro_coupon:null},price_1TrgihS8aAcsk3TG4HnG4gbY:{tier:"scale",label:"Scale",price_id:"price_1TrgihS8aAcsk3TG4HnG4gbY",monthly_usd:100,credits_mc:8e7,concurrency:20,intro_coupon:null},price_1TwqIVS8aAcsk3TG0FE3ddDM:{tier:"sonic",label:"Sonic Annual",price_id:"price_1TwqIVS8aAcsk3TG0FE3ddDM",monthly_usd:25,billed_usd:300,credits_mc:25e7,concurrency:3,intro_coupon:null,billing_interval:"year",credits_never_expire:!0,private_offer:!0},price_1TwqIWS8aAcsk3TGbGbtlLak:{tier:"sonic-memory",label:"Sonic Annual + Memory Pro",price_id:"price_1TwqIWS8aAcsk3TGbGbtlLak",monthly_usd:34.5,billed_usd:414,credits_mc:25e7,concurrency:3,intro_coupon:null,billing_interval:"year",credits_never_expire:!0,includes_memory:!0,private_offer:!0}},g=Object.fromEntries(Object.values(d).map(e=>[e.tier,e])),f={sonic:{slug:"sonic",label:"Sonic annual partner offer",headline:"A full year of MCP Scraper, with Credits that never expire.",tier_keys:["sonic","sonic-memory"]}};function Q(e){let r=Object.values(f).find(o=>o.tier_keys.includes(e));return r?`/${r.slug}/claim`:null}function Z(e){return e?Object.values(f).find(r=>r.tier_keys.includes(e))??null:null}function Ce(e,r){let o=Z(e);return o?!o.tier_keys.includes(r):!1}function he(e){if(!e)return null;let r=g[e];return r?{tier:r.tier,label:r.label,amount_usd:O(r),interval:k(r),monthly_equivalent_usd:r.monthly_usd,credits_per_interval:M(r),concurrency:r.concurrency,credits_never_expire:!!r.credits_never_expire,includes_memory:!!r.includes_memory,private_offer:!!r.private_offer,offer_claim_path:Q(r.tier)}:null}function Te(e){let r=f[e];if(!r)return null;let o=r.tier_keys.map(n=>g[n]).filter(n=>!!n).map(n=>({tier:n.tier,label:n.label,amount_usd:O(n),interval:k(n),monthly_equivalent_usd:n.monthly_usd,credits:M(n),credits_never_expire:!!n.credits_never_expire,includes_memory:!!n.includes_memory,concurrency:n.concurrency}));return o.length?{slug:r.slug,label:r.label,headline:r.headline,options:o}:null}function ye(){let e=g.starter,r=$.map(({aliases:s,...c})=>({...c,...c.notes?{notes:c.notes.replace(/ via fal\.ai/g,"")}:{}})),o=Object.fromEntries(r.map(s=>[s.key,{credits:s.credits,unit:s.unit,label:s.label,...s.notes?{notes:s.notes}:{}}])),n=Object.values(d).filter(s=>!s.private_offer).map(s=>({tier:s.tier,label:s.label,monthlyUsd:s.monthly_usd,credits:s.credits_mc/a,concurrency:s.concurrency}));return{...p,operationCosts:o,publicTiers:n,concurrencyPack:{ratePolicyVersion:L,monthlyUsd:x,currency:P,interval:U,billingUnit:"pack",slotsPerPack:b},scaleCreditPack:{ratePolicyVersion:j,eligibleTier:"scale",incrementUsd:D,creditsPerIncrement:R,maxQuantity:z,currency:"usd",expiryMonths:3},creditsPerDollarStarter:Math.round(i(e.credits_mc)/e.monthly_usd),costs:r,scheduling:p.addons.scheduling,connectedAccounts:p.addons.connectedAccounts,listingWrites:Y}}var x=5,P="usd",U="month",b=2,L="2026-08-14.concurrency-pack-v1",X="$5/month for 2 extra browsers",ve="MCP Scraper browser concurrency pack \u2014 each $5 monthly pack adds 2 browsers.",Ae="Billed monthly until canceled. A receipt is emailed after each successful payment.",h="npx -y -p mcp-scraper@latest mcp-scraper-cli billing concurrency checkout",j="2026-09-21.scale-credit-pack-v1",D=10,R=8e4,Se=R*a,z=100;function Ne(){let e=process.env.CONCURRENCY_BILLING_URL??"https://mcpscraper.dev/billing";return{product:"Extra browser concurrency pack",rate_policy_version:L,price_label:X,unit_amount_usd:x,currency:P,interval:U,billing_unit:"pack",pack_quantity:1,slots_per_pack:b,extra_slots:b,billing_url:e,terminal_command:h,terminal_command_with_api_key_env:`MCP_SCRAPER_API_KEY=sk_live_your_key ${h}`}}function i(e){return e/a}var J="npx -y -p mcp-scraper@latest mcp-scraper-cli billing subscribe";function ee(){return Object.values(d).filter(e=>!e.private_offer).sort((e,r)=>e.credits_mc-r.credits_mc)}function Ie(e,r){let o=process.env.TOPUP_URL??"https://mcpscraper.dev/billing",n=i(e),s=i(r),c=ee().map(u=>u.tier).join("|"),l=`${J} <${c}>`;return{error:"insufficient_balance",error_code:"insufficient_balance",message:`Insufficient credits. Balance: ${n} credits. This call requires ${s} credits. Credits refresh when your plan renews. Scale customers can buy ${R} Credits per $${D} increment; other customers can move to a larger plan at ${o} (terminal: ${l})`,balance_credits:n,required_credits:s,topup_url:o,upgrade_command:l}}var re=1e7*1,we={price_1TnMSTS8aAcsk3TGBgiwuvqL:{plan:"pro",label:"Pro",price_id:"price_1TnMSTS8aAcsk3TGBgiwuvqL",interval:"month",monthly_usd:19,quota_bytes:5e9},price_1TnMSTS8aAcsk3TGWBVU2agY:{plan:"pro",label:"Pro",price_id:"price_1TnMSTS8aAcsk3TGWBVU2agY",interval:"year",monthly_usd:19,quota_bytes:5e9}},ke={free:re,pro:5e9,team:5e10},te=3;var Oe=19/te,Me={TOPUP:"topup",SUBSCRIPTION:"subscription",SIGNUP_GRANT:"signup_grant",MONTHLY_REFRESH:"monthly_free_refresh",PAA:"paa",PAA_REFUND:"paa_refund",SERP:"serp",SERP_REFUND:"serp_refund",REFUND:"refund",TRANSCRIPTION:"transcription",TRANSCRIPTION_HOLD:"transcription_hold",TRANSCRIPTION_REFUND:"transcription_refund",YT_CHANNEL:"yt_channel",FB_AD:"fb_ad",MAPS_SEARCH:"maps_search",DIRECTORY_WORKFLOW_HOLD:"directory_workflow_hold",DIRECTORY_WORKFLOW_REFUND:"directory_workflow_refund",LEAD_LIST_ENRICHMENT_HOLD:"lead_list_enrichment_hold",LEAD_LIST_ENRICHMENT_REFUND:"lead_list_enrichment_refund",MAPS_PLACE:"maps_place",MAPS_REVIEW:"maps_review",MAPS_REVIEW_REFUND:"maps_review_refund",EXTRACT_SITE:"extract_site",EXTRACT_SITE_REFUND:"extract_site_refund",EXTRACT_URL:"page_scrape",URL_MAP:"url_map",EXTRACT_SITE_HOLD:"extract_site_hold",YT_CHANNEL_REFUND:"yt_channel_refund",FB_AD_REFUND:"fb_ad_refund",URL_MAP_REFUND:"url_map_refund",FB_SEARCH:"fb_search",FB_REELS_INVENTORY:"fb_reels_inventory",FB_REELS_INVENTORY_REFUND:"fb_reels_inventory_refund",FB_TRANSCRIBE:"fb_transcribe",TIKTOK_TRANSCRIBE:"tiktok_transcribe",FB_SEARCH_REFUND:"fb_search_refund",FB_SEARCH_RESULT:"fb_search_result",FB_SEARCH_RESULT_REFUND:"fb_search_result_refund",FB_PAGE_INTEL:"fb_page_intel",FB_PAGE_INTEL_REFUND:"fb_page_intel_refund",FB_PAGE_INTEL_AD:"fb_page_intel_ad",FB_PAGE_INTEL_AD_REFUND:"fb_page_intel_ad_refund",FB_TRANSCRIBE_REFUND:"fb_transcribe_refund",TIKTOK_TRANSCRIBE_REFUND:"tiktok_transcribe_refund",GOOGLE_ADS_SEARCH:"google_ads_search",GOOGLE_ADS_SEARCH_REFUND:"google_ads_search_refund",GOOGLE_ADS_INTEL:"google_ads_intel",GOOGLE_ADS_INTEL_REFUND:"google_ads_intel_refund",GOOGLE_ADS_TRANSCRIBE:"google_ads_transcribe",GOOGLE_ADS_TRANSCRIBE_REFUND:"google_ads_transcribe_refund",INSTAGRAM_PROFILE:"instagram_profile",INSTAGRAM_PROFILE_REFUND:"instagram_profile_refund",INSTAGRAM_MEDIA:"instagram_media",INSTAGRAM_MEDIA_REFUND:"instagram_media_refund",INSTAGRAM_TRANSCRIBE:"instagram_transcribe",INSTAGRAM_TRANSCRIBE_REFUND:"instagram_transcribe_refund",BROWSER_SESSION:"browser_session",REDDIT_THREAD:"reddit_thread",REDDIT_THREAD_REFUND:"reddit_thread_refund",REDDIT_COMMENT:"reddit_comment",REDDIT_COMMENT_REFUND:"reddit_comment_refund",VIDEO_ANALYSIS:"video_analysis",VIDEO_ANALYSIS_REFUND:"video_analysis_refund",VIDEO_ANALYSIS_ADJUST:"video_analysis_adjust",VIDEO_ANALYSIS_LLM:"video_analysis_llm",MEMORY_AI:"memory_ai",MEMORY_AI_REFUND:"memory_ai_refund",TRUSTPILOT_REVIEWS:"trustpilot_reviews",TRUSTPILOT_REVIEWS_REFUND:"trustpilot_reviews_refund",TRUSTPILOT_REVIEW:"trustpilot_review",TRUSTPILOT_REVIEW_REFUND:"trustpilot_review_refund",G2_REVIEWS:"g2_reviews",G2_REVIEWS_REFUND:"g2_reviews_refund",G2_REVIEW:"g2_review",G2_REVIEW_REFUND:"g2_review_refund",DIFF_PAGE:"diff_page",DIFF_PAGE_REFUND:"diff_page_refund",LOCAL_SOURCEBOOK_CREATE:"local_sourcebook_create",LOCAL_SOURCEBOOK_CREATE_REFUND:"local_sourcebook_create_refund",LOCAL_SOURCEBOOK_EDIT:"local_sourcebook_edit",LOCAL_SOURCEBOOK_EDIT_REFUND:"local_sourcebook_edit_refund",COMMONS_CREATE:"commons_create",COMMONS_CREATE_REFUND:"commons_create_refund",COMMONS_EDIT:"commons_edit",COMMONS_EDIT_REFUND:"commons_edit_refund",SITE_AUDIT_LLM:"site_audit_llm",SCHEDULED_RUN:"scheduled_run",CONNECTED_USAGE:"connected_usage",CONNECTED_ACCOUNT_FEE:"connected_account_fee",ADMIN_GRANT:"admin_grant",ADMIN_REVOKE:"admin_revoke"},xe=3,Pe=1.5,Ue=1.5,B=2e4,F=200;function Le(e){let r=Math.max(1,Math.ceil((e||60)/60));return B+F*r}var De=B+F*30,Be=d.price_1TmiHRS8aAcsk3TGwmSNfNIa.credits_mc/d.price_1TmiHRS8aAcsk3TGwmSNfNIa.monthly_usd;export{t as a,ne as b,ae as c,oe as d,a as e,ce as f,le as g,_e as h,de as i,Y as j,C as k,H as l,ue as m,K as n,pe as o,me as p,W as q,y as r,be as s,v as t,A as u,S as v,N as w,ge as x,fe as y,Re as z,$ as A,Ee as B,k as C,O as D,M as E,d as F,g as G,Q as H,Z as I,Ce as J,he as K,Te as L,ye as M,x as N,b as O,ve as P,Ae as Q,D as R,R as S,Se as T,z as U,Ne as V,i as W,Ie as X,we as Y,ke as Z,Me as _,xe as $,Pe as aa,Ue as ba,Le as ca,De as da,Be as ea};
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{f as F,h as V,i as re,j as ae,l as $}from"./chunk-
|
|
1
|
+
import{f as F,h as V,i as re,j as ae,l as $}from"./chunk-QTDLZTQ7.js";import{_ as R,e as U,q as _,t as G,u as ne,v as oe,w as M,y as ie}from"./chunk-72YO6PKN.js";import{h as L}from"./chunk-2SP57VCG.js";import{b as K}from"./chunk-6DTXIZY2.js";import{R as te,Sb as D,Xb as x,t as C}from"./chunk-WJ4XFLS4.js";import{createHash as Te,randomUUID as Ne}from"crypto";import{z as f}from"zod";async function at(e,t){!Number.isFinite(t)||t<=0||await K({op:R.CONNECTED_ACCOUNT_FEE,userId:Number(e)},()=>L({vendor:"nango_connection",units:t,unitType:"connection_month"}))}async function se(e,t){let n=[{vendor:"nango_function_run",units:t.functionRuns,unitType:"function_run"},{vendor:"nango_proxy_request",units:t.proxyRequests,unitType:"proxy_request"},{vendor:"nango_compute",units:t.computeMilliseconds/1e3,unitType:"compute_second"}];await K({op:R.CONNECTED_USAGE,userId:Number(e)},async()=>{for(let o of n)!Number.isFinite(o.units)||o.units<=0||await L(o)})}var E="connected_usage",Re="mastra_scheduler",Ie=f.object({toolName:f.string().trim().min(1).max(200).optional(),operationKind:f.enum(["read","action","sync","export","proxy","other"]).optional(),outcome:f.enum(["success","error","partial"]).optional(),requestId:f.string().trim().min(1).max(200).optional(),occurrenceId:f.string().trim().min(1).max(200).optional(),scheduleId:f.string().trim().min(1).max(200).optional(),nangoOperationId:f.string().trim().min(1).max(200).optional(),startedAt:f.string().datetime({offset:!0}).optional(),completedAt:f.string().datetime({offset:!0}).optional()}).strict(),Ae=f.object({identity:f.string().trim().min(1).max(320),idempotencyKey:f.string().trim().min(1).max(500),ratePolicyVersion:f.string().trim().min(1).max(100),connectionId:f.string().trim().min(1).max(500),providerConfigKey:f.string().trim().min(1).max(200),functionRuns:f.number().int().safe().nonnegative(),proxyRequests:f.number().int().safe().nonnegative(),computeMilliseconds:f.number().int().safe().nonnegative(),sourceSurface:f.enum(["main_mcp","mastra_scheduler","nango_reconciliation"]).default(Re),metadata:Ie.optional()}).strict(),Pe=f.object({identity:f.string().trim().min(1).max(320),ratePolicyVersion:f.string().trim().min(1).max(100)}).strict(),h=class extends Error{constructor(n,o,a,i={}){super(o);this.code=n;this.httpStatus=a;this.details=i;this.name="ConnectedUsageBillingError"}code;httpStatus;details};function z(e){return Te("sha256").update(e).digest("hex")}function xe(e){return`connected-usage:${z(e)}`}function Ue(e){return`connected-usage-debit:${z(e)}`}function Me(e){return e.trim().toLowerCase()}async function ce(e){let t=await te(Me(e));if(!t)throw new h("user_not_found","billing identity was not found",404);return t}function Oe(e){let t={toolName:e.metadata.toolName??null,operationKind:e.metadata.operationKind??null,outcome:e.metadata.outcome??null,requestId:e.metadata.requestId??null,occurrenceId:e.metadata.occurrenceId??null,scheduleId:e.metadata.scheduleId??null,nangoOperationId:e.metadata.nangoOperationId??null,startedAt:e.metadata.startedAt??null,completedAt:e.metadata.completedAt??null};return z(JSON.stringify({ratePolicyVersion:_,connectionId:e.connectionId,providerConfigKey:e.providerConfigKey,usage:e.usage,sourceSurface:e.sourceSurface,metadata:t}))}function A(e){let t=JSON.stringify(e);if(t.length>4e3)throw new h("invalid_request","connected usage metadata is too large",400);return t}function W(e){try{let t=JSON.parse(e.metadata??"");if(t.contract!=="connected_usage_receipt"||t.ratePolicyVersion!==_||typeof t.fingerprint!="string"||typeof t.connectionId!="string"||typeof t.providerConfigKey!="string"||!t.usage||!t.charge)throw new Error("invalid connected usage receipt metadata");return t}catch{throw new h("idempotency_key_conflict","the idempotency key is already bound to an incompatible billing receipt",409)}}async function H(e){let t=await C().execute({sql:"SELECT id, user_id, billing_class, source_surface, status, amount_mc, metadata FROM billing_events WHERE idempotency_key = ? LIMIT 1",args:[e]});return t.rows[0]?t.rows[0]:null}async function de(e,t){return(await C().execute({sql:`SELECT 1
|
|
2
2
|
FROM billing_events
|
|
3
3
|
WHERE user_id = ?
|
|
4
4
|
AND billing_class = ?
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
var o="0.91.0";export{o as a};
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{a as U}from"./chunk-XG6GCEUE.js";import{b as N,c as P}from"./chunk-HE45FFBU.js";import{h as L,t as E}from"./chunk-
|
|
1
|
+
import{a as U}from"./chunk-XG6GCEUE.js";import{b as N,c as P}from"./chunk-HE45FFBU.js";import{h as L,t as E}from"./chunk-WJ4XFLS4.js";import{createWriteStream as Z,mkdirSync as ee,rmSync as te}from"fs";import{homedir as re}from"os";import{join as k,extname as B,basename as F}from"path";import{pipeline as ne}from"stream/promises";import{Readable as oe,Transform as ie}from"stream";var se=["doubleclick.net","googlesyndication.com","googletagmanager.com","google-analytics.com","googletagservices.com","adservice.google","googletag.","pagead2.googlesyndication","facebook.net/tr","connect.facebook.net","fbcdn.net/rsrc","analytics.twitter.com","static.ads-twitter.com","ads.twitter.com","t.co/i/adsct","pixel.advertising.com","hotjar.com","clarity.ms","quantserve.com","scorecardresearch.com","newrelic.com","nr-data.net","segment.io","segment.com","amplitude.com","mixpanel.com","heap.io","fullstory.com","moatads.com","criteo.com","adsrvr.org","rubiconproject.com","pubmatic.com","openx.net","appnexus.com","amazon-adsystem.com","media.net","yieldmo.com","triplelift.com","sharethrough.com","prebid.","smaato.net","indexworm.com","casalemedia.com","outbrain.com","taboola.com","revcontent.com","mgid.com","tawk.to","intercom.io","drift.com","hs-scripts.com","zopim.com","livechatinc.com","userlike.com","onetrust.com","cookielaw.org","cookieinformation.com","trustarc.com","/ads/","/ad/","/banner/","/banners/","/pixel/","/beacon/","/tracking/","/tracker/","/remarketing/","/conversion/","1x1.gif","spacer.gif","blank.gif","transparent.gif"],ae=new Set([".jpg",".jpeg",".png",".webp",".gif",".avif",".svg",".tiff"]),de=new Set([".mp4",".webm",".mov",".avi",".m4v",".ogv",".mkv"]),ce=new Set([".mp3",".wav",".ogg",".aac",".m4a",".flac",".opus"]);function le(e){let t=e.toLowerCase();return se.some(r=>t.includes(r))}function ue(e){return e.startsWith("data:")}function q(e){try{let t=B(new URL(e).pathname).toLowerCase();if(ae.has(t))return"image";if(de.has(t))return"video";if(ce.has(t))return"audio"}catch{}return null}function H(e){let t=e.toLowerCase();return t.startsWith("image/")?"image":t.startsWith("video/")?"video":t.startsWith("audio/")?"audio":null}function me(e,t){if(!e||ue(e))return null;try{return new URL(e,t).href}catch{return null}}function O(e,t){try{let r=new URL(e);return F(r.pathname).replace(/[^a-zA-Z0-9._-]/g,"_").slice(0,80)||`asset-${t}`}catch{return`asset-${t}`}}function I(e,t){let r=e?.replace(/&(?:amp|#0?38|#x26);/gi,"&").replace(/\s+/g," ").trim();return r?r.slice(0,t):null}function ge(e){let t=e.match(/[-_/](\d{2,5})x(\d{2,5})(?=[._/?#-]|$)/i);return t?{width:Number(t[1]),height:Number(t[2])}:null}function pe(e){try{let t=new URL(e);t.hash="",t.pathname=t.pathname.replace(/-(?:\d{2,5}x\d{2,5}|scaled|e\d{6,})(?=\.[a-z0-9]{2,6}$)/i,"");for(let r of["w","width","h","height","resize"])t.searchParams.delete(r);return t.searchParams.sort(),t.href}catch{return e}}function S(e){let t=(e.width??0)*(e.height??0),r=e.discoveryMethods.some(g=>g.includes("lightbox"))?1e12:0,n=/-(?:\d{2,5}x\d{2,5}|scaled|e\d{6,})(?=\.[a-z0-9]{2,6}(?:[?#]|$))/i.test(e.url)?0:1e11;return r+n+t}function R(e,t){let r=e.get(t.url);if(!r){e.set(t.url,t);return}r.discoveryMethods=[...new Set([...r.discoveryMethods,...t.discoveryMethods])],r.altTexts=[...new Set([...r.altTexts,...t.altTexts])],r.contexts=[...new Set([...r.contexts,...t.contexts])],!r.type&&t.type&&(r.type=t.type),(t.width??0)*(t.height??0)>(r.width??0)*(r.height??0)&&(r.width=t.width,r.height=t.height)}function he(e,t){let r=new Map,n=(d,i,o=null,h,l,u)=>{let s=d?me(d.trim().replace(/&/g,"&"),t):null;if(!s||le(s))return;let m=u??ge(s);R(r,{url:s,type:o??q(s),discoveryMethods:[i],altTexts:h?[h]:[],contexts:l?[l]:[],width:m?.width??null,height:m?.height??null})},g=(d,i,o,h)=>{for(let l of(d??"").split(",")){let[u,s]=l.trim().split(/\s+/);if(!u)continue;let m=s?.match(/^(\d+)w$/)?.[1];n(u,i,o,h,null,m?{width:Number(m),height:0}:null)}},a=(d,i="")=>{let o=d.$;for(let l of d.images){let u=l.attributes,s=I(l.alt,500),m=l.width&&l.height?{width:l.width,height:l.height}:null;for(let p of["src","data-src","data-lazy-src","data-original","data-bg","data-background","data-bg-url","data-lazy-bg","data-echo"]){let b=u[p];n(b,`static-${p}${i}`,"image",s,null,m)}g(u.srcset,`static-srcset${i}`,"image",s),g(u["data-srcset"],`static-data-srcset${i}`,"image",s)}o("source").each((l,u)=>{let s=o(u),m=s.attr("type")??"",p=m.startsWith("video/")?"video":m.startsWith("audio/")?"audio":"image";n(s.attr("src"),`static-source${i}`,p),g(s.attr("srcset"),`static-source-srcset${i}`,p)}),o("video,audio").each((l,u)=>{let s=o(u),m=u.tagName.toLowerCase();n(s.attr("src"),`static-${m}${i}`,m),m==="video"&&n(s.attr("poster"),`static-video-poster${i}`,"image")}),o("meta").each((l,u)=>{let s=o(u),m=(s.attr("property")??s.attr("name")??"").toLowerCase();(m==="og:image"||m==="og:image:url"||m==="twitter:image"||m==="twitter:image:src")&&n(s.attr("content"),`static-${m}${i}`,"image")}),o("link[href]").each((l,u)=>{let s=o(u);(s.attr("rel")??"").toLowerCase().split(/\s+/).some(p=>p==="icon"||p==="apple-touch-icon")&&n(s.attr("href"),`static-icon${i}`,"image")}),o("svg image").each((l,u)=>{let s=o(u);n(s.attr("href")??s.attr("xlink:href"),`static-svg-image${i}`,"image")}),o("a[href]").has("img").each((l,u)=>{let s=o(u),m=s.attr("href");if(m&&(/lightbox|gallery|fancybox|glightbox|swipebox|elementor-open-lightbox/i.test(`${s.attr("class")??""} ${s.attr("data-elementor-open-lightbox")??""}`)||/\.(?:avif|gif|jpe?g|png|svg|webp)(?:[?#]|$)/i.test(m))){let p=I(s.find("img").first().attr("alt"),500);n(m,`static-lightbox-target${i}`,"image",p)}});let h=o("style").toArray().map(l=>o(l).text());o("[style]").each((l,u)=>{h.push(o(u).attr("style")??"")});for(let l of h.join(`
|
|
2
2
|
`).matchAll(/(?:background(?:-image)?\s*:\s*)?url\(\s*(?:["']|"|�?34;|'|�?39;)?([^"')&]+)(?:["']|"|�?34;|'|�?39;)?\s*\)/gi))n(l[1],`static-css-url${i}`,"image");for(let l of d.sourceHtml.matchAll(/["'](https?:\/\/[^"'\s<>]+\.(?:avif|gif|jpe?g|png|svg|webp|mp4|webm|mp3|ogg)(?:\?[^"'\s<>]*)?)["']/gi))n(l[1],`static-bare-url${i}`)},c=typeof e=="string"?P(e):e;return a(c),c.$("script").each((d,i)=>{let o=c.$(i).text(),h=o.replace(/\\u003[cC]/g,"<").replace(/\\u003[eE]/g,">").replace(/\\u0026/gi,"&").replace(/\\["']/g,l=>l.slice(1)).replace(/\\\//g,"/");h!==o&&/<(?:img|source|video|audio)\b/i.test(h)&&a(P(h),"-json-unescaped")}),[...r.values()]}function fe(e,t,r=100){let n=new Map;for(let a of e)R(n,a);for(let a of t?.assets??[])R(n,{...a,type:a.type});let g=new Map;for(let a of n.values()){let c=pe(a.url),d=g.get(c)??[];d.push(a),g.set(c,d)}return[...g.values()].map(a=>{let c=[...a].sort((i,o)=>S(o)-S(i)||i.url.localeCompare(o.url)),d={...c[0]};return d.discoveryMethods=[...new Set(a.flatMap(i=>i.discoveryMethods))],d.altTexts=[...new Set(a.flatMap(i=>i.altTexts))],d.contexts=[...new Set(a.flatMap(i=>i.contexts))],{...d,variants:c.slice(1).map(i=>i.url)}}).sort((a,c)=>S(c)-S(a)||a.url.localeCompare(c.url)).slice(0,Math.max(1,r))}async function be(e,t,r,n={}){let g=Math.max(1,Math.min(n.maxBytes??52428800,104857600)),a=e,c=null;for(let p=0;p<=5;p++){let b=await N(a,{field:"media URL"});if(b.error||!b.parsed)throw new Error(b.error??"Media URL was rejected");if(c=await fetch(b.parsed.href,{signal:AbortSignal.timeout(15e3),redirect:"manual"}),c.status>=300&&c.status<400){let w=c.headers.get("location");if(!w)throw new Error(`HTTP ${c.status} redirect did not include Location`);a=new URL(w,b.parsed.href).href,c=null;continue}break}if(!c)throw new Error("Media download exceeded five redirects");if(!c.ok)throw new Error(`HTTP ${c.status}`);if(!c.body)throw new Error("Empty response body");let d=c.headers.get("content-type")?.split(";")[0].trim()??null;if(n.expectedType&&(!d||H(d)!==n.expectedType))throw new Error(`Expected ${n.expectedType} content but received ${d??"no content-type"}`);let i=Number(c.headers.get("content-length")??0);if(Number.isFinite(i)&&i>g)throw new Error(`Media asset exceeds ${g} byte limit`);let o=k(t,r);if(d&&!B(r)){let b={"image/jpeg":".jpg","image/png":".png","image/webp":".webp","image/gif":".gif","image/svg+xml":".svg","image/avif":".avif","video/mp4":".mp4","video/webm":".webm","audio/mpeg":".mp3","audio/ogg":".ogg","audio/wav":".wav"}[d];b&&(o=o+b)}let h=Z(o);await new Promise((p,b)=>{h.once("open",()=>p()),h.once("error",b)});let l=0,u=new ie({transform(p,b,w){let _=p.length;if(l+_>g){w(new Error(`Media asset exceeds ${g} byte limit`));return}if(n.consumeBytes&&!n.consumeBytes(_)){w(new Error("Media export total byte limit exceeded"));return}l+=_,w(null,p)}});try{await ne(oe.fromWeb(c.body),u,h)}catch(p){throw h.destroy(),te(o,{force:!0}),p}let{statSync:s}=await import("fs"),m=s(o).size;return{savedPath:o,sizeBytes:m,mimeType:d}}async function Oe(e,t,r={}){let n=r.types??["image","video","audio"],g=new Set(n),a=he(e,t),c=Math.max(1,Math.min(r.maxAssets??100,250)),d=fe(a,r.rendered,2e3),i=d.slice(0,c),o=new Set([...a.map(f=>f.url),...(r.rendered?.assets??[]).map(f=>f.url)]).size,h=i.flatMap(f=>{let y=f.type??q(f.url);return y&&g.has(y)?[{...f,type:y}]:[]}),l=o-h.length,u=d.length>c,s=r.rendered?.exhausted===!0&&!u,m=u?"asset_limit":r.rendered?.stopReason??"render_unavailable",p=u?[`Media discovery found ${d.length} responsive families; only the requested ${c} records were retained.`]:r.rendered?r.rendered.exhausted?[]:[`Rendered media discovery stopped with ${r.rendered.stopReason}; the inventory may be incomplete.`]:["Rendered media discovery was unavailable; this manifest contains static-source evidence only."],b={pageUrl:t,staticFound:a.length,renderedFound:r.rendered?.assets.length??0,filteredCount:l,totalFound:o,completeness:s?"complete":"partial",exhausted:s,stopReason:m,scrollRounds:r.rendered?.scrollRounds??0,warnings:p,artifact:null};if(r.outputDir===null)return{...b,outputDir:null,assets:h.map((f,y)=>({...f,mimeType:null,filename:O(f.url,y),savedPath:null,sizeBytes:null,sha256:null,downloadStatus:"not_attempted",downloadError:null,inlinePreview:null}))};let w=(()=>{try{return new URL(t).hostname.replace(/^www\./,"")}catch{return"unknown"}})(),_=new Date().toISOString().replace(/[:.]/g,"-").slice(0,19),C=r.outputDir??k(re(),"Downloads","mcp-scraper","media",`${_}-${w}`);ee(C,{recursive:!0});let T=[];return await Promise.allSettled(h.map(async(f,y)=>{let $=O(f.url,y);try{let{savedPath:v,sizeBytes:Q,mimeType:D}=await be(f.url,C,$),G=D?H(D)??f.type:f.type;T.push({...f,type:G,mimeType:D,filename:F(v),savedPath:v,sizeBytes:Q,sha256:null,downloadStatus:"downloaded",downloadError:null,inlinePreview:null})}catch(v){T.push({...f,mimeType:null,filename:$,savedPath:null,sizeBytes:null,sha256:null,downloadStatus:"failed",downloadError:v instanceof Error?v.message:"media_download_failed",inlinePreview:null})}})),T.sort((f,y)=>f.savedPath&&!y.savedPath?-1:!f.savedPath&&y.savedPath?1:f.url.localeCompare(y.url)),{...b,outputDir:C,assets:T}}var ye=new Set(["browser_navigation_blocked","browser_navigation_failed","browser_timeout","browser_session_failed","browser_result_missing"]);function we(e){return e instanceof Error?e.code:void 0}function _e(e){return e instanceof Error?e.message:String(e)}function j(e){let t=e?.match(/^browser_http_(\d+)$/);return t?Number(t[1]):null}function ve(e){return e===404?"page_not_found":e===403?"page_forbidden":e===429?"page_rate_limited":e>=500&&e<=599?"page_server_error":"page_http_error"}function Ee(e){let t=_e(e),r=we(e);if(U(e,t))return{errorCode:"vendor_unavailable",httpStatus:null,retryable:!0};let n=j(r)??j(t.match(/^(browser_http_\d+)/)?.[1]);return n!=null?{errorCode:ve(n),httpStatus:n,retryable:n===429||n>=500}:r==="browser_challenge_unresolved"||/^browser_challenge_unresolved/.test(t)?{errorCode:"bot_check_unresolved",httpStatus:null,retryable:!0}:r==="browser_response_too_large"||/exceeds \d+ byte limit/i.test(t)?{errorCode:"page_too_large",httpStatus:null,retryable:!1}:/browser (?:has been )?closed|context closed|target page, context or browser has been closed|session closed/i.test(t)?{errorCode:"browser_session_interrupted",httpStatus:null,retryable:!0}:ye.has(r??"")||/timeout|navigation_failed|ERR_/i.test(t)?{errorCode:"page_unreachable",httpStatus:null,retryable:!0}:{errorCode:"extraction_failed",httpStatus:null,retryable:!1}}function z(e,t){switch(e){case"page_not_found":return`The page could not be found (HTTP ${t??404}). It may have been moved or deleted.`;case"page_forbidden":return`The site refused the request (HTTP ${t??403}).`;case"page_rate_limited":return"The site is rate-limiting requests right now (HTTP 429). Retrying after a short wait usually works.";case"page_server_error":return`The page's own server returned an error (HTTP ${t}). This is on the target site's side, not something a retry here can fix.`;case"page_http_error":return`The page returned an unexpected HTTP ${t} response.`;case"bot_check_unresolved":return"This site has automated-traffic protection (a bot/CAPTCHA check) that could not be resolved in time. Some sites are simply not extractable this way.";case"page_too_large":return"The page is larger than we can safely process.";case"page_unreachable":return"The page did not respond in time, or the connection could not be completed.";case"browser_session_interrupted":return"The browser session closed before extraction completed. A retry uses a fresh browser session.";case"vendor_unavailable":return"MCP Scraper's internal browser services require MCP Scraper team attention. Servers or IPs are down until this is fixed \u2014 this is not caused by your request. Please retry in a few minutes.";case"extraction_failed":return"The page could not be extracted. Please retry, or contact support if this persists for the same URL."}}function Fe(e,t={}){let r=z(e.errorCode,e.httpStatus);return{...L({errorCode:e.errorCode,retryable:e.retryable,chargeStatus:t.chargeStatus,details:{...t.details??{},...e.httpStatus!=null?{http_status:e.httpStatus}:{}}}),message:r,error:r}}function qe(e){return e.httpStatus===404?404:e.httpStatus===403?403:e.httpStatus===429?429:e.httpStatus!=null&&e.httpStatus>=500&&e.httpStatus<=599?500:e.httpStatus!=null?502:e.errorCode==="vendor_unavailable"?503:502}var xe="wayback_capture_missing";function He(e,t){if(e==null)return{failureCode:null,failureReason:null};if(e===xe)return{failureCode:e,failureReason:t??null};let r=new Error(t??e);r.code=e;let n=Ee(r);return{failureCode:n.errorCode,failureReason:z(n.errorCode,n.httpStatus)}}import{neon as Te}from"@neondatabase/serverless";import{waitUntil as Se}from"@vercel/functions";var Me="jina",M=null,X=!1;function Y(){return(process.env.JINA_EMBED_MODEL??"jina-embeddings-v5-omni-small").trim()}function J(){return Number((process.env.JINA_EMBED_DIM??"1024").trim())}function A(){return!!(process.env.JINA_API_KEY?.trim()&&process.env.MEMORY_DATABASE_URL?.trim())}function x(){if(M)return M;let e=process.env.MEMORY_DATABASE_URL?.trim();if(!e)throw new Error("MEMORY_DATABASE_URL is not set; Commons semantic search needs the shared Postgres.");return M=Te(e),M}async function K(){if(X)return;let e=J();await x().query("CREATE EXTENSION IF NOT EXISTS vector"),await x().query(`
|
|
3
3
|
CREATE TABLE IF NOT EXISTS commons_index_vectors (
|
|
4
4
|
document_id TEXT PRIMARY KEY,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
import{g as _,i as c}from"./chunk-
|
|
1
|
+
import{g as _,i as c}from"./chunk-WJ4XFLS4.js";var g={\u006b\u0065\u0072\u006e\u0065\u006c:"browserRuntime"},a=/error|message/i,w=(new RegExp("(?:spending|billing|organization|account).{0,80}(?:cap|limit|blocked|disabled)|(?:quota|capacity).{0,40}(?:reached|exceeded)|https?:\\/\\/\\S*(?:dashboard|on\u006b\u0065\u0072\u006e\u0065\u006c)", "i")),E=/\b(?:ENOENT|ENOTDIR|EACCES|EPERM|EISDIR|EMFILE|ENFILE)\b|\b(?:mkdir|open|stat|lstat|readFile|writeFile)\s+['"]?(?:\/|[A-Za-z]:\\)|(?:^|[\s'"])(?:\/Users|\/home|\/tmp|\/var|\/private|\/opt|\/srv|\/root)\//i,k=new Set(["\u006b\u0065\u0072\u006e\u0065\u006c_session_id","browser_session_id","\u006b\u0065\u0072\u006e\u0065\u006cSessionId","browserSessionId","sessionId","\u006b\u0065\u0072\u006e\u0065\u006c_delete_started","session_cleanup_started","\u006b\u0065\u0072\u006e\u0065\u006cDeleteStarted","sessionCleanupStarted","\u006b\u0065\u0072\u006e\u0065\u006c_delete_succeeded","session_cleanup_succeeded","\u006b\u0065\u0072\u006e\u0065\u006cDeleteSucceeded","sessionCleanupSucceeded","\u006b\u0065\u0072\u006e\u0065\u006c_delete_error","session_cleanup_error","\u006b\u0065\u0072\u006e\u0065\u006cDeleteError","sessionCleanupError","\u006b\u0065\u0072\u006e\u0065\u006cProxyId","proxyId","requestedProxyIdSuffix","retrievedProxyIdSuffix","provider","providerName","service","serviceName","vendor","vendorName","organization","organizationId","accountId","browser_provider","provider_session_id","provider_session_observed_at","provider_lookup_status","provider_status","provider_error_code","provider_error_message","provider_reason_source","provider_duration_ms","provider_bandwidth_bytes","provider_captcha","provider_checked_at","provider_detail","provider_detail_json","provider_disconnect_observed","provider_disconnect_message","browserProvider","providerSessionId","providerLookupStatus","providerStatus","providerErrorCode","providerErrorMessage","providerReasonSource","providerDurationMs","providerBandwidthBytes","providerCaptcha","providerCheckedAt","providerDetail","providerDisconnectObserved","providerDisconnectMessage"]),S=new Set(["hosted_url","hostedUrl","live_view_url","liveViewUrl","cdp_ws_url","cdpWsUrl","browser_live_view_url"]),f=(new RegExp("\\b(?:wss?|https?):\\/\\/[^\\s\"'<>]*\\bon\u006b\u0065\u0072\u006e\u0065\u006c\\.com[^\\s\"'<>]*", "gi"));function b(e){return e.replace(f,"[browser-service]")}var R=new Set(["\u0062\u0072\u0069\u0067\u0068\u0074_\u0064\u0061\u0074\u0061_browser_api","\u006b\u0065\u0072\u006e\u0065\u006c","local"]),m=new Set(["pending","complete","not_found","unavailable_credentials","temporary_error","invalid_payload","not_applicable"]),h=new Set(["active","running","complete","completed","failed","closed","stopped","terminated","timed_out"]),I=new Set(["provider_session_api","client_runtime","extractor","unknown"]),y=new Set(["browser_disconnected","worker_disconnect","job_killed","cdp_cmd_timeout","network_inactivity_timeout","session_timeout","client_timeout","navigate_domains_limit","cdp_error","browser_session_interrupted","provider_detail_pending","provider_lookup_unavailable","provider_reported_no_error","provider_session_not_found","provider_invalid_json","provider_invalid_payload","provider_lookup_timeout","provider_lookup_network_error","unknown"]),T=new Set(["browser_provider","provider_session_id","provider_lookup_status","provider_status","provider_error_code","provider_error_message","provider_reason_source","provider_duration_ms","provider_bandwidth_bytes","provider_captcha","provider_checked_at","provider_detail","provider_detail_json","provider_disconnect_observed","provider_disconnect_message"]),P=512,A=320;function t(e,r){return typeof e=="string"&&r.has(e)?e:null}function O(e){return typeof e!="string"?null:y.has(e)||/^provider_http_(?:429|5\d\d)$/.test(e)?e:null}function D(e){return typeof e!="string"||e.length<1||e.length>P||/^(?:authorization|bearer|password|secret|token|api[_-]?key|account)/i.test(e)?null:/^[A-Za-z0-9._:-]+$/.test(e)&&!e.includes("://")?e:null}function p(e){return typeof e=="number"&&Number.isSafeInteger(e)&&e>=0?e:null}function l(e){if(typeof e!="string"||e.length>40)return null;let r=Date.parse(e);return Number.isFinite(r)&&new Date(r).toISOString()===e?e:null}function u(e){if(typeof e!="string")return null;let r=e.replace(/[\u0000-\u001f\u007f]+/g," ").replace(/\s+/g," ").trim();if(!r||r.startsWith("{")||r.startsWith("["))return null;let o=r.replace(/\b(?:wss?|https?):\/\/\S+/gi,"[redacted endpoint]").replace(/\b(?:authorization|proxy-authorization)\s*:\s*(?:Bearer\s+)?\S+/gi,"[redacted credential]").replace(/\bBearer\s+\S+/gi,"[redacted credential]").replace(/\b(?:api[_-]?key|password|secret|token|account[_-]?id)\s*[=:]\s*[^\s,;]+/gi,"[redacted credential]").replace(/\b(?:account|organization)\s+(?:id\s+)?[A-Za-z0-9_-]{4,}/gi,"[redacted account]");return b(o).slice(0,A)}function v(e){return typeof e=="boolean"?e:null}function N(e){return{browser_provider:t(e.browser_provider,R),provider_session_id:D(e.provider_session_id),provider_session_observed_at:l(e.provider_session_observed_at),provider_lookup_status:t(e.provider_lookup_status,m),provider_status:t(e.provider_status,h),provider_error_code:O(e.provider_error_code),provider_error_message:u(e.provider_error_message),provider_reason_source:t(e.provider_reason_source,I),provider_duration_ms:p(e.provider_duration_ms),provider_bandwidth_bytes:p(e.provider_bandwidth_bytes),provider_captcha:v(e.provider_captcha),provider_checked_at:l(e.provider_checked_at),provider_disconnect_observed:v(e.provider_disconnect_observed),provider_disconnect_message:u(e.provider_disconnect_message)}}function i(e,r=""){if(typeof e=="string"){let o=b(e);return a.test(r)&&(w.test(e)||E.test(e))?o=_("service_unavailable"):a.test(r)&&(o=c(o)),o}if(Array.isArray(e))return e.map(o=>i(o,r));if(e!==null&&typeof e=="object"){let o={};for(let[n,d]of Object.entries(e)){if(k.has(n))continue;let s=g[n]??n;if(S.has(n)){o[s]=null;continue}o[s]=i(d,n)}return o}return e}function L(e){return e.map(r=>{if(r===null||typeof r!="object"||Array.isArray(r))return i(r);let o=r,n=i(o);return Object.keys(o).some(s=>T.has(s))?{...n,...N(o)}:n})}function C(e){let r=e?.diagnostics;return r?.debug?{...e,diagnostics:{...r,debug:i(r.debug)}}:e}export{i as a,L as b,C as c};
|