mcp-scraper 0.33.6 → 0.33.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +2 -2
  2. package/dist/bin/api-server.cjs +4088 -3097
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +3 -3
  5. package/dist/bin/mcp-scraper-cli.cjs +12 -2
  6. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  7. package/dist/bin/mcp-scraper-cli.js +12 -2
  8. package/dist/bin/mcp-scraper-cli.js.map +1 -1
  9. package/dist/bin/mcp-scraper-install.cjs +2 -2
  10. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  11. package/dist/bin/mcp-scraper-install.js +2 -2
  12. package/dist/bin/mcp-stdio-server.cjs +109 -6
  13. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  14. package/dist/bin/mcp-stdio-server.js +6 -6
  15. package/dist/bin/paa-harvest.cjs +106 -13
  16. package/dist/bin/paa-harvest.cjs.map +1 -1
  17. package/dist/bin/paa-harvest.js +3 -3
  18. package/dist/{chunk-DUS4ENOY.js → chunk-2EDFOQD7.js} +2 -2
  19. package/dist/{chunk-DUS4ENOY.js.map → chunk-2EDFOQD7.js.map} +1 -1
  20. package/dist/{chunk-C2OD4ELT.js → chunk-3HBPKR5G.js} +2 -2
  21. package/dist/{chunk-AIK7WQD5.js → chunk-62DQAWPF.js} +141 -3
  22. package/dist/chunk-62DQAWPF.js.map +1 -0
  23. package/dist/{chunk-6ZUK4H66.js → chunk-H6QECE2U.js} +109 -6
  24. package/dist/chunk-H6QECE2U.js.map +1 -0
  25. package/dist/chunk-HAMN42SP.js +7 -0
  26. package/dist/chunk-HAMN42SP.js.map +1 -0
  27. package/dist/{chunk-LKIP7VDN.js → chunk-JBSGBSGT.js} +46 -3
  28. package/dist/chunk-JBSGBSGT.js.map +1 -0
  29. package/dist/{chunk-XL6YIWJ3.js → chunk-SUOHUXQS.js} +3 -3
  30. package/dist/{chunk-WJJTGRK4.js → chunk-V36LS5YV.js} +2 -2
  31. package/dist/{chunk-A54ELKJK.js → chunk-XVCETVYY.js} +129 -15
  32. package/dist/chunk-XVCETVYY.js.map +1 -0
  33. package/dist/{db-5SZQUI7A.js → db-YAI5AQOI.js} +16 -2
  34. package/dist/{extract-bundle-YNXTFCC3.js → extract-bundle-GPZRLBVY.js} +2 -2
  35. package/dist/index.cjs +106 -13
  36. package/dist/index.cjs.map +1 -1
  37. package/dist/index.d.cts +1 -1
  38. package/dist/index.d.ts +1 -1
  39. package/dist/index.js +3 -3
  40. package/dist/{server-SHOYO52X.js → server-B35CQJB3.js} +1358 -772
  41. package/dist/server-B35CQJB3.js.map +1 -0
  42. package/dist/{site-extract-repository-ZDTO7ODZ.js → site-extract-repository-JHMVHENZ.js} +4 -4
  43. package/dist/{worker-BJ64GCDE.js → worker-3LYNVGLF.js} +6 -6
  44. package/docs/mcp-tool-manifest.generated.json +185 -3
  45. package/package.json +1 -1
  46. package/dist/chunk-6ZUK4H66.js.map +0 -1
  47. package/dist/chunk-A54ELKJK.js.map +0 -1
  48. package/dist/chunk-AIK7WQD5.js.map +0 -1
  49. package/dist/chunk-F65IIUHI.js +0 -7
  50. package/dist/chunk-F65IIUHI.js.map +0 -1
  51. package/dist/chunk-LKIP7VDN.js.map +0 -1
  52. package/dist/server-SHOYO52X.js.map +0 -1
  53. /package/dist/{chunk-C2OD4ELT.js.map → chunk-3HBPKR5G.js.map} +0 -0
  54. /package/dist/{chunk-XL6YIWJ3.js.map → chunk-SUOHUXQS.js.map} +0 -0
  55. /package/dist/{chunk-WJJTGRK4.js.map → chunk-V36LS5YV.js.map} +0 -0
  56. /package/dist/{db-5SZQUI7A.js.map → db-YAI5AQOI.js.map} +0 -0
  57. /package/dist/{extract-bundle-YNXTFCC3.js.map → extract-bundle-GPZRLBVY.js.map} +0 -0
  58. /package/dist/{site-extract-repository-ZDTO7ODZ.js.map → site-extract-repository-JHMVHENZ.js.map} +0 -0
  59. /package/dist/{worker-BJ64GCDE.js.map → worker-3LYNVGLF.js.map} +0 -0
@@ -278,6 +278,9 @@ var HttpMcpToolExecutor = class {
278
278
  redditThread(input) {
279
279
  return this.call("/reddit/thread", input, this.httpTimeoutOverrideMs ?? 24e4);
280
280
  }
281
+ redditTrending(input) {
282
+ return this.call("/reddit/trending", input, this.httpTimeoutOverrideMs ?? 3e5);
283
+ }
281
284
  videoFrameAnalysis(input) {
282
285
  return this.call("/video/analyze", input);
283
286
  }
@@ -564,7 +567,7 @@ render();
564
567
  }
565
568
 
566
569
  // src/version.ts
567
- var PACKAGE_VERSION = "0.33.6";
570
+ var PACKAGE_VERSION = "0.33.7";
568
571
 
569
572
  // src/mcp/browser-agent-tool-schemas.ts
570
573
  var import_zod = require("zod");
@@ -3107,8 +3110,8 @@ var SUBSCRIPTION_TIERS = {
3107
3110
  "price_1TmiHRS8aAcsk3TGwmSNfNIa": { tier: "starter", label: "Starter", price_id: "price_1TmiHRS8aAcsk3TGwmSNfNIa", monthly_usd: 12, credits_mc: 8e6, concurrency: 3, intro_coupon: "mcp-starter-1dollar-intro-12" },
3108
3111
  "price_1TrgihS8aAcsk3TG5cglrq4D": { tier: "growth", label: "Growth", price_id: "price_1TrgihS8aAcsk3TG5cglrq4D", monthly_usd: 40, credits_mc: 26666700, concurrency: 10, intro_coupon: null },
3109
3112
  "price_1TrgihS8aAcsk3TG4HnG4gbY": { tier: "scale", label: "Scale", price_id: "price_1TrgihS8aAcsk3TG4HnG4gbY", monthly_usd: 100, credits_mc: 8e7, concurrency: 20, intro_coupon: null },
3110
- "price_1TwAl2S8aAcsk3TGC8xvpM3S": { tier: "sonic", label: "Sonic Annual", price_id: "price_1TwAl2S8aAcsk3TGC8xvpM3S", monthly_usd: 25, credits_mc: 25e7, concurrency: 3, intro_coupon: null, billing_interval: "year", credits_never_expire: true, private_offer: true },
3111
- "price_1TwAl3S8aAcsk3TGnORwFd87": { tier: "sonic-memory", label: "Sonic Annual + Memory Pro", price_id: "price_1TwAl3S8aAcsk3TGnORwFd87", monthly_usd: 34.5, credits_mc: 25e7, concurrency: 3, intro_coupon: null, billing_interval: "year", credits_never_expire: true, includes_memory: true, private_offer: true }
3113
+ "price_1TwqIVS8aAcsk3TG0FE3ddDM": { tier: "sonic", label: "Sonic Annual", price_id: "price_1TwqIVS8aAcsk3TG0FE3ddDM", monthly_usd: 25, billed_usd: 300, credits_mc: 25e7, concurrency: 3, intro_coupon: null, billing_interval: "year", credits_never_expire: true, private_offer: true },
3114
+ "price_1TwqIWS8aAcsk3TGbGbtlLak": { tier: "sonic-memory", label: "Sonic Annual + Memory Pro", price_id: "price_1TwqIWS8aAcsk3TGbGbtlLak", monthly_usd: 34.5, billed_usd: 414, credits_mc: 25e7, concurrency: 3, intro_coupon: null, billing_interval: "year", credits_never_expire: true, includes_memory: true, private_offer: true }
3112
3115
  };
3113
3116
  var SUBSCRIPTION_TIER_BY_KEY = Object.fromEntries(
3114
3117
  Object.values(SUBSCRIPTION_TIERS).map((t) => [t.tier, t])
@@ -4300,6 +4303,63 @@ ${commentMd || "_No comments captured._"}`
4300
4303
  }
4301
4304
  };
4302
4305
  }
4306
+ function formatRedditTrending(raw, input) {
4307
+ const parsed = parseData(raw);
4308
+ if ("error" in parsed) return { content: [{ type: "text", text: parsed.error }], isError: true };
4309
+ const d = parsed.data;
4310
+ const threads = d.rankedThreads ?? [];
4311
+ const questions = (d.questions ?? []).filter((q) => q.question);
4312
+ const totals = {
4313
+ threads: d.totals?.threads ?? threads.length,
4314
+ upvotes: d.totals?.upvotes ?? 0,
4315
+ comments: d.totals?.comments ?? 0
4316
+ };
4317
+ const subreddit = d.subreddit ?? input.subreddit ?? null;
4318
+ const threadBlocks = threads.map((t, i) => [
4319
+ `### ${i + 1}. ${t.title || "Untitled"}`,
4320
+ `**${t.subreddit || "r/?"}** \xB7 ${t.score ?? 0} upvotes \xB7 ${t.commentCount ?? 0} comments \xB7 engagement ${t.engagementScore ?? 0}${t.ageText ? ` \xB7 ${t.ageText}` : ""}`,
4321
+ t.url ? `${t.url}` : "",
4322
+ t.topQuestions?.length ? `**Top questions:**
4323
+ ${t.topQuestions.map((q) => `- ${q}`).join("\n")}` : ""
4324
+ ].filter(Boolean).join("\n")).join("\n\n");
4325
+ const questionList = questions.map((q) => `- ${q.question}${q.threadUrl ? `
4326
+ ${q.threadUrl}` : ""}`).join("\n");
4327
+ const full = [
4328
+ `# Reddit Trending: "${d.topic || input.topic}"${subreddit ? ` in r/${subreddit}` : ""}`,
4329
+ `**${totals.threads} threads \xB7 ${totals.upvotes} upvotes \xB7 ${totals.comments} comments** \xB7 last 30 days \xB7 ${d.threadsScraped ?? 0} threads scraped`,
4330
+ `
4331
+ ## Ranked threads
4332
+ ${threadBlocks || "_No threads found._"}`,
4333
+ `
4334
+ ## Questions people asked
4335
+ ${questionList || "_No questions extracted._"}`,
4336
+ `
4337
+ ---
4338
+ \u{1F4A1} Deep-dive a winner's full comment tree: pass its url to \`reddit_thread\``
4339
+ ].join("\n");
4340
+ return {
4341
+ ...oneBlock(full),
4342
+ structuredContent: {
4343
+ topic: d.topic ?? input.topic,
4344
+ subreddit,
4345
+ window: d.window ?? "30d",
4346
+ totals,
4347
+ rankedThreads: threads.map((t) => ({
4348
+ title: t.title ?? "",
4349
+ url: t.url ?? "",
4350
+ subreddit: t.subreddit ?? "",
4351
+ score: Number(t.score ?? 0),
4352
+ commentCount: Number(t.commentCount ?? 0),
4353
+ engagementScore: Number(t.engagementScore ?? 0),
4354
+ ageText: t.ageText ?? "",
4355
+ topQuestions: t.topQuestions ?? []
4356
+ })),
4357
+ questions: questions.map((q) => ({ question: q.question ?? "", threadUrl: q.threadUrl ?? "" })),
4358
+ threadsScraped: Number(d.threadsScraped ?? 0),
4359
+ searchUrl: d.searchUrl ?? ""
4360
+ }
4361
+ };
4362
+ }
4303
4363
  function formatFacebookAdSearch(raw, input) {
4304
4364
  const parsed = parseData(raw);
4305
4365
  if ("error" in parsed) return { content: [{ type: "text", text: parsed.error }], isError: true };
@@ -5668,8 +5728,11 @@ seam is noted so you can chain them.
5668
5728
  url the user gives).
5669
5729
 
5670
5730
  ## Reddit
5671
- - A reddit.com thread/post URL -> **reddit_thread**. Returns the post plus its comment tree and handles
5672
- Reddit's bot wall itself (no login needed). Find threads first with \`search_serp\`.
5731
+ - Read ONE known reddit.com thread/post URL -> **reddit_thread**. Returns the post plus its comment tree and
5732
+ handles Reddit's bot wall itself (no login needed). Find threads first with \`search_serp\` or \`reddit_trending\`.
5733
+ - DISCOVER what a niche is talking about (topic, no known thread) -> **reddit_trending** (takes a topic, optional
5734
+ subreddit; returns the last 30 days' top threads ranked by engagement plus the questions people asked \u2014
5735
+ feed winning \`rankedThreads[].url\` values into \`reddit_thread\` for the full comment tree).
5673
5736
 
5674
5737
  ## Other sites & logins (browser agent)
5675
5738
  For an arbitrary site or a logged-in dashboard with no dedicated tool, use the browser_* agent. **First
@@ -6217,6 +6280,13 @@ var RedditThreadInputSchema = {
6217
6280
  url: import_zod4.z.string().min(1).describe("A reddit.com thread/post URL (www, old, new Reddit, or redd.it)."),
6218
6281
  maxComments: import_zod4.z.number().int().min(1).max(2e3).optional().describe("Optional cap on comments returned. Omit to return all captured comments.")
6219
6282
  };
6283
+ var RedditTrendingInputSchema = {
6284
+ topic: import_zod4.z.string().min(1).describe('Topic to scan, in plain words (e.g. "crm for small business"). Not a URL \u2014 pass a known thread URL to reddit_thread instead.'),
6285
+ subreddit: import_zod4.z.string().min(1).optional().describe('Bare subreddit name to scope the scan to one community, e.g. "SEO" (no r/ prefix, no URL). Omit to scan all of Reddit.'),
6286
+ maxThreads: import_zod4.z.number().int().min(1).max(25).default(5).describe("Top-ranked threads to keep. Default 5 \u2014 each scraped thread adds ~30s and the request has a hard ~300s ceiling, so raise this above 5 only with includeComments:false."),
6287
+ includeComments: import_zod4.z.boolean().default(true).describe("Scrape each ranked thread for its body and comments to extract questions. Set false for a fast, cheap discovery-only sweep (ranking + metadata only), then call reddit_thread on the winners."),
6288
+ maxCommentsPerThread: import_zod4.z.number().int().min(1).max(200).default(50).describe("Comments captured per scraped thread when includeComments is true. Default 50. Billed per captured comment.")
6289
+ };
6220
6290
  var VideoFrameAnalysisInputSchema = {
6221
6291
  sourceUrl: import_zod4.z.string().min(1).describe("A YouTube, Facebook, Instagram, TikTok, or Vimeo URL (downloaded automatically), or a direct video file URL (.mp4/.webm/.mov). Videos up to 30 minutes are supported."),
6222
6292
  intervalS: import_zod4.z.number().min(1).max(30).optional().describe("Preferred seconds between sampled frames (1-30, default 2). Automatically widened for long videos so the whole duration is covered within the frame budget."),
@@ -6804,6 +6874,32 @@ var RedditThreadOutputSchema = {
6804
6874
  body: import_zod4.z.string()
6805
6875
  }))
6806
6876
  };
6877
+ var RedditTrendingOutputSchema = {
6878
+ topic: import_zod4.z.string(),
6879
+ subreddit: NullableString2,
6880
+ window: import_zod4.z.string(),
6881
+ totals: import_zod4.z.object({
6882
+ threads: import_zod4.z.number().int().min(0),
6883
+ upvotes: import_zod4.z.number().int().min(0),
6884
+ comments: import_zod4.z.number().int().min(0)
6885
+ }),
6886
+ rankedThreads: import_zod4.z.array(import_zod4.z.object({
6887
+ title: import_zod4.z.string(),
6888
+ url: import_zod4.z.string(),
6889
+ subreddit: import_zod4.z.string(),
6890
+ score: import_zod4.z.number().int().min(0),
6891
+ commentCount: import_zod4.z.number().int().min(0),
6892
+ engagementScore: import_zod4.z.number().int().min(0),
6893
+ ageText: import_zod4.z.string(),
6894
+ topQuestions: import_zod4.z.array(import_zod4.z.string())
6895
+ })),
6896
+ questions: import_zod4.z.array(import_zod4.z.object({
6897
+ question: import_zod4.z.string(),
6898
+ threadUrl: import_zod4.z.string()
6899
+ })),
6900
+ threadsScraped: import_zod4.z.number().int().min(0),
6901
+ searchUrl: import_zod4.z.string()
6902
+ };
6807
6903
  var FacebookPageIntelOutputSchema = {
6808
6904
  advertiserName: NullableString2,
6809
6905
  inputMode: import_zod4.z.enum(["pageId", "libraryId", "query"]),
@@ -8094,6 +8190,13 @@ function registerPaaExtractorMcpTools(server2, executor, options = {}) {
8094
8190
  outputSchema: recordOutputSchema("reddit_thread", RedditThreadOutputSchema),
8095
8191
  annotations: liveWebToolAnnotations("Reddit Thread + Comments")
8096
8192
  }, async (input) => formatRedditThread(await executor.redditThread(input), input));
8193
+ server2.registerTool("reddit_trending", {
8194
+ title: "Reddit Trending",
8195
+ description: "Discover the top Reddit conversations about a topic from the last 30 days: searches Reddit (optionally one subreddit), ranks threads by engagement (upvotes + 2x comments), scrapes the top ones, and extracts the real questions people asked. Each scraped thread takes ~30s, so keep maxThreads small; for a wide scan set includeComments:false to rank cheaply first, then read the winners with reddit_thread. Not for reading one known thread URL \u2014 use reddit_thread for that.",
8196
+ inputSchema: RedditTrendingInputSchema,
8197
+ outputSchema: recordOutputSchema("reddit_trending", RedditTrendingOutputSchema),
8198
+ annotations: liveWebToolAnnotations("Reddit Trending")
8199
+ }, async (input) => formatRedditTrending(await executor.redditTrending(input), input));
8097
8200
  server2.registerTool("video_frame_analysis", {
8098
8201
  title: "Video Breakdown (frame-by-frame + transcript)",
8099
8202
  description: "Produce a deep frame-by-frame + transcript breakdown of a video \u2014 pacing, hook, visual style, and how to replicate it. Accepts a YouTube, Facebook, Instagram, TikTok, or Vimeo URL directly (downloaded for you), or a direct video file URL (.mp4/.webm/.mov). Costs $1 per 120 frames requested (max 480 = $4; refunded down if the video can't use them; refunded fully on failure): returns a runId immediately; poll video_frame_analysis_status until done. Videos up to 30 minutes.",
@@ -10944,7 +11047,7 @@ function renderInstallTerminal(options) {
10944
11047
  "1/1 install surfaces ready",
10945
11048
  colorize("Newest: any approved connection read can become an indexed Memory snapshot in one call. OAuth stays tenant-isolated and provider content is redacted and marked untrusted.", "lime", color),
10946
11049
  "",
10947
- `${colorize("Tools", "cyan", color)} ${colorize("(165 MCP tools)", "muted", color)}`,
11050
+ `${colorize("Tools", "cyan", color)} ${colorize("(166 MCP tools)", "muted", color)}`,
10948
11051
  toolRow("search", ["harvest_paa", "search_serp", "maps_search", "maps_place_intel"], color),
10949
11052
  toolRow("extract", ["extract_url", "map_site_urls", "extract_site", "audit_site", "directory_workflow"], color),
10950
11053
  toolRow("build", ["rank_tracker_workflow", "cron plan", "database prompt"], color),