mcp-scraper 0.33.6 → 0.33.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/bin/api-server.cjs +4088 -3097
- package/dist/bin/api-server.cjs.map +1 -1
- package/dist/bin/api-server.js +3 -3
- package/dist/bin/mcp-scraper-cli.cjs +12 -2
- package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
- package/dist/bin/mcp-scraper-cli.js +12 -2
- package/dist/bin/mcp-scraper-cli.js.map +1 -1
- package/dist/bin/mcp-scraper-install.cjs +2 -2
- package/dist/bin/mcp-scraper-install.cjs.map +1 -1
- package/dist/bin/mcp-scraper-install.js +2 -2
- package/dist/bin/mcp-stdio-server.cjs +109 -6
- package/dist/bin/mcp-stdio-server.cjs.map +1 -1
- package/dist/bin/mcp-stdio-server.js +6 -6
- package/dist/bin/paa-harvest.cjs +106 -13
- package/dist/bin/paa-harvest.cjs.map +1 -1
- package/dist/bin/paa-harvest.js +3 -3
- package/dist/{chunk-DUS4ENOY.js → chunk-2EDFOQD7.js} +2 -2
- package/dist/{chunk-DUS4ENOY.js.map → chunk-2EDFOQD7.js.map} +1 -1
- package/dist/{chunk-C2OD4ELT.js → chunk-3HBPKR5G.js} +2 -2
- package/dist/{chunk-AIK7WQD5.js → chunk-62DQAWPF.js} +141 -3
- package/dist/chunk-62DQAWPF.js.map +1 -0
- package/dist/{chunk-6ZUK4H66.js → chunk-H6QECE2U.js} +109 -6
- package/dist/chunk-H6QECE2U.js.map +1 -0
- package/dist/chunk-HAMN42SP.js +7 -0
- package/dist/chunk-HAMN42SP.js.map +1 -0
- package/dist/{chunk-LKIP7VDN.js → chunk-JBSGBSGT.js} +46 -3
- package/dist/chunk-JBSGBSGT.js.map +1 -0
- package/dist/{chunk-XL6YIWJ3.js → chunk-SUOHUXQS.js} +3 -3
- package/dist/{chunk-WJJTGRK4.js → chunk-V36LS5YV.js} +2 -2
- package/dist/{chunk-A54ELKJK.js → chunk-XVCETVYY.js} +129 -15
- package/dist/chunk-XVCETVYY.js.map +1 -0
- package/dist/{db-5SZQUI7A.js → db-YAI5AQOI.js} +16 -2
- package/dist/{extract-bundle-YNXTFCC3.js → extract-bundle-GPZRLBVY.js} +2 -2
- package/dist/index.cjs +106 -13
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +3 -3
- package/dist/{server-SHOYO52X.js → server-B35CQJB3.js} +1358 -772
- package/dist/server-B35CQJB3.js.map +1 -0
- package/dist/{site-extract-repository-ZDTO7ODZ.js → site-extract-repository-JHMVHENZ.js} +4 -4
- package/dist/{worker-BJ64GCDE.js → worker-3LYNVGLF.js} +6 -6
- package/docs/mcp-tool-manifest.generated.json +185 -3
- package/package.json +1 -1
- package/dist/chunk-6ZUK4H66.js.map +0 -1
- package/dist/chunk-A54ELKJK.js.map +0 -1
- package/dist/chunk-AIK7WQD5.js.map +0 -1
- package/dist/chunk-F65IIUHI.js +0 -7
- package/dist/chunk-F65IIUHI.js.map +0 -1
- package/dist/chunk-LKIP7VDN.js.map +0 -1
- package/dist/server-SHOYO52X.js.map +0 -1
- /package/dist/{chunk-C2OD4ELT.js.map → chunk-3HBPKR5G.js.map} +0 -0
- /package/dist/{chunk-XL6YIWJ3.js.map → chunk-SUOHUXQS.js.map} +0 -0
- /package/dist/{chunk-WJJTGRK4.js.map → chunk-V36LS5YV.js.map} +0 -0
- /package/dist/{db-5SZQUI7A.js.map → db-YAI5AQOI.js.map} +0 -0
- /package/dist/{extract-bundle-YNXTFCC3.js.map → extract-bundle-GPZRLBVY.js.map} +0 -0
- /package/dist/{site-extract-repository-ZDTO7ODZ.js.map → site-extract-repository-JHMVHENZ.js.map} +0 -0
- /package/dist/{worker-BJ64GCDE.js.map → worker-3LYNVGLF.js.map} +0 -0
|
@@ -278,6 +278,9 @@ var HttpMcpToolExecutor = class {
|
|
|
278
278
|
redditThread(input) {
|
|
279
279
|
return this.call("/reddit/thread", input, this.httpTimeoutOverrideMs ?? 24e4);
|
|
280
280
|
}
|
|
281
|
+
redditTrending(input) {
|
|
282
|
+
return this.call("/reddit/trending", input, this.httpTimeoutOverrideMs ?? 3e5);
|
|
283
|
+
}
|
|
281
284
|
videoFrameAnalysis(input) {
|
|
282
285
|
return this.call("/video/analyze", input);
|
|
283
286
|
}
|
|
@@ -564,7 +567,7 @@ render();
|
|
|
564
567
|
}
|
|
565
568
|
|
|
566
569
|
// src/version.ts
|
|
567
|
-
var PACKAGE_VERSION = "0.33.
|
|
570
|
+
var PACKAGE_VERSION = "0.33.7";
|
|
568
571
|
|
|
569
572
|
// src/mcp/browser-agent-tool-schemas.ts
|
|
570
573
|
var import_zod = require("zod");
|
|
@@ -3107,8 +3110,8 @@ var SUBSCRIPTION_TIERS = {
|
|
|
3107
3110
|
"price_1TmiHRS8aAcsk3TGwmSNfNIa": { tier: "starter", label: "Starter", price_id: "price_1TmiHRS8aAcsk3TGwmSNfNIa", monthly_usd: 12, credits_mc: 8e6, concurrency: 3, intro_coupon: "mcp-starter-1dollar-intro-12" },
|
|
3108
3111
|
"price_1TrgihS8aAcsk3TG5cglrq4D": { tier: "growth", label: "Growth", price_id: "price_1TrgihS8aAcsk3TG5cglrq4D", monthly_usd: 40, credits_mc: 26666700, concurrency: 10, intro_coupon: null },
|
|
3109
3112
|
"price_1TrgihS8aAcsk3TG4HnG4gbY": { tier: "scale", label: "Scale", price_id: "price_1TrgihS8aAcsk3TG4HnG4gbY", monthly_usd: 100, credits_mc: 8e7, concurrency: 20, intro_coupon: null },
|
|
3110
|
-
"
|
|
3111
|
-
"
|
|
3113
|
+
"price_1TwqIVS8aAcsk3TG0FE3ddDM": { tier: "sonic", label: "Sonic Annual", price_id: "price_1TwqIVS8aAcsk3TG0FE3ddDM", monthly_usd: 25, billed_usd: 300, credits_mc: 25e7, concurrency: 3, intro_coupon: null, billing_interval: "year", credits_never_expire: true, private_offer: true },
|
|
3114
|
+
"price_1TwqIWS8aAcsk3TGbGbtlLak": { tier: "sonic-memory", label: "Sonic Annual + Memory Pro", price_id: "price_1TwqIWS8aAcsk3TGbGbtlLak", monthly_usd: 34.5, billed_usd: 414, credits_mc: 25e7, concurrency: 3, intro_coupon: null, billing_interval: "year", credits_never_expire: true, includes_memory: true, private_offer: true }
|
|
3112
3115
|
};
|
|
3113
3116
|
var SUBSCRIPTION_TIER_BY_KEY = Object.fromEntries(
|
|
3114
3117
|
Object.values(SUBSCRIPTION_TIERS).map((t) => [t.tier, t])
|
|
@@ -4300,6 +4303,63 @@ ${commentMd || "_No comments captured._"}`
|
|
|
4300
4303
|
}
|
|
4301
4304
|
};
|
|
4302
4305
|
}
|
|
4306
|
+
function formatRedditTrending(raw, input) {
|
|
4307
|
+
const parsed = parseData(raw);
|
|
4308
|
+
if ("error" in parsed) return { content: [{ type: "text", text: parsed.error }], isError: true };
|
|
4309
|
+
const d = parsed.data;
|
|
4310
|
+
const threads = d.rankedThreads ?? [];
|
|
4311
|
+
const questions = (d.questions ?? []).filter((q) => q.question);
|
|
4312
|
+
const totals = {
|
|
4313
|
+
threads: d.totals?.threads ?? threads.length,
|
|
4314
|
+
upvotes: d.totals?.upvotes ?? 0,
|
|
4315
|
+
comments: d.totals?.comments ?? 0
|
|
4316
|
+
};
|
|
4317
|
+
const subreddit = d.subreddit ?? input.subreddit ?? null;
|
|
4318
|
+
const threadBlocks = threads.map((t, i) => [
|
|
4319
|
+
`### ${i + 1}. ${t.title || "Untitled"}`,
|
|
4320
|
+
`**${t.subreddit || "r/?"}** \xB7 ${t.score ?? 0} upvotes \xB7 ${t.commentCount ?? 0} comments \xB7 engagement ${t.engagementScore ?? 0}${t.ageText ? ` \xB7 ${t.ageText}` : ""}`,
|
|
4321
|
+
t.url ? `${t.url}` : "",
|
|
4322
|
+
t.topQuestions?.length ? `**Top questions:**
|
|
4323
|
+
${t.topQuestions.map((q) => `- ${q}`).join("\n")}` : ""
|
|
4324
|
+
].filter(Boolean).join("\n")).join("\n\n");
|
|
4325
|
+
const questionList = questions.map((q) => `- ${q.question}${q.threadUrl ? `
|
|
4326
|
+
${q.threadUrl}` : ""}`).join("\n");
|
|
4327
|
+
const full = [
|
|
4328
|
+
`# Reddit Trending: "${d.topic || input.topic}"${subreddit ? ` in r/${subreddit}` : ""}`,
|
|
4329
|
+
`**${totals.threads} threads \xB7 ${totals.upvotes} upvotes \xB7 ${totals.comments} comments** \xB7 last 30 days \xB7 ${d.threadsScraped ?? 0} threads scraped`,
|
|
4330
|
+
`
|
|
4331
|
+
## Ranked threads
|
|
4332
|
+
${threadBlocks || "_No threads found._"}`,
|
|
4333
|
+
`
|
|
4334
|
+
## Questions people asked
|
|
4335
|
+
${questionList || "_No questions extracted._"}`,
|
|
4336
|
+
`
|
|
4337
|
+
---
|
|
4338
|
+
\u{1F4A1} Deep-dive a winner's full comment tree: pass its url to \`reddit_thread\``
|
|
4339
|
+
].join("\n");
|
|
4340
|
+
return {
|
|
4341
|
+
...oneBlock(full),
|
|
4342
|
+
structuredContent: {
|
|
4343
|
+
topic: d.topic ?? input.topic,
|
|
4344
|
+
subreddit,
|
|
4345
|
+
window: d.window ?? "30d",
|
|
4346
|
+
totals,
|
|
4347
|
+
rankedThreads: threads.map((t) => ({
|
|
4348
|
+
title: t.title ?? "",
|
|
4349
|
+
url: t.url ?? "",
|
|
4350
|
+
subreddit: t.subreddit ?? "",
|
|
4351
|
+
score: Number(t.score ?? 0),
|
|
4352
|
+
commentCount: Number(t.commentCount ?? 0),
|
|
4353
|
+
engagementScore: Number(t.engagementScore ?? 0),
|
|
4354
|
+
ageText: t.ageText ?? "",
|
|
4355
|
+
topQuestions: t.topQuestions ?? []
|
|
4356
|
+
})),
|
|
4357
|
+
questions: questions.map((q) => ({ question: q.question ?? "", threadUrl: q.threadUrl ?? "" })),
|
|
4358
|
+
threadsScraped: Number(d.threadsScraped ?? 0),
|
|
4359
|
+
searchUrl: d.searchUrl ?? ""
|
|
4360
|
+
}
|
|
4361
|
+
};
|
|
4362
|
+
}
|
|
4303
4363
|
function formatFacebookAdSearch(raw, input) {
|
|
4304
4364
|
const parsed = parseData(raw);
|
|
4305
4365
|
if ("error" in parsed) return { content: [{ type: "text", text: parsed.error }], isError: true };
|
|
@@ -5668,8 +5728,11 @@ seam is noted so you can chain them.
|
|
|
5668
5728
|
url the user gives).
|
|
5669
5729
|
|
|
5670
5730
|
## Reddit
|
|
5671
|
-
-
|
|
5672
|
-
Reddit's bot wall itself (no login needed). Find threads first with \`search_serp\`.
|
|
5731
|
+
- Read ONE known reddit.com thread/post URL -> **reddit_thread**. Returns the post plus its comment tree and
|
|
5732
|
+
handles Reddit's bot wall itself (no login needed). Find threads first with \`search_serp\` or \`reddit_trending\`.
|
|
5733
|
+
- DISCOVER what a niche is talking about (topic, no known thread) -> **reddit_trending** (takes a topic, optional
|
|
5734
|
+
subreddit; returns the last 30 days' top threads ranked by engagement plus the questions people asked \u2014
|
|
5735
|
+
feed winning \`rankedThreads[].url\` values into \`reddit_thread\` for the full comment tree).
|
|
5673
5736
|
|
|
5674
5737
|
## Other sites & logins (browser agent)
|
|
5675
5738
|
For an arbitrary site or a logged-in dashboard with no dedicated tool, use the browser_* agent. **First
|
|
@@ -6217,6 +6280,13 @@ var RedditThreadInputSchema = {
|
|
|
6217
6280
|
url: import_zod4.z.string().min(1).describe("A reddit.com thread/post URL (www, old, new Reddit, or redd.it)."),
|
|
6218
6281
|
maxComments: import_zod4.z.number().int().min(1).max(2e3).optional().describe("Optional cap on comments returned. Omit to return all captured comments.")
|
|
6219
6282
|
};
|
|
6283
|
+
var RedditTrendingInputSchema = {
|
|
6284
|
+
topic: import_zod4.z.string().min(1).describe('Topic to scan, in plain words (e.g. "crm for small business"). Not a URL \u2014 pass a known thread URL to reddit_thread instead.'),
|
|
6285
|
+
subreddit: import_zod4.z.string().min(1).optional().describe('Bare subreddit name to scope the scan to one community, e.g. "SEO" (no r/ prefix, no URL). Omit to scan all of Reddit.'),
|
|
6286
|
+
maxThreads: import_zod4.z.number().int().min(1).max(25).default(5).describe("Top-ranked threads to keep. Default 5 \u2014 each scraped thread adds ~30s and the request has a hard ~300s ceiling, so raise this above 5 only with includeComments:false."),
|
|
6287
|
+
includeComments: import_zod4.z.boolean().default(true).describe("Scrape each ranked thread for its body and comments to extract questions. Set false for a fast, cheap discovery-only sweep (ranking + metadata only), then call reddit_thread on the winners."),
|
|
6288
|
+
maxCommentsPerThread: import_zod4.z.number().int().min(1).max(200).default(50).describe("Comments captured per scraped thread when includeComments is true. Default 50. Billed per captured comment.")
|
|
6289
|
+
};
|
|
6220
6290
|
var VideoFrameAnalysisInputSchema = {
|
|
6221
6291
|
sourceUrl: import_zod4.z.string().min(1).describe("A YouTube, Facebook, Instagram, TikTok, or Vimeo URL (downloaded automatically), or a direct video file URL (.mp4/.webm/.mov). Videos up to 30 minutes are supported."),
|
|
6222
6292
|
intervalS: import_zod4.z.number().min(1).max(30).optional().describe("Preferred seconds between sampled frames (1-30, default 2). Automatically widened for long videos so the whole duration is covered within the frame budget."),
|
|
@@ -6804,6 +6874,32 @@ var RedditThreadOutputSchema = {
|
|
|
6804
6874
|
body: import_zod4.z.string()
|
|
6805
6875
|
}))
|
|
6806
6876
|
};
|
|
6877
|
+
var RedditTrendingOutputSchema = {
|
|
6878
|
+
topic: import_zod4.z.string(),
|
|
6879
|
+
subreddit: NullableString2,
|
|
6880
|
+
window: import_zod4.z.string(),
|
|
6881
|
+
totals: import_zod4.z.object({
|
|
6882
|
+
threads: import_zod4.z.number().int().min(0),
|
|
6883
|
+
upvotes: import_zod4.z.number().int().min(0),
|
|
6884
|
+
comments: import_zod4.z.number().int().min(0)
|
|
6885
|
+
}),
|
|
6886
|
+
rankedThreads: import_zod4.z.array(import_zod4.z.object({
|
|
6887
|
+
title: import_zod4.z.string(),
|
|
6888
|
+
url: import_zod4.z.string(),
|
|
6889
|
+
subreddit: import_zod4.z.string(),
|
|
6890
|
+
score: import_zod4.z.number().int().min(0),
|
|
6891
|
+
commentCount: import_zod4.z.number().int().min(0),
|
|
6892
|
+
engagementScore: import_zod4.z.number().int().min(0),
|
|
6893
|
+
ageText: import_zod4.z.string(),
|
|
6894
|
+
topQuestions: import_zod4.z.array(import_zod4.z.string())
|
|
6895
|
+
})),
|
|
6896
|
+
questions: import_zod4.z.array(import_zod4.z.object({
|
|
6897
|
+
question: import_zod4.z.string(),
|
|
6898
|
+
threadUrl: import_zod4.z.string()
|
|
6899
|
+
})),
|
|
6900
|
+
threadsScraped: import_zod4.z.number().int().min(0),
|
|
6901
|
+
searchUrl: import_zod4.z.string()
|
|
6902
|
+
};
|
|
6807
6903
|
var FacebookPageIntelOutputSchema = {
|
|
6808
6904
|
advertiserName: NullableString2,
|
|
6809
6905
|
inputMode: import_zod4.z.enum(["pageId", "libraryId", "query"]),
|
|
@@ -8094,6 +8190,13 @@ function registerPaaExtractorMcpTools(server2, executor, options = {}) {
|
|
|
8094
8190
|
outputSchema: recordOutputSchema("reddit_thread", RedditThreadOutputSchema),
|
|
8095
8191
|
annotations: liveWebToolAnnotations("Reddit Thread + Comments")
|
|
8096
8192
|
}, async (input) => formatRedditThread(await executor.redditThread(input), input));
|
|
8193
|
+
server2.registerTool("reddit_trending", {
|
|
8194
|
+
title: "Reddit Trending",
|
|
8195
|
+
description: "Discover the top Reddit conversations about a topic from the last 30 days: searches Reddit (optionally one subreddit), ranks threads by engagement (upvotes + 2x comments), scrapes the top ones, and extracts the real questions people asked. Each scraped thread takes ~30s, so keep maxThreads small; for a wide scan set includeComments:false to rank cheaply first, then read the winners with reddit_thread. Not for reading one known thread URL \u2014 use reddit_thread for that.",
|
|
8196
|
+
inputSchema: RedditTrendingInputSchema,
|
|
8197
|
+
outputSchema: recordOutputSchema("reddit_trending", RedditTrendingOutputSchema),
|
|
8198
|
+
annotations: liveWebToolAnnotations("Reddit Trending")
|
|
8199
|
+
}, async (input) => formatRedditTrending(await executor.redditTrending(input), input));
|
|
8097
8200
|
server2.registerTool("video_frame_analysis", {
|
|
8098
8201
|
title: "Video Breakdown (frame-by-frame + transcript)",
|
|
8099
8202
|
description: "Produce a deep frame-by-frame + transcript breakdown of a video \u2014 pacing, hook, visual style, and how to replicate it. Accepts a YouTube, Facebook, Instagram, TikTok, or Vimeo URL directly (downloaded for you), or a direct video file URL (.mp4/.webm/.mov). Costs $1 per 120 frames requested (max 480 = $4; refunded down if the video can't use them; refunded fully on failure): returns a runId immediately; poll video_frame_analysis_status until done. Videos up to 30 minutes.",
|
|
@@ -10944,7 +11047,7 @@ function renderInstallTerminal(options) {
|
|
|
10944
11047
|
"1/1 install surfaces ready",
|
|
10945
11048
|
colorize("Newest: any approved connection read can become an indexed Memory snapshot in one call. OAuth stays tenant-isolated and provider content is redacted and marked untrusted.", "lime", color),
|
|
10946
11049
|
"",
|
|
10947
|
-
`${colorize("Tools", "cyan", color)} ${colorize("(
|
|
11050
|
+
`${colorize("Tools", "cyan", color)} ${colorize("(166 MCP tools)", "muted", color)}`,
|
|
10948
11051
|
toolRow("search", ["harvest_paa", "search_serp", "maps_search", "maps_place_intel"], color),
|
|
10949
11052
|
toolRow("extract", ["extract_url", "map_site_urls", "extract_site", "audit_site", "directory_workflow"], color),
|
|
10950
11053
|
toolRow("build", ["rank_tracker_workflow", "cron plan", "database prompt"], color),
|