mcp-scraper 0.3.10 → 0.3.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -3
- package/dist/bin/api-server.cjs +115 -40
- package/dist/bin/api-server.cjs.map +1 -1
- package/dist/bin/api-server.js +2 -2
- package/dist/bin/browser-agent-stdio-server.cjs +206 -31
- package/dist/bin/browser-agent-stdio-server.cjs.map +1 -1
- package/dist/bin/browser-agent-stdio-server.js +3 -2
- package/dist/bin/browser-agent-stdio-server.js.map +1 -1
- package/dist/bin/mcp-scraper-cli.cjs +1 -1
- package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
- package/dist/bin/mcp-scraper-cli.js +1 -1
- package/dist/bin/mcp-scraper-combined-stdio-server.cjs +262 -60
- package/dist/bin/mcp-scraper-combined-stdio-server.cjs.map +1 -1
- package/dist/bin/mcp-scraper-combined-stdio-server.js +4 -3
- package/dist/bin/mcp-scraper-combined-stdio-server.js.map +1 -1
- package/dist/bin/mcp-scraper-install.cjs +1 -1
- package/dist/bin/mcp-scraper-install.cjs.map +1 -1
- package/dist/bin/mcp-scraper-install.js +1 -1
- package/dist/bin/mcp-stdio-server.cjs +38 -11
- package/dist/bin/mcp-stdio-server.cjs.map +1 -1
- package/dist/bin/mcp-stdio-server.js +2 -2
- package/dist/bin/paa-harvest.cjs +21 -1
- package/dist/bin/paa-harvest.cjs.map +1 -1
- package/dist/bin/paa-harvest.js +1 -1
- package/dist/chunk-3YGKXXUG.js +133 -0
- package/dist/chunk-3YGKXXUG.js.map +1 -0
- package/dist/{chunk-UWSG3C5J.js → chunk-4OPKIDON.js} +22 -2
- package/dist/chunk-4OPKIDON.js.map +1 -0
- package/dist/{chunk-CED7X4WB.js → chunk-7R7VBQRV.js} +39 -12
- package/dist/chunk-7R7VBQRV.js.map +1 -0
- package/dist/{chunk-QPFF3V2R.js → chunk-BW3DGFNQ.js} +36 -9
- package/dist/chunk-BW3DGFNQ.js.map +1 -0
- package/dist/chunk-FB5ZJGFN.js +7 -0
- package/dist/chunk-FB5ZJGFN.js.map +1 -0
- package/dist/index.cjs +21 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +1 -1
- package/dist/{server-QTV2EUKA.js → server-CUPDJLWM.js} +44 -142
- package/dist/server-CUPDJLWM.js.map +1 -0
- package/dist/{worker-56IXWOQU.js → worker-FG7ZWEGA.js} +2 -2
- package/docs/mcp-tool-craft-lint.generated.md +1 -1
- package/docs/mcp-tool-manifest.generated.json +3 -3
- package/package.json +1 -1
- package/dist/chunk-CED7X4WB.js.map +0 -1
- package/dist/chunk-NEGW2ZEJ.js +0 -7
- package/dist/chunk-NEGW2ZEJ.js.map +0 -1
- package/dist/chunk-QPFF3V2R.js.map +0 -1
- package/dist/chunk-UWSG3C5J.js.map +0 -1
- package/dist/server-QTV2EUKA.js.map +0 -1
- /package/dist/{worker-56IXWOQU.js.map → worker-FG7ZWEGA.js.map} +0 -0
package/README.md
CHANGED
|
@@ -186,8 +186,8 @@ env = { MCP_SCRAPER_API_KEY = "sk_live_your_key" }
|
|
|
186
186
|
- `facebook_page_intel`
|
|
187
187
|
- `facebook_ad_transcribe` — transcribe a direct Facebook ad video URL returned by `facebook_page_intel`.
|
|
188
188
|
- `facebook_video_transcribe` — transcribe an organic Facebook reel, video, watch, post, or share URL, including `fb.watch` links. The tool renders the page, extracts the best matching public Facebook CDN MP4 URL, then returns transcript text, timestamped chunks, selected quality, video metadata, and the extracted MP4 URL for follow-up download.
|
|
189
|
-
- `instagram_profile_content` — discover Instagram profile grid content links for a handle or profile URL. Returns collected post/reel/tv URLs, profile counts, type counts, shortcodes, browser details, pagination attempts, stop reason, and limitations.
|
|
190
|
-
- `instagram_media_download` — extract and download one Instagram post/reel/tv URL. Returns text/caption, image URL/downloads, selected video/audio MP4 tracks, optional muxed MP4 when `ffmpeg` is available, optional transcript, and browser details.
|
|
189
|
+
- `instagram_profile_content` — discover Instagram profile grid content links for a handle or profile URL, optionally through a saved hosted browser `profile` for authenticated access. Returns collected post/reel/tv URLs, profile counts, type counts, shortcodes, browser details, pagination attempts, stop reason, and limitations.
|
|
190
|
+
- `instagram_media_download` — extract and download one Instagram post/reel/tv URL, optionally through a saved hosted browser `profile` for authenticated access. Returns text/caption, image URL/downloads, selected video/audio MP4 tracks, optional muxed MP4 when `ffmpeg` is available, optional transcript, and browser details.
|
|
191
191
|
- `maps_search` — search Google Maps for multiple business/profile candidates. Use for GMB/GBP prospect lists, competitors, categories, and anything needing more than the Google 3-pack. In default `proxyMode: "location"`, retryable failures rotate to a new residential proxy and new browser session for up to 5 attempts. `maxResults` defaults to 10 and is capped at 50.
|
|
192
192
|
- `maps_place_intel` — hydrate one known/named Google Maps business with profile details and optional reviews. Use after `maps_search` when a selected candidate needs full details.
|
|
193
193
|
- `directory_workflow` — build city-by-city directory/prospecting datasets from Census place selection plus Google Maps searches. Use it for requests like "all cities over 100k population in Tennessee, then get 20 roofers from Maps." In default `proxyMode: "location"`, each city search rotates retryable failures to a new residential proxy and new browser session for up to 5 attempts. The saved CSV includes `source_location`, `result_position`, `business_name`, `review_stars`, `review_count`, `category`, `address`, `phone`, `hours_status`, `website_url`, `directions_url`, `place_url`, `cid`, `cid_decimal`, Census population, and ZIP groups.
|
|
@@ -218,6 +218,7 @@ env = { MCP_SCRAPER_API_KEY = "sk_live_your_key" }
|
|
|
218
218
|
- `browser_replay_download` — download and save the replay MP4 under `MCP_SCRAPER_OUTPUT_DIR/browser-replays`.
|
|
219
219
|
- `browser_replay_mark` — while recording, locate a DOM target and return a replay-timed annotation object.
|
|
220
220
|
- `browser_replay_annotate` — download a replay MP4, render timed boxes, circles, underlines, arrows, and labels using annotation objects from `browser_replay_mark` or exact bounds from `browser_locate`, and save a new annotated MP4 under `MCP_SCRAPER_OUTPUT_DIR/browser-replays`.
|
|
221
|
+
- `browser_capture_fanout` — capture ChatGPT/Claude AI-search fan-out from an open logged-in hosted session. Use `export=true` for durable artifacts. It writes full `fanout.json`, query/source/citation/domain/snippet CSVs, TSV, and `report.html` under `MCP_SCRAPER_OUTPUT_DIR/fanout`, returning paths relative to `MCP_SCRAPER_OUTPUT_DIR` so users can share portable paths.
|
|
221
222
|
- `browser_close`
|
|
222
223
|
- `browser_list_sessions`
|
|
223
224
|
|
|
@@ -247,7 +248,7 @@ The `mcp-scraper` and `mcp-scraper-combined` NPX stdio servers also expose saved
|
|
|
247
248
|
- `BROWSER_AGENT_PROFILE_NAME` is optional and sets the default saved hosted browser profile for `browser-agent` and `mcp-scraper-combined` stdio sessions. Aliases: `BROWSER_SERVICE_PROFILE_NAME`, `KERNEL_BROWSER_PROFILE_NAME`, `KERNEL_PROFILE_NAME`.
|
|
248
249
|
- `BROWSER_AGENT_PROFILE_SAVE_CHANGES=true` is optional hosted setup behavior. It persists cookies and storage back to the named profile when `browser_close` deletes the hosted browser session. Aliases: `BROWSER_SERVICE_PROFILE_SAVE_CHANGES`, `KERNEL_BROWSER_PROFILE_SAVE_CHANGES`, `KERNEL_PROFILE_SAVE_CHANGES`.
|
|
249
250
|
|
|
250
|
-
Every web intelligence tool call made through `mcp-scraper` or `mcp-scraper-combined` saves a full Markdown report to disk by default and returns the file path in the MCP response. The hosted `/mcp` endpoint returns reports inline only and never writes files. Browser replay downloads are saved by `browser_replay_download` under `MCP_SCRAPER_OUTPUT_DIR/browser-replays`.
|
|
251
|
+
Every web intelligence tool call made through `mcp-scraper` or `mcp-scraper-combined` saves a full Markdown report to disk by default and returns the file path in the MCP response. The hosted `/mcp` endpoint returns reports inline only and never writes files. Browser replay downloads are saved by `browser_replay_download` under `MCP_SCRAPER_OUTPUT_DIR/browser-replays`. AI fan-out exports are saved by `browser_capture_fanout` with `export=true` under `MCP_SCRAPER_OUTPUT_DIR/fanout`, and return relative paths.
|
|
251
252
|
|
|
252
253
|
## Updating Existing Installs
|
|
253
254
|
|
package/dist/bin/api-server.cjs
CHANGED
|
@@ -9164,6 +9164,16 @@ function proxyIdSuffix(proxyId) {
|
|
|
9164
9164
|
function errorText(err) {
|
|
9165
9165
|
return err instanceof Error ? err.message : String(err);
|
|
9166
9166
|
}
|
|
9167
|
+
function isKernelProfileConflict(err) {
|
|
9168
|
+
return /\b409\b|conflict|already exists/i.test(errorText(err));
|
|
9169
|
+
}
|
|
9170
|
+
async function ensureKernelProfile(kernelClient, name) {
|
|
9171
|
+
try {
|
|
9172
|
+
await kernelClient.profiles.create({ name });
|
|
9173
|
+
} catch (err) {
|
|
9174
|
+
if (!isKernelProfileConflict(err)) throw err;
|
|
9175
|
+
}
|
|
9176
|
+
}
|
|
9167
9177
|
function rankCheckContextOptions(config) {
|
|
9168
9178
|
return {
|
|
9169
9179
|
viewport: config.viewport,
|
|
@@ -9266,10 +9276,19 @@ var init_BrowserDriver = __esm({
|
|
|
9266
9276
|
if (config.kernelApiKey) {
|
|
9267
9277
|
this.kernelClient = new import_sdk4.default({ apiKey: config.kernelApiKey });
|
|
9268
9278
|
const timeoutSeconds = positiveIntFromEnv("KERNEL_BROWSER_TIMEOUT_SECONDS", DEFAULT_KERNEL_BROWSER_TIMEOUT_SECONDS);
|
|
9279
|
+
if (config.kernelProfileName && config.kernelProfileSaveChanges === true) {
|
|
9280
|
+
await ensureKernelProfile(this.kernelClient, config.kernelProfileName);
|
|
9281
|
+
}
|
|
9269
9282
|
const kernelBrowser = await this.kernelClient.browsers.create({
|
|
9270
9283
|
stealth: true,
|
|
9271
9284
|
timeout_seconds: timeoutSeconds,
|
|
9272
|
-
...config.kernelProxyId ? { proxy_id: config.kernelProxyId } : {}
|
|
9285
|
+
...config.kernelProxyId ? { proxy_id: config.kernelProxyId } : {},
|
|
9286
|
+
...config.kernelProfileName ? {
|
|
9287
|
+
profile: {
|
|
9288
|
+
name: config.kernelProfileName,
|
|
9289
|
+
...typeof config.kernelProfileSaveChanges === "boolean" ? { save_changes: config.kernelProfileSaveChanges } : {}
|
|
9290
|
+
}
|
|
9291
|
+
} : {}
|
|
9273
9292
|
});
|
|
9274
9293
|
this.kernelSessionId = kernelBrowser.session_id;
|
|
9275
9294
|
let defaultProxyDisabled = null;
|
|
@@ -9313,6 +9332,7 @@ var init_BrowserDriver = __esm({
|
|
|
9313
9332
|
timeout_seconds: timeoutSeconds,
|
|
9314
9333
|
proxy_mode: proxyMode,
|
|
9315
9334
|
proxy_id_present: Boolean(config.kernelProxyId),
|
|
9335
|
+
profile_name_present: Boolean(config.kernelProfileName),
|
|
9316
9336
|
proxy_resolution_source: config.kernelProxyResolution?.source
|
|
9317
9337
|
}));
|
|
9318
9338
|
if (this.debugEnabled) {
|
|
@@ -13121,8 +13141,8 @@ function ownerFromMetadata(metadata) {
|
|
|
13121
13141
|
return match?.[1]?.trim() || null;
|
|
13122
13142
|
}
|
|
13123
13143
|
async function collectInstagramProfileContentFromPage(page, opts) {
|
|
13124
|
-
const maxItems = Math.min(
|
|
13125
|
-
const maxScrolls = Math.min(
|
|
13144
|
+
const maxItems = Math.min(2e3, Math.max(1, opts.maxItems ?? 50));
|
|
13145
|
+
const maxScrolls = Math.min(250, Math.max(0, opts.maxScrolls ?? 10));
|
|
13126
13146
|
const scrollDelayMs = Math.min(5e3, Math.max(250, opts.scrollDelayMs ?? 1200));
|
|
13127
13147
|
const stableScrollLimit = Math.min(10, Math.max(1, opts.stableScrollLimit ?? 4));
|
|
13128
13148
|
const seen = /* @__PURE__ */ new Set();
|
|
@@ -13343,21 +13363,32 @@ var init_InstagramContentExtractor = __esm({
|
|
|
13343
13363
|
function invalidRequest2(message) {
|
|
13344
13364
|
return { error_code: "invalid_request", message };
|
|
13345
13365
|
}
|
|
13346
|
-
|
|
13366
|
+
function resolveSaveProfileChanges(body) {
|
|
13367
|
+
return body.saveProfileChanges ?? body.save_profile_changes ?? browserServiceProfileSaveChanges();
|
|
13368
|
+
}
|
|
13369
|
+
function resolveProfileName(body) {
|
|
13370
|
+
const explicit = body.profile?.trim();
|
|
13371
|
+
return explicit || browserServiceProfileName();
|
|
13372
|
+
}
|
|
13373
|
+
async function kernelLaunchOptsDirect(profileName, saveProfileChanges) {
|
|
13347
13374
|
return {
|
|
13348
13375
|
headless: true,
|
|
13349
13376
|
kernelApiKey: browserServiceApiKey(),
|
|
13377
|
+
...profileName ? { kernelProfileName: profileName } : {},
|
|
13378
|
+
...typeof saveProfileChanges === "boolean" ? { kernelProfileSaveChanges: saveProfileChanges } : {},
|
|
13350
13379
|
viewport: { width: 1280, height: 900 },
|
|
13351
13380
|
locale: "en-US"
|
|
13352
13381
|
};
|
|
13353
13382
|
}
|
|
13354
|
-
async function resolveInstagramLaunch() {
|
|
13383
|
+
async function resolveInstagramLaunch(body) {
|
|
13384
|
+
const profileName = resolveProfileName(body);
|
|
13385
|
+
const saveProfileChanges = resolveSaveProfileChanges(body);
|
|
13355
13386
|
return {
|
|
13356
|
-
config: await kernelLaunchOptsDirect(),
|
|
13387
|
+
config: await kernelLaunchOptsDirect(profileName, saveProfileChanges),
|
|
13357
13388
|
browser: {
|
|
13358
13389
|
mode: "hosted",
|
|
13359
13390
|
requestedMode: "hosted",
|
|
13360
|
-
profileName: null,
|
|
13391
|
+
profileName: profileName ?? null,
|
|
13361
13392
|
profileSource: "hosted",
|
|
13362
13393
|
profileDirConfigured: false,
|
|
13363
13394
|
executablePathConfigured: false
|
|
@@ -13449,8 +13480,11 @@ var init_instagram_routes = __esm({
|
|
|
13449
13480
|
InstagramProfileContentBodySchema = import_zod17.z.object({
|
|
13450
13481
|
handle: import_zod17.z.string().trim().optional(),
|
|
13451
13482
|
url: import_zod17.z.string().trim().optional(),
|
|
13452
|
-
|
|
13453
|
-
|
|
13483
|
+
profile: import_zod17.z.string().trim().min(1).optional(),
|
|
13484
|
+
saveProfileChanges: import_zod17.z.boolean().optional(),
|
|
13485
|
+
save_profile_changes: import_zod17.z.boolean().optional(),
|
|
13486
|
+
maxItems: import_zod17.z.number().int().min(1).max(2e3).default(50),
|
|
13487
|
+
maxScrolls: import_zod17.z.number().int().min(0).max(250).default(10),
|
|
13454
13488
|
scrollDelayMs: import_zod17.z.number().int().min(250).max(5e3).default(1200),
|
|
13455
13489
|
stableScrollLimit: import_zod17.z.number().int().min(1).max(10).default(4)
|
|
13456
13490
|
}).refine((d) => !!d.handle || !!d.url, {
|
|
@@ -13459,6 +13493,9 @@ var init_instagram_routes = __esm({
|
|
|
13459
13493
|
InstagramMediaTypeSchema = import_zod17.z.enum(["image", "video", "audio"]);
|
|
13460
13494
|
InstagramMediaDownloadBodySchema = import_zod17.z.object({
|
|
13461
13495
|
url: import_zod17.z.string().trim().min(1, "url is required"),
|
|
13496
|
+
profile: import_zod17.z.string().trim().min(1).optional(),
|
|
13497
|
+
saveProfileChanges: import_zod17.z.boolean().optional(),
|
|
13498
|
+
save_profile_changes: import_zod17.z.boolean().optional(),
|
|
13462
13499
|
mediaTypes: import_zod17.z.array(InstagramMediaTypeSchema).default(["image", "video", "audio"]),
|
|
13463
13500
|
downloadMedia: import_zod17.z.boolean().default(true),
|
|
13464
13501
|
downloadAllTracks: import_zod17.z.boolean().default(false),
|
|
@@ -13489,7 +13526,7 @@ var init_instagram_routes = __esm({
|
|
|
13489
13526
|
const { ok, balance_mc } = await debitMc(user.id, MC_COSTS.instagram_profile, LedgerOperation.INSTAGRAM_PROFILE, target.profileUrl);
|
|
13490
13527
|
if (!ok) return c.json(insufficientBalanceResponse(balance_mc, MC_COSTS.instagram_profile), 402);
|
|
13491
13528
|
debited = true;
|
|
13492
|
-
const launch = await resolveInstagramLaunch();
|
|
13529
|
+
const launch = await resolveInstagramLaunch(body);
|
|
13493
13530
|
await driver.launch(launch.config);
|
|
13494
13531
|
await driver.navigateTo(target.profileUrl);
|
|
13495
13532
|
const page = driver.getPage();
|
|
@@ -13536,7 +13573,7 @@ var init_instagram_routes = __esm({
|
|
|
13536
13573
|
const { ok, balance_mc } = await debitMc(user.id, MC_COSTS.instagram_media, LedgerOperation.INSTAGRAM_MEDIA, sourceUrl.href);
|
|
13537
13574
|
if (!ok) return c.json(insufficientBalanceResponse(balance_mc, MC_COSTS.instagram_media), 402);
|
|
13538
13575
|
mediaDebited = true;
|
|
13539
|
-
const launch = await resolveInstagramLaunch();
|
|
13576
|
+
const launch = await resolveInstagramLaunch(body);
|
|
13540
13577
|
await driver.launch(launch.config);
|
|
13541
13578
|
const page = driver.getPage();
|
|
13542
13579
|
const capturedMediaUrls = [];
|
|
@@ -16133,7 +16170,7 @@ function formatInstagramProfileContent(raw, input) {
|
|
|
16133
16170
|
const itemRows = items.slice(0, 100).map(
|
|
16134
16171
|
(item, i) => `| ${i + 1} | ${item.type} | \`${item.shortcode}\` | ${item.url} | ${cell(item.firstSeenStage ?? "")} |`
|
|
16135
16172
|
).join("\n");
|
|
16136
|
-
const browserLabel = "hosted browser";
|
|
16173
|
+
const browserLabel = browser.profileName ? `hosted browser profile ${browser.profileName}` : "hosted browser";
|
|
16137
16174
|
const full = [
|
|
16138
16175
|
`# Instagram Profile Content: ${d.handle ?? input.handle ?? input.url ?? "profile"}`,
|
|
16139
16176
|
`**Collected:** ${items.length} items \xB7 posts ${typeCounts?.post ?? 0} \xB7 reels ${typeCounts?.reel ?? 0} \xB7 tv ${typeCounts?.tv ?? 0}`,
|
|
@@ -16208,7 +16245,7 @@ function formatInstagramMediaDownload(raw, input) {
|
|
|
16208
16245
|
const transcriptText = transcript?.text ?? "";
|
|
16209
16246
|
const chunks = transcript?.chunks ?? [];
|
|
16210
16247
|
const browser = structuredInstagramBrowser(d.browser);
|
|
16211
|
-
const browserLabel = "hosted browser";
|
|
16248
|
+
const browserLabel = browser.profileName ? `hosted browser profile ${browser.profileName}` : "hosted browser";
|
|
16212
16249
|
const downloadRows = downloads.map((download, i) => {
|
|
16213
16250
|
const status = download.error ? `error: ${cell(download.error)}` : `${download.sizeBytes ?? 0} bytes`;
|
|
16214
16251
|
return `| ${i + 1} | ${download.kind} | ${download.savedPath ? `\`${download.savedPath}\`` : "\u2014"} | ${status} |`;
|
|
@@ -21921,7 +21958,7 @@ var PACKAGE_VERSION;
|
|
|
21921
21958
|
var init_version = __esm({
|
|
21922
21959
|
"src/version.ts"() {
|
|
21923
21960
|
"use strict";
|
|
21924
|
-
PACKAGE_VERSION = "0.3.
|
|
21961
|
+
PACKAGE_VERSION = "0.3.12";
|
|
21925
21962
|
}
|
|
21926
21963
|
});
|
|
21927
21964
|
|
|
@@ -21992,13 +22029,17 @@ var init_mcp_tool_schemas = __esm({
|
|
|
21992
22029
|
InstagramProfileContentInputSchema = {
|
|
21993
22030
|
handle: import_zod27.z.string().min(1).optional().describe("Instagram handle, with or without @. Provide handle or url."),
|
|
21994
22031
|
url: import_zod27.z.string().url().optional().describe("Instagram profile URL, e.g. https://www.instagram.com/nasaartemis/. Provide handle or url."),
|
|
21995
|
-
|
|
21996
|
-
|
|
22032
|
+
profile: import_zod27.z.string().min(1).optional().describe("Optional saved hosted browser profile name to load authenticated Instagram access. If omitted, the server uses its configured default profile when present."),
|
|
22033
|
+
saveProfileChanges: import_zod27.z.boolean().optional().describe("Whether to save browser changes back to the hosted profile. Leave unset unless intentionally updating the saved login session."),
|
|
22034
|
+
maxItems: import_zod27.z.number().int().min(1).max(2e3).default(50).describe("Maximum profile grid post/reel/tv URLs to collect. Default 50, maximum 2000. Use higher values only when the user asks for a fuller archive."),
|
|
22035
|
+
maxScrolls: import_zod27.z.number().int().min(0).max(250).default(10).describe("Maximum pagination scroll attempts. Default 10, maximum 250. Increase for long profiles when Instagram continues loading more grid links."),
|
|
21997
22036
|
scrollDelayMs: import_zod27.z.number().int().min(250).max(5e3).default(1200).describe("Delay after each pagination scroll before collecting newly loaded links. Default 1200ms. Increase to 2000-3000ms when Instagram loads slowly."),
|
|
21998
22037
|
stableScrollLimit: import_zod27.z.number().int().min(1).max(10).default(4).describe("Stop after this many consecutive scrolls with no new links or scroll progress. Default 4.")
|
|
21999
22038
|
};
|
|
22000
22039
|
InstagramMediaDownloadInputSchema = {
|
|
22001
22040
|
url: import_zod27.z.string().url().describe("Instagram post, reel, or tv URL, e.g. https://www.instagram.com/reel/SHORTCODE/. The tool renders the page, extracts text, image metadata, and Instagram CDN media tracks."),
|
|
22041
|
+
profile: import_zod27.z.string().min(1).optional().describe("Optional saved hosted browser profile name to load authenticated Instagram access. If omitted, the server uses its configured default profile when present."),
|
|
22042
|
+
saveProfileChanges: import_zod27.z.boolean().optional().describe("Whether to save browser changes back to the hosted profile. Leave unset unless intentionally updating the saved login session."),
|
|
22002
22043
|
mediaTypes: import_zod27.z.array(import_zod27.z.enum(["image", "video", "audio"])).default(["image", "video", "audio"]).describe("Which media types to download when downloadMedia is true. Reels commonly expose separate video-only and audio-only MP4 tracks."),
|
|
22003
22044
|
downloadMedia: import_zod27.z.boolean().default(true).describe("Download extracted text/media files to the MCP Scraper output directory when the API server can write files. Always returns extracted media URLs even when false."),
|
|
22004
22045
|
downloadAllTracks: import_zod27.z.boolean().default(false).describe("Download every captured Instagram MP4 track instead of only the selected best video and audio tracks. Use false by default to avoid duplicate bitrates."),
|
|
@@ -22410,8 +22451,8 @@ var init_mcp_tool_schemas = __esm({
|
|
|
22410
22451
|
executablePathConfigured: import_zod27.z.boolean()
|
|
22411
22452
|
});
|
|
22412
22453
|
InstagramPaginationOutput = import_zod27.z.object({
|
|
22413
|
-
maxItems: import_zod27.z.number().int().min(1).max(
|
|
22414
|
-
maxScrolls: import_zod27.z.number().int().min(0).max(
|
|
22454
|
+
maxItems: import_zod27.z.number().int().min(1).max(2e3),
|
|
22455
|
+
maxScrolls: import_zod27.z.number().int().min(0).max(250),
|
|
22415
22456
|
attemptedScrolls: import_zod27.z.number().int().min(0),
|
|
22416
22457
|
stableScrolls: import_zod27.z.number().int().min(0),
|
|
22417
22458
|
stableScrollLimit: import_zod27.z.number().int().min(1).max(10),
|
|
@@ -23288,6 +23329,29 @@ function youtubeVideoIdFromUrl(url) {
|
|
|
23288
23329
|
}
|
|
23289
23330
|
return null;
|
|
23290
23331
|
}
|
|
23332
|
+
async function readResponseData(res) {
|
|
23333
|
+
const text = await res.text();
|
|
23334
|
+
if (!text.trim()) return null;
|
|
23335
|
+
try {
|
|
23336
|
+
return JSON.parse(text);
|
|
23337
|
+
} catch {
|
|
23338
|
+
return text;
|
|
23339
|
+
}
|
|
23340
|
+
}
|
|
23341
|
+
function httpErrorPayload(path6, res, data) {
|
|
23342
|
+
const objectData = data && typeof data === "object" && !Array.isArray(data) ? data : null;
|
|
23343
|
+
const bodyMessage = objectData ? objectData.message ?? objectData.error ?? objectData.error_code : data;
|
|
23344
|
+
return {
|
|
23345
|
+
...objectData ?? { body: data },
|
|
23346
|
+
error: objectData?.error ?? objectData?.error_code ?? "mcp_http_error",
|
|
23347
|
+
error_type: "http",
|
|
23348
|
+
retryable: res.status === 429 || res.status >= 500,
|
|
23349
|
+
status: res.status,
|
|
23350
|
+
statusText: res.statusText,
|
|
23351
|
+
path: path6,
|
|
23352
|
+
message: typeof bodyMessage === "string" && bodyMessage.trim() ? bodyMessage : `MCP Scraper HTTP ${res.status}${res.statusText ? ` ${res.statusText}` : ""} for ${path6}`
|
|
23353
|
+
};
|
|
23354
|
+
}
|
|
23291
23355
|
var HttpMcpToolExecutor;
|
|
23292
23356
|
var init_http_mcp_tool_executor = __esm({
|
|
23293
23357
|
"src/mcp/http-mcp-tool-executor.ts"() {
|
|
@@ -23320,9 +23384,9 @@ var init_http_mcp_tool_executor = __esm({
|
|
|
23320
23384
|
body: JSON.stringify(body),
|
|
23321
23385
|
signal: AbortSignal.timeout(timeoutMs)
|
|
23322
23386
|
});
|
|
23323
|
-
const data = await res
|
|
23387
|
+
const data = await readResponseData(res);
|
|
23324
23388
|
if (!res.ok) {
|
|
23325
|
-
return { content: [{ type: "text", text: JSON.stringify(data) }], isError: true };
|
|
23389
|
+
return { content: [{ type: "text", text: JSON.stringify(httpErrorPayload(path6, res, data)) }], isError: true };
|
|
23326
23390
|
}
|
|
23327
23391
|
return { content: [{ type: "text", text: JSON.stringify(data) }] };
|
|
23328
23392
|
} catch (err) {
|
|
@@ -23355,9 +23419,9 @@ var init_http_mcp_tool_executor = __esm({
|
|
|
23355
23419
|
},
|
|
23356
23420
|
signal: AbortSignal.timeout(timeoutMs)
|
|
23357
23421
|
});
|
|
23358
|
-
const data = await res
|
|
23422
|
+
const data = await readResponseData(res);
|
|
23359
23423
|
if (!res.ok) {
|
|
23360
|
-
return { content: [{ type: "text", text: JSON.stringify(data) }], isError: true };
|
|
23424
|
+
return { content: [{ type: "text", text: JSON.stringify(httpErrorPayload(path6, res, data)) }], isError: true };
|
|
23361
23425
|
}
|
|
23362
23426
|
return { content: [{ type: "text", text: JSON.stringify(data) }] };
|
|
23363
23427
|
} catch (err) {
|
|
@@ -24432,7 +24496,9 @@ function writeTable(path6, rows, delimiter) {
|
|
|
24432
24496
|
}
|
|
24433
24497
|
function exportFanout(enriched) {
|
|
24434
24498
|
const stamp = safe(enriched.capturedAt.replace(/[:.]/g, "-"));
|
|
24435
|
-
const
|
|
24499
|
+
const outputDir = outputBaseDir3();
|
|
24500
|
+
const relativeDir = (0, import_node_path11.join)("fanout", `${stamp}-${safe(enriched.platform)}`);
|
|
24501
|
+
const dir = (0, import_node_path11.join)(outputDir, relativeDir);
|
|
24436
24502
|
(0, import_node_fs8.mkdirSync)(dir, { recursive: true });
|
|
24437
24503
|
const queryRows = enriched.queries.map((q, i) => ({ index: i + 1, query: q }));
|
|
24438
24504
|
const citationRows = enriched.aggregates.citationOrder.map((c) => {
|
|
@@ -24440,25 +24506,32 @@ function exportFanout(enriched) {
|
|
|
24440
24506
|
return { rank: c.rank, domain: c.domain, url: c.url, timesCited: c.timesCited, siteType: src?.siteType ?? "", title: src?.title ?? "" };
|
|
24441
24507
|
});
|
|
24442
24508
|
const sourceRows2 = enriched.browsedUrls.map((s) => ({ domain: s.domain, url: s.url, cited: s.cited, timesCited: s.timesCited, siteType: s.siteType, title: s.title, round: s.round ?? "" }));
|
|
24509
|
+
const browsedOnlyRows = enriched.browsedOnly.map((s) => ({ domain: s.domain, url: s.url, cited: s.cited, timesCited: s.timesCited, siteType: s.siteType, title: s.title, round: s.round ?? "" }));
|
|
24510
|
+
const snippetRows = enriched.snippets.map((s) => ({ domain: s.domain, url: s.url, title: s.title, text: s.text }));
|
|
24443
24511
|
const domainRows = enriched.aggregates.topSites.map((d) => ({ domain: d.domain, count: d.count, cited: d.cited, timesCited: d.timesCited, siteType: d.siteType }));
|
|
24444
|
-
const
|
|
24445
|
-
|
|
24446
|
-
|
|
24447
|
-
|
|
24448
|
-
|
|
24449
|
-
|
|
24450
|
-
|
|
24451
|
-
|
|
24452
|
-
|
|
24512
|
+
const relativePaths = {
|
|
24513
|
+
relativeTo: "MCP_SCRAPER_OUTPUT_DIR or ~/Downloads/mcp-scraper",
|
|
24514
|
+
dir: relativeDir,
|
|
24515
|
+
json: (0, import_node_path11.join)(relativeDir, "fanout.json"),
|
|
24516
|
+
queriesCsv: (0, import_node_path11.join)(relativeDir, "queries.csv"),
|
|
24517
|
+
queriesTsv: (0, import_node_path11.join)(relativeDir, "queries.tsv"),
|
|
24518
|
+
citationsCsv: (0, import_node_path11.join)(relativeDir, "citations.csv"),
|
|
24519
|
+
sourcesCsv: (0, import_node_path11.join)(relativeDir, "sources.csv"),
|
|
24520
|
+
browsedOnlyCsv: (0, import_node_path11.join)(relativeDir, "browsed-only.csv"),
|
|
24521
|
+
snippetsCsv: (0, import_node_path11.join)(relativeDir, "snippets.csv"),
|
|
24522
|
+
domainsCsv: (0, import_node_path11.join)(relativeDir, "domains.csv"),
|
|
24523
|
+
report: (0, import_node_path11.join)(relativeDir, "report.html")
|
|
24453
24524
|
};
|
|
24454
|
-
(0, import_node_fs8.writeFileSync)(
|
|
24455
|
-
writeTable(
|
|
24456
|
-
writeTable(
|
|
24457
|
-
writeTable(
|
|
24458
|
-
writeTable(
|
|
24459
|
-
writeTable(
|
|
24460
|
-
(0,
|
|
24461
|
-
|
|
24525
|
+
(0, import_node_fs8.writeFileSync)((0, import_node_path11.join)(outputDir, relativePaths.json), JSON.stringify(enriched, null, 2));
|
|
24526
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.queriesCsv), queryRows, ",");
|
|
24527
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.queriesTsv), queryRows, " ");
|
|
24528
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.citationsCsv), citationRows, ",");
|
|
24529
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.sourcesCsv), sourceRows2, ",");
|
|
24530
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.browsedOnlyCsv), browsedOnlyRows, ",");
|
|
24531
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.snippetsCsv), snippetRows, ",");
|
|
24532
|
+
writeTable((0, import_node_path11.join)(outputDir, relativePaths.domainsCsv), domainRows, ",");
|
|
24533
|
+
(0, import_node_fs8.writeFileSync)((0, import_node_path11.join)(outputDir, relativePaths.report), renderReportHtml(enriched));
|
|
24534
|
+
return relativePaths;
|
|
24462
24535
|
}
|
|
24463
24536
|
function esc(s) {
|
|
24464
24537
|
return String(s).replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """);
|
|
@@ -24575,6 +24648,7 @@ async function runFanoutCapture(page, input) {
|
|
|
24575
24648
|
const cap = new FanoutCdpCapture();
|
|
24576
24649
|
await cap.attach(page);
|
|
24577
24650
|
try {
|
|
24651
|
+
if (input.reset) cap.reset();
|
|
24578
24652
|
const baseline = cap.answerStreamCount();
|
|
24579
24653
|
if (input.prompt) await sendPrompt(page, input.prompt).catch(() => false);
|
|
24580
24654
|
const waitMs = input.wait_ms ?? (input.prompt ? 9e4 : 8e3);
|
|
@@ -25256,6 +25330,7 @@ function buildBrowserAgentRoutes() {
|
|
|
25256
25330
|
prompt: typeof body.prompt === "string" ? body.prompt : void 0,
|
|
25257
25331
|
wait_ms: typeof body.wait_ms === "number" ? body.wait_ms : void 0,
|
|
25258
25332
|
first_party_domain: typeof body.first_party_domain === "string" ? body.first_party_domain : void 0,
|
|
25333
|
+
reset: body.reset === true,
|
|
25259
25334
|
export: body.export === true
|
|
25260
25335
|
};
|
|
25261
25336
|
const t0 = Date.now();
|