mcp-scraper 0.2.12 → 0.2.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -2
- package/dist/bin/api-server.cjs +1836 -633
- package/dist/bin/api-server.cjs.map +1 -1
- package/dist/bin/api-server.js +3 -3
- package/dist/bin/browser-agent-stdio-server.cjs +1 -1
- package/dist/bin/browser-agent-stdio-server.cjs.map +1 -1
- package/dist/bin/browser-agent-stdio-server.js +2 -2
- package/dist/bin/mcp-scraper-cli.cjs +799 -12
- package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
- package/dist/bin/mcp-scraper-cli.js +90 -9
- package/dist/bin/mcp-scraper-cli.js.map +1 -1
- package/dist/bin/mcp-scraper-combined-stdio-server.cjs +47 -5
- package/dist/bin/mcp-scraper-combined-stdio-server.cjs.map +1 -1
- package/dist/bin/mcp-scraper-combined-stdio-server.js +3 -3
- package/dist/bin/mcp-scraper-install.cjs +4 -3
- package/dist/bin/mcp-scraper-install.cjs.map +1 -1
- package/dist/bin/mcp-scraper-install.js +4 -3
- package/dist/bin/mcp-scraper-install.js.map +1 -1
- package/dist/bin/mcp-stdio-server.cjs +47 -5
- package/dist/bin/mcp-stdio-server.cjs.map +1 -1
- package/dist/bin/mcp-stdio-server.js +2 -2
- package/dist/{chunk-ROS67BNV.js → chunk-BSYPATSM.js} +21 -2
- package/dist/chunk-BSYPATSM.js.map +1 -0
- package/dist/chunk-F44RBOJ5.js +1575 -0
- package/dist/chunk-F44RBOJ5.js.map +1 -0
- package/dist/{chunk-P5RYXUV7.js → chunk-IWQTNY2E.js} +48 -6
- package/dist/chunk-IWQTNY2E.js.map +1 -0
- package/dist/chunk-ROCXJOVY.js +7 -0
- package/dist/chunk-ROCXJOVY.js.map +1 -0
- package/dist/{chunk-7GCCOT3M.js → chunk-TL7YTFLH.js} +15 -1
- package/dist/chunk-TL7YTFLH.js.map +1 -0
- package/dist/{chunk-PB7ENA3G.js → chunk-ZV2XXYR7.js} +2 -2
- package/dist/{db-BVHYI57K.js → db-P5X6UQ3E.js} +2 -2
- package/dist/{server-I2H5WG5I.js → server-7QP6HQJJ.js} +556 -164
- package/dist/server-7QP6HQJJ.js.map +1 -0
- package/dist/{worker-TDJQ6TH3.js → worker-OZSWIS3F.js} +3 -3
- package/docs/specs/cli-agent-wiring-spec.md +4 -1
- package/docs/specs/seo-cli-growth-roadmap-spec.md +4 -2
- package/package.json +1 -1
- package/dist/chunk-7GCCOT3M.js.map +0 -1
- package/dist/chunk-L6IS63WS.js +0 -869
- package/dist/chunk-L6IS63WS.js.map +0 -1
- package/dist/chunk-LK27MYGD.js +0 -7
- package/dist/chunk-LK27MYGD.js.map +0 -1
- package/dist/chunk-P5RYXUV7.js.map +0 -1
- package/dist/chunk-ROS67BNV.js.map +0 -1
- package/dist/server-I2H5WG5I.js.map +0 -1
- /package/dist/{chunk-PB7ENA3G.js.map → chunk-ZV2XXYR7.js.map} +0 -0
- /package/dist/{db-BVHYI57K.js.map → db-P5X6UQ3E.js.map} +0 -0
- /package/dist/{worker-TDJQ6TH3.js.map → worker-OZSWIS3F.js.map} +0 -0
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
|
|
4
4
|
// src/cli/human-cli.ts
|
|
5
5
|
var import_commander = require("commander");
|
|
6
|
+
var import_node_child_process2 = require("child_process");
|
|
6
7
|
var import_promises3 = require("fs/promises");
|
|
7
8
|
var import_node_path3 = require("path");
|
|
8
9
|
|
|
9
10
|
// src/version.ts
|
|
10
|
-
var PACKAGE_VERSION = "0.2.
|
|
11
|
+
var PACKAGE_VERSION = "0.2.14";
|
|
11
12
|
|
|
12
13
|
// src/cli/agent-config.ts
|
|
13
14
|
function apiKeyValue(options) {
|
|
@@ -234,6 +235,34 @@ var AGENT_PROMPTS = {
|
|
|
234
235
|
"",
|
|
235
236
|
"Use directory_workflow when the user wants cities selected by population and Google Maps candidates per city. Keep query and location separate. Preserve result_position, source_location, review stars/count, categories, and profile URLs for downstream CSV or directory use."
|
|
236
237
|
].join("\n"),
|
|
238
|
+
"map-comparison": [
|
|
239
|
+
"# MCP Scraper Maps Comparison Prompt",
|
|
240
|
+
"",
|
|
241
|
+
"Run the map-comparison workflow when the user wants to compare local Maps competitors in a city or across selected markets. Use `maps-results.csv`, `map-comparison.csv`, and `profile-insights.csv` as source of truth.",
|
|
242
|
+
"",
|
|
243
|
+
"Ground recommendations in result_position, review_stars, review_count, category, website presence, review topics, and profile attributes. Treat review gaps and missing websites as opportunity signals, not guaranteed ranking factors."
|
|
244
|
+
].join("\n"),
|
|
245
|
+
"serp-comparison": [
|
|
246
|
+
"# MCP Scraper SERP Comparison Prompt",
|
|
247
|
+
"",
|
|
248
|
+
"Run the serp-comparison workflow when the user wants to know why competitors outrank a page or what the SERP rewards. Use organic results, extracted page headings, PAA questions, and AI Overview citations before making recommendations.",
|
|
249
|
+
"",
|
|
250
|
+
"Tie each recommendation to `content-gaps.csv`, `page-comparison.csv`, `paa-questions.csv`, or `ai-overview-citations.csv`. Do not invent missing sections or citations."
|
|
251
|
+
].join("\n"),
|
|
252
|
+
"paa-expansion-brief": [
|
|
253
|
+
"# MCP Scraper PAA Expansion Brief Prompt",
|
|
254
|
+
"",
|
|
255
|
+
"Run the paa-expansion-brief workflow when the user wants to figure out what to write from People Also Ask expansion. Use `section-map.csv` to structure the brief and `paa-questions.csv` for exact customer-language headings.",
|
|
256
|
+
"",
|
|
257
|
+
"Answer the highest-priority questions directly, preserve source URLs when present, and separate evidence-backed sections from assumptions."
|
|
258
|
+
].join("\n"),
|
|
259
|
+
"ai-overview-language": [
|
|
260
|
+
"# MCP Scraper AI Overview Language Prompt",
|
|
261
|
+
"",
|
|
262
|
+
"Run the ai-overview-language workflow when the user wants to know how to phrase content for AI Overview inclusion. Use `claim-patterns.csv`, `language-guidance.csv`, `ai-overview-citations.csv`, and PAA follow-ups as evidence.",
|
|
263
|
+
"",
|
|
264
|
+
"Recommend concise answer blocks, criteria/step language, citation hooks, and follow-up sections. Do not claim the target will be cited; frame the output as evidence-based language guidance."
|
|
265
|
+
].join("\n"),
|
|
237
266
|
"ai-citation-monitor": [
|
|
238
267
|
"# MCP Scraper AI Citation Monitor Prompt",
|
|
239
268
|
"",
|
|
@@ -412,22 +441,25 @@ async function openWorkflowReport(id, outputDir) {
|
|
|
412
441
|
}
|
|
413
442
|
|
|
414
443
|
// src/workflows/registry.ts
|
|
415
|
-
var
|
|
444
|
+
var import_zod5 = require("zod");
|
|
416
445
|
|
|
417
446
|
// src/workflows/http-client.ts
|
|
418
447
|
var WorkflowHttpClient = class {
|
|
419
|
-
constructor(apiUrl, apiKey, fetchImpl = fetch) {
|
|
448
|
+
constructor(apiUrl, apiKey, fetchImpl = fetch, extraHeaders = {}) {
|
|
420
449
|
this.apiUrl = apiUrl;
|
|
421
450
|
this.apiKey = apiKey;
|
|
422
451
|
this.fetchImpl = fetchImpl;
|
|
452
|
+
this.extraHeaders = extraHeaders;
|
|
423
453
|
}
|
|
424
454
|
apiUrl;
|
|
425
455
|
apiKey;
|
|
426
456
|
fetchImpl;
|
|
457
|
+
extraHeaders;
|
|
427
458
|
async post(path, body, timeoutMs = 18e4) {
|
|
428
459
|
const res = await this.fetchImpl(`${this.apiUrl.replace(/\/$/, "")}${path}`, {
|
|
429
460
|
method: "POST",
|
|
430
461
|
headers: {
|
|
462
|
+
...this.extraHeaders,
|
|
431
463
|
"Content-Type": "application/json",
|
|
432
464
|
"x-api-key": this.apiKey
|
|
433
465
|
},
|
|
@@ -1020,11 +1052,714 @@ ${summary}
|
|
|
1020
1052
|
}
|
|
1021
1053
|
};
|
|
1022
1054
|
|
|
1055
|
+
// src/workflows/workflows/comparison-briefs.ts
|
|
1056
|
+
var import_zod4 = require("zod");
|
|
1057
|
+
|
|
1058
|
+
// src/workflows/workflows/seo-workflow-utils.ts
|
|
1059
|
+
var STOP_WORDS = /* @__PURE__ */ new Set([
|
|
1060
|
+
"about",
|
|
1061
|
+
"after",
|
|
1062
|
+
"also",
|
|
1063
|
+
"because",
|
|
1064
|
+
"been",
|
|
1065
|
+
"best",
|
|
1066
|
+
"both",
|
|
1067
|
+
"from",
|
|
1068
|
+
"have",
|
|
1069
|
+
"into",
|
|
1070
|
+
"more",
|
|
1071
|
+
"most",
|
|
1072
|
+
"near",
|
|
1073
|
+
"only",
|
|
1074
|
+
"over",
|
|
1075
|
+
"than",
|
|
1076
|
+
"that",
|
|
1077
|
+
"their",
|
|
1078
|
+
"them",
|
|
1079
|
+
"then",
|
|
1080
|
+
"there",
|
|
1081
|
+
"these",
|
|
1082
|
+
"they",
|
|
1083
|
+
"this",
|
|
1084
|
+
"what",
|
|
1085
|
+
"when",
|
|
1086
|
+
"where",
|
|
1087
|
+
"which",
|
|
1088
|
+
"while",
|
|
1089
|
+
"with",
|
|
1090
|
+
"your",
|
|
1091
|
+
"will",
|
|
1092
|
+
"would",
|
|
1093
|
+
"should",
|
|
1094
|
+
"could",
|
|
1095
|
+
"does",
|
|
1096
|
+
"were",
|
|
1097
|
+
"cost",
|
|
1098
|
+
"costs"
|
|
1099
|
+
]);
|
|
1100
|
+
function normalizeDomain2(value) {
|
|
1101
|
+
if (!value) return null;
|
|
1102
|
+
try {
|
|
1103
|
+
const url = new URL(value.includes("://") ? value : `https://${value}`);
|
|
1104
|
+
return url.hostname.replace(/^www\./, "").toLowerCase();
|
|
1105
|
+
} catch {
|
|
1106
|
+
return value.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0]?.toLowerCase() || null;
|
|
1107
|
+
}
|
|
1108
|
+
}
|
|
1109
|
+
function domainFromUrl2(url) {
|
|
1110
|
+
return normalizeDomain2(url) ?? "";
|
|
1111
|
+
}
|
|
1112
|
+
function numberFrom2(value) {
|
|
1113
|
+
if (value === null || value === void 0 || value === "") return null;
|
|
1114
|
+
const parsed = Number(String(value).replace(/[^\d.]/g, ""));
|
|
1115
|
+
return Number.isFinite(parsed) ? parsed : null;
|
|
1116
|
+
}
|
|
1117
|
+
function median2(values) {
|
|
1118
|
+
const nums = values.filter((v) => v !== null).sort((a, b) => a - b);
|
|
1119
|
+
if (!nums.length) return null;
|
|
1120
|
+
return nums[Math.floor(nums.length / 2)] ?? null;
|
|
1121
|
+
}
|
|
1122
|
+
function textTerms(text, limit = 12) {
|
|
1123
|
+
const counts = /* @__PURE__ */ new Map();
|
|
1124
|
+
for (const token of text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/)) {
|
|
1125
|
+
if (token.length < 4 || STOP_WORDS.has(token)) continue;
|
|
1126
|
+
counts.set(token, (counts.get(token) ?? 0) + 1);
|
|
1127
|
+
}
|
|
1128
|
+
return [...counts.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, limit).map(([term, count]) => `${term} (${count})`).join("; ");
|
|
1129
|
+
}
|
|
1130
|
+
function classifyQuestion(question) {
|
|
1131
|
+
const q = question.toLowerCase();
|
|
1132
|
+
if (/\b(cost|price|pricing|charge|expensive|cheap)\b/.test(q)) return "cost";
|
|
1133
|
+
if (/\b(best|top|recommended|reviews?|compare|versus|vs)\b/.test(q)) return "comparison";
|
|
1134
|
+
if (/\b(how|steps?|process|way to)\b/.test(q)) return "process";
|
|
1135
|
+
if (/\b(why|worth|important|benefit)\b/.test(q)) return "why";
|
|
1136
|
+
if (/\b(near me|city|local|nearby)\b/.test(q)) return "local";
|
|
1137
|
+
if (/\b(can|does|do|is|are|should|will)\b/.test(q)) return "decision";
|
|
1138
|
+
return "definition";
|
|
1139
|
+
}
|
|
1140
|
+
function questionRows(paa) {
|
|
1141
|
+
return (paa?.flat ?? []).filter((row) => row.question).map((row, index) => {
|
|
1142
|
+
const url = row.source_cite ?? "";
|
|
1143
|
+
const domain = normalizeDomain2(row.source_site ?? "") ?? domainFromUrl2(url);
|
|
1144
|
+
return {
|
|
1145
|
+
position: index + 1,
|
|
1146
|
+
intent: classifyQuestion(row.question ?? ""),
|
|
1147
|
+
question: row.question ?? "",
|
|
1148
|
+
answer_excerpt: (row.answer ?? "").slice(0, 320),
|
|
1149
|
+
source_title: row.source_title ?? "",
|
|
1150
|
+
source_domain: domain,
|
|
1151
|
+
source_url: url
|
|
1152
|
+
};
|
|
1153
|
+
});
|
|
1154
|
+
}
|
|
1155
|
+
function sourceDomainRows(rows) {
|
|
1156
|
+
const byDomain = /* @__PURE__ */ new Map();
|
|
1157
|
+
for (const row of rows) {
|
|
1158
|
+
const domain = String(row.source_domain ?? row.domain ?? "");
|
|
1159
|
+
if (!domain) continue;
|
|
1160
|
+
const entry = byDomain.get(domain) ?? { questions: 0, urls: /* @__PURE__ */ new Set(), intents: /* @__PURE__ */ new Map() };
|
|
1161
|
+
entry.questions += row.question ? 1 : 0;
|
|
1162
|
+
if (row.source_url || row.url) entry.urls.add(String(row.source_url ?? row.url));
|
|
1163
|
+
if (row.intent) entry.intents.set(String(row.intent), (entry.intents.get(String(row.intent)) ?? 0) + 1);
|
|
1164
|
+
byDomain.set(domain, entry);
|
|
1165
|
+
}
|
|
1166
|
+
return [...byDomain.entries()].map(([domain, entry]) => ({
|
|
1167
|
+
domain,
|
|
1168
|
+
question_mentions: entry.questions,
|
|
1169
|
+
source_url_count: entry.urls.size,
|
|
1170
|
+
top_intents: [...entry.intents.entries()].sort((a, b) => b[1] - a[1]).map(([intent, count]) => `${intent} (${count})`).join("; ")
|
|
1171
|
+
})).sort((a, b) => Number(b.question_mentions) - Number(a.question_mentions) || String(a.domain).localeCompare(String(b.domain)));
|
|
1172
|
+
}
|
|
1173
|
+
function splitSentences(text) {
|
|
1174
|
+
return (text ?? "").replace(/\s+/g, " ").split(/(?<=[.!?])\s+/).map((sentence) => sentence.trim()).filter((sentence) => sentence.length > 20).slice(0, 20);
|
|
1175
|
+
}
|
|
1176
|
+
function classifySentence(sentence) {
|
|
1177
|
+
const s = sentence.toLowerCase();
|
|
1178
|
+
if (/\bis\b|\bare\b|\bmeans\b|\brefers to\b/.test(s)) return "definition";
|
|
1179
|
+
if (/\binclude\b|\bconsider\b|\bfactors?\b|\bcriteria\b/.test(s)) return "criteria";
|
|
1180
|
+
if (/\bfirst\b|\bthen\b|\bsteps?\b|\bprocess\b/.test(s)) return "process";
|
|
1181
|
+
if (/\bvs\b|\bthan\b|\bcompare\b|\bdifference\b/.test(s)) return "comparison";
|
|
1182
|
+
if (/\bcost\b|\bprice\b|\baverage\b|\brange\b/.test(s)) return "cost";
|
|
1183
|
+
return "claim";
|
|
1184
|
+
}
|
|
1185
|
+
function pageSummaryRow(page, source) {
|
|
1186
|
+
return {
|
|
1187
|
+
position: source.position ?? "",
|
|
1188
|
+
domain: source.domain,
|
|
1189
|
+
url: source.url,
|
|
1190
|
+
serp_title: source.title ?? "",
|
|
1191
|
+
page_title: page.title ?? "",
|
|
1192
|
+
h1: page.h1 ?? "",
|
|
1193
|
+
meta_description: page.metaDescription ?? "",
|
|
1194
|
+
word_count: page.wordCount ?? "",
|
|
1195
|
+
heading_count: page.headings?.length ?? 0,
|
|
1196
|
+
schema_types: (page.schemaTypes ?? []).join("; ")
|
|
1197
|
+
};
|
|
1198
|
+
}
|
|
1199
|
+
async function mapLimit2(items, limit, fn) {
|
|
1200
|
+
const out = new Array(items.length);
|
|
1201
|
+
let next = 0;
|
|
1202
|
+
async function worker() {
|
|
1203
|
+
while (next < items.length) {
|
|
1204
|
+
const index = next;
|
|
1205
|
+
next += 1;
|
|
1206
|
+
out[index] = await fn(items[index], index);
|
|
1207
|
+
}
|
|
1208
|
+
}
|
|
1209
|
+
await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, () => worker()));
|
|
1210
|
+
return out;
|
|
1211
|
+
}
|
|
1212
|
+
|
|
1213
|
+
// src/workflows/workflows/comparison-briefs.ts
|
|
1214
|
+
var ProxyModeSchema = import_zod4.z.enum(["location", "configured", "none"]);
|
|
1215
|
+
var MapComparisonInputSchema = import_zod4.z.object({
|
|
1216
|
+
query: import_zod4.z.string().min(1),
|
|
1217
|
+
location: import_zod4.z.string().optional(),
|
|
1218
|
+
state: import_zod4.z.string().optional(),
|
|
1219
|
+
minPopulation: import_zod4.z.number().int().min(0).default(1e5),
|
|
1220
|
+
maxCities: import_zod4.z.number().int().min(1).max(100).default(5),
|
|
1221
|
+
maxResultsPerCity: import_zod4.z.number().int().min(1).max(50).default(20),
|
|
1222
|
+
hydrateTop: import_zod4.z.number().int().min(0).max(10).default(5),
|
|
1223
|
+
maxReviews: import_zod4.z.number().int().min(0).max(500).default(25),
|
|
1224
|
+
concurrency: import_zod4.z.number().int().min(1).max(5).default(5),
|
|
1225
|
+
proxyMode: ProxyModeSchema.default("location"),
|
|
1226
|
+
returnPartial: import_zod4.z.boolean().default(true)
|
|
1227
|
+
}).refine((input) => input.location || input.state, {
|
|
1228
|
+
message: "Either location or state is required for map-comparison"
|
|
1229
|
+
});
|
|
1230
|
+
var SerpComparisonInputSchema = import_zod4.z.object({
|
|
1231
|
+
keyword: import_zod4.z.string().min(1),
|
|
1232
|
+
domain: import_zod4.z.string().optional(),
|
|
1233
|
+
url: import_zod4.z.string().url().optional(),
|
|
1234
|
+
location: import_zod4.z.string().optional(),
|
|
1235
|
+
maxResults: import_zod4.z.number().int().min(1).max(20).default(10),
|
|
1236
|
+
maxQuestions: import_zod4.z.number().int().min(1).max(200).default(40),
|
|
1237
|
+
extractTop: import_zod4.z.number().int().min(0).max(10).default(5),
|
|
1238
|
+
includePaa: import_zod4.z.boolean().default(true),
|
|
1239
|
+
includeAiOverview: import_zod4.z.boolean().default(true),
|
|
1240
|
+
returnPartial: import_zod4.z.boolean().default(true)
|
|
1241
|
+
});
|
|
1242
|
+
var PaaExpansionBriefInputSchema = import_zod4.z.object({
|
|
1243
|
+
keyword: import_zod4.z.string().min(1),
|
|
1244
|
+
location: import_zod4.z.string().optional(),
|
|
1245
|
+
maxQuestions: import_zod4.z.number().int().min(1).max(300).default(80),
|
|
1246
|
+
depth: import_zod4.z.number().int().min(1).max(6).default(3),
|
|
1247
|
+
returnPartial: import_zod4.z.boolean().default(true)
|
|
1248
|
+
});
|
|
1249
|
+
var AiOverviewLanguageInputSchema = import_zod4.z.object({
|
|
1250
|
+
keyword: import_zod4.z.string().min(1),
|
|
1251
|
+
domain: import_zod4.z.string().optional(),
|
|
1252
|
+
url: import_zod4.z.string().url().optional(),
|
|
1253
|
+
location: import_zod4.z.string().optional(),
|
|
1254
|
+
maxQuestions: import_zod4.z.number().int().min(1).max(200).default(40),
|
|
1255
|
+
extractTop: import_zod4.z.number().int().min(0).max(8).default(3),
|
|
1256
|
+
returnPartial: import_zod4.z.boolean().default(true)
|
|
1257
|
+
});
|
|
1258
|
+
function businessRowsFromMaps(location, query, results) {
|
|
1259
|
+
return results.map((result) => ({
|
|
1260
|
+
source_query: query,
|
|
1261
|
+
source_location: location,
|
|
1262
|
+
city: location.split(",")[0]?.trim() ?? location,
|
|
1263
|
+
state: location.split(",")[1]?.trim() ?? "",
|
|
1264
|
+
population: "",
|
|
1265
|
+
result_position: result.position,
|
|
1266
|
+
business_name: result.name,
|
|
1267
|
+
review_stars: result.rating ?? "",
|
|
1268
|
+
review_count: result.reviewCount ?? "",
|
|
1269
|
+
category: result.category ?? "",
|
|
1270
|
+
address: result.address ?? "",
|
|
1271
|
+
phone: result.phone ?? "",
|
|
1272
|
+
website_url: result.websiteUrl ?? "",
|
|
1273
|
+
place_url: result.placeUrl ?? "",
|
|
1274
|
+
cid: result.cid ?? "",
|
|
1275
|
+
cid_decimal: result.cidDecimal ?? "",
|
|
1276
|
+
result_status: "ok",
|
|
1277
|
+
error: ""
|
|
1278
|
+
}));
|
|
1279
|
+
}
|
|
1280
|
+
function marketRows(rows) {
|
|
1281
|
+
const byLocation = /* @__PURE__ */ new Map();
|
|
1282
|
+
for (const row of rows) {
|
|
1283
|
+
const key = String(row.source_location ?? row.location ?? "");
|
|
1284
|
+
if (!key) continue;
|
|
1285
|
+
const list = byLocation.get(key) ?? [];
|
|
1286
|
+
list.push(row);
|
|
1287
|
+
byLocation.set(key, list);
|
|
1288
|
+
}
|
|
1289
|
+
return [...byLocation.entries()].map(([location, list]) => {
|
|
1290
|
+
const counts = list.map((row) => numberFrom2(row.review_count));
|
|
1291
|
+
const ratings = list.map((row) => numberFrom2(row.review_stars));
|
|
1292
|
+
const topThree = list.slice(0, 3).map((row) => numberFrom2(row.review_count)).filter((v) => v !== null);
|
|
1293
|
+
const categories = /* @__PURE__ */ new Map();
|
|
1294
|
+
for (const row of list) {
|
|
1295
|
+
const category = String(row.category ?? "");
|
|
1296
|
+
if (category) categories.set(category, (categories.get(category) ?? 0) + 1);
|
|
1297
|
+
}
|
|
1298
|
+
return {
|
|
1299
|
+
source_location: location,
|
|
1300
|
+
result_count: list.filter((row) => row.business_name).length,
|
|
1301
|
+
median_review_count: median2(counts) ?? "",
|
|
1302
|
+
median_rating: median2(ratings) ?? "",
|
|
1303
|
+
top_three_average_review_count: topThree.length ? Math.round(topThree.reduce((a, b) => a + b, 0) / topThree.length) : "",
|
|
1304
|
+
top_categories: [...categories.entries()].sort((a, b) => b[1] - a[1]).slice(0, 5).map(([cat, count]) => `${cat} (${count})`).join("; "),
|
|
1305
|
+
websites_present: list.filter((row) => row.website_url).length
|
|
1306
|
+
};
|
|
1307
|
+
});
|
|
1308
|
+
}
|
|
1309
|
+
function comparisonRows(rows) {
|
|
1310
|
+
const benchmarkByLocation = /* @__PURE__ */ new Map();
|
|
1311
|
+
for (const market of marketRows(rows)) {
|
|
1312
|
+
benchmarkByLocation.set(String(market.source_location), numberFrom2(market.top_three_average_review_count) ?? 0);
|
|
1313
|
+
}
|
|
1314
|
+
return rows.filter((row) => row.business_name).map((row) => {
|
|
1315
|
+
const reviews = numberFrom2(row.review_count) ?? 0;
|
|
1316
|
+
const benchmark = benchmarkByLocation.get(String(row.source_location)) ?? 0;
|
|
1317
|
+
const websiteMissing = !row.website_url;
|
|
1318
|
+
const rank = numberFrom2(row.result_position) ?? 999;
|
|
1319
|
+
return {
|
|
1320
|
+
source_location: row.source_location,
|
|
1321
|
+
result_position: row.result_position,
|
|
1322
|
+
business_name: row.business_name,
|
|
1323
|
+
category: row.category,
|
|
1324
|
+
review_stars: row.review_stars,
|
|
1325
|
+
review_count: row.review_count,
|
|
1326
|
+
review_gap_to_top3_average: benchmark ? Math.max(0, benchmark - reviews) : "",
|
|
1327
|
+
website_url: row.website_url,
|
|
1328
|
+
place_url: row.place_url,
|
|
1329
|
+
comparison_note: rank <= 3 ? "visible leader" : websiteMissing ? "ranking without website" : reviews < benchmark ? "review-light competitor" : "visible competitor"
|
|
1330
|
+
};
|
|
1331
|
+
});
|
|
1332
|
+
}
|
|
1333
|
+
function organicRows(serp, targetDomain) {
|
|
1334
|
+
return (serp?.organicResults ?? []).map((result) => {
|
|
1335
|
+
const url = result.url ?? "";
|
|
1336
|
+
const domain = normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(url);
|
|
1337
|
+
return {
|
|
1338
|
+
position: result.position ?? "",
|
|
1339
|
+
title: result.title ?? "",
|
|
1340
|
+
url,
|
|
1341
|
+
domain,
|
|
1342
|
+
snippet: result.snippet ?? "",
|
|
1343
|
+
is_target: targetDomain ? domain === targetDomain : false
|
|
1344
|
+
};
|
|
1345
|
+
});
|
|
1346
|
+
}
|
|
1347
|
+
function pageGapRows(targetPage, competitorPages) {
|
|
1348
|
+
const targetHeadingText = new Set((targetPage?.headings ?? []).map((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim()));
|
|
1349
|
+
const targetTerms = new Set((targetPage?.headings ?? []).flatMap((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/).filter(Boolean)));
|
|
1350
|
+
const rows = [];
|
|
1351
|
+
for (const { source, page } of competitorPages) {
|
|
1352
|
+
for (const heading of page.headings ?? []) {
|
|
1353
|
+
if (heading.level > 3 || !heading.text) continue;
|
|
1354
|
+
const normalized = heading.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim();
|
|
1355
|
+
const terms = normalized.split(/\s+/).filter((term) => term.length > 3);
|
|
1356
|
+
const overlap = terms.filter((term) => targetTerms.has(term)).length;
|
|
1357
|
+
const covered = targetHeadingText.has(normalized) || overlap >= Math.max(2, Math.ceil(terms.length / 2));
|
|
1358
|
+
if (covered && targetPage) continue;
|
|
1359
|
+
rows.push({
|
|
1360
|
+
source_position: source.position ?? "",
|
|
1361
|
+
source_domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url),
|
|
1362
|
+
source_url: source.url ?? "",
|
|
1363
|
+
heading_level: heading.level,
|
|
1364
|
+
competitor_heading: heading.text,
|
|
1365
|
+
target_coverage: targetPage ? "not found in target headings" : "no target page extracted",
|
|
1366
|
+
terms: textTerms(heading.text, 6)
|
|
1367
|
+
});
|
|
1368
|
+
}
|
|
1369
|
+
}
|
|
1370
|
+
return rows.slice(0, 150);
|
|
1371
|
+
}
|
|
1372
|
+
async function extractPages(ctx, sources, warnings, labelPrefix) {
|
|
1373
|
+
return mapLimit2(sources.filter((source) => source.url), 2, async (source, index) => {
|
|
1374
|
+
try {
|
|
1375
|
+
const page = await ctx.client.post("/extract-url", { url: source.url }, 18e4);
|
|
1376
|
+
await ctx.artifacts.writeJson(`${labelPrefix} ${index + 1}`, `raw/extract-url/${labelPrefix.toLowerCase()}-${index + 1}.json`, page);
|
|
1377
|
+
return { source, page };
|
|
1378
|
+
} catch (err) {
|
|
1379
|
+
warnings.push(`Page extraction failed for ${source.url}: ${err instanceof Error ? err.message : String(err)}`);
|
|
1380
|
+
return null;
|
|
1381
|
+
}
|
|
1382
|
+
}).then((items) => items.filter((item) => item !== null));
|
|
1383
|
+
}
|
|
1384
|
+
var mapComparisonWorkflowDefinition = {
|
|
1385
|
+
id: "map-comparison",
|
|
1386
|
+
title: "Maps Comparison",
|
|
1387
|
+
description: "Compare Google Maps competitors by rank, reviews, stars, categories, websites, and profile/review signals.",
|
|
1388
|
+
inputSchema: MapComparisonInputSchema,
|
|
1389
|
+
async run(input, ctx) {
|
|
1390
|
+
await ctx.artifacts.writeManifest("running", {}, [], []);
|
|
1391
|
+
const warnings = [];
|
|
1392
|
+
let rows;
|
|
1393
|
+
let directory = null;
|
|
1394
|
+
let mapsSearch = null;
|
|
1395
|
+
if (input.location) {
|
|
1396
|
+
mapsSearch = await ctx.client.post("/maps/search", {
|
|
1397
|
+
query: input.query,
|
|
1398
|
+
location: input.location,
|
|
1399
|
+
maxResults: input.maxResultsPerCity,
|
|
1400
|
+
proxyMode: input.proxyMode
|
|
1401
|
+
}, 24e4);
|
|
1402
|
+
rows = businessRowsFromMaps(input.location, input.query, mapsSearch.results);
|
|
1403
|
+
await ctx.artifacts.writeJson("Maps search raw JSON", "raw/maps-search.json", mapsSearch);
|
|
1404
|
+
} else {
|
|
1405
|
+
directory = await ctx.client.post("/directory/run", {
|
|
1406
|
+
query: input.query,
|
|
1407
|
+
state: input.state,
|
|
1408
|
+
minPopulation: input.minPopulation,
|
|
1409
|
+
maxCities: input.maxCities,
|
|
1410
|
+
maxResultsPerCity: input.maxResultsPerCity,
|
|
1411
|
+
concurrency: input.concurrency,
|
|
1412
|
+
proxyMode: input.proxyMode,
|
|
1413
|
+
saveCsv: true
|
|
1414
|
+
}, 9e5);
|
|
1415
|
+
rows = directoryRows(directory);
|
|
1416
|
+
await ctx.artifacts.writeJson("Directory raw JSON", "raw/directory-workflow.json", directory);
|
|
1417
|
+
await ctx.artifacts.writeCsv("Directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, rows);
|
|
1418
|
+
warnings.push(...directory.warnings);
|
|
1419
|
+
}
|
|
1420
|
+
const compareRows = comparisonRows(rows);
|
|
1421
|
+
const selected = compareRows.slice(0, input.hydrateTop * Math.max(1, input.location ? 1 : input.maxCities));
|
|
1422
|
+
const hydrated = await mapLimit2(selected, 3, async (row, index) => {
|
|
1423
|
+
try {
|
|
1424
|
+
const detail = await ctx.client.post("/maps/place", {
|
|
1425
|
+
businessName: row.business_name,
|
|
1426
|
+
location: row.source_location,
|
|
1427
|
+
includeReviews: input.maxReviews > 0,
|
|
1428
|
+
maxReviews: Math.max(1, input.maxReviews)
|
|
1429
|
+
}, 18e4);
|
|
1430
|
+
await ctx.artifacts.writeJson(`${row.business_name} profile`, `raw/maps-place-intel/${index + 1}-${String(row.business_name).toLowerCase().replace(/[^a-z0-9]+/g, "-")}.json`, detail);
|
|
1431
|
+
return { row, detail, error: "" };
|
|
1432
|
+
} catch (err) {
|
|
1433
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
1434
|
+
warnings.push(`Profile hydration failed for ${row.business_name}: ${message}`);
|
|
1435
|
+
return { row, detail: null, error: message };
|
|
1436
|
+
}
|
|
1437
|
+
});
|
|
1438
|
+
const profileRows = hydrated.map(({ row, detail, error }) => ({
|
|
1439
|
+
source_location: row.source_location,
|
|
1440
|
+
result_position: row.result_position,
|
|
1441
|
+
business_name: row.business_name,
|
|
1442
|
+
category: detail?.category ?? row.category,
|
|
1443
|
+
review_stars: detail?.rating ?? row.review_stars,
|
|
1444
|
+
review_count: detail?.reviewCount ?? row.review_count,
|
|
1445
|
+
website_url: detail?.website ?? row.website_url,
|
|
1446
|
+
review_topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
|
|
1447
|
+
about_attributes: (detail?.aboutAttributes ?? []).map((a) => `${a.section}: ${a.attribute}`).join("; "),
|
|
1448
|
+
reviews_status: detail?.reviewsStatus ?? "",
|
|
1449
|
+
error
|
|
1450
|
+
}));
|
|
1451
|
+
const markets = marketRows(rows);
|
|
1452
|
+
await ctx.artifacts.writeCsv("Maps results CSV", "maps-results.csv", ["source_query", "source_location", "city", "state", "population", "result_position", "business_name", "review_stars", "review_count", "category", "address", "phone", "website_url", "place_url", "cid", "cid_decimal", "result_status", "error"], rows);
|
|
1453
|
+
await ctx.artifacts.writeCsv("Comparison CSV", "map-comparison.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "website_url", "place_url", "comparison_note"], compareRows);
|
|
1454
|
+
await ctx.artifacts.writeCsv("Profile insights CSV", "profile-insights.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "website_url", "review_topics", "about_attributes", "reviews_status", "error"], profileRows);
|
|
1455
|
+
await ctx.artifacts.writeJson("Maps comparison evidence", "evidence.json", { input, directory, mapsSearch, rows, compareRows, profileRows, markets, warnings });
|
|
1456
|
+
await ctx.artifacts.writeText("Brief", "brief.md", [
|
|
1457
|
+
`# Maps Comparison: ${input.query}`,
|
|
1458
|
+
"",
|
|
1459
|
+
`Markets: ${markets.map((row) => row.source_location).join(", ")}`,
|
|
1460
|
+
"",
|
|
1461
|
+
"## How to Use",
|
|
1462
|
+
"- Compare rank position against review count and category patterns.",
|
|
1463
|
+
"- Treat review gaps and missing websites as opportunity signals, not guarantees.",
|
|
1464
|
+
"- Use profile topics and attributes as evidence for local content and GBP improvements."
|
|
1465
|
+
].join("\n"));
|
|
1466
|
+
const summary = `${compareRows.length} Maps competitors compared across ${markets.length} market(s); ${profileRows.filter((row) => !row.error).length} profiles hydrated.`;
|
|
1467
|
+
const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
|
|
1468
|
+
title: "Maps Comparison",
|
|
1469
|
+
subtitle: `${input.query}${input.location ? ` \xB7 ${input.location}` : input.state ? ` \xB7 ${input.state}` : ""}`,
|
|
1470
|
+
summary,
|
|
1471
|
+
warnings,
|
|
1472
|
+
tables: [
|
|
1473
|
+
{ title: "Market Benchmarks", columns: ["source_location", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "websites_present"], rows: markets },
|
|
1474
|
+
{ title: "Competitor Comparison", columns: ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "comparison_note"], rows: compareRows.slice(0, 120) },
|
|
1475
|
+
{ title: "Profile Insights", columns: ["source_location", "business_name", "review_topics", "about_attributes", "error"], rows: profileRows }
|
|
1476
|
+
]
|
|
1477
|
+
}));
|
|
1478
|
+
const status2 = warnings.length ? "partial" : "succeeded";
|
|
1479
|
+
const counts = { markets: markets.length, competitors: compareRows.length, hydratedProfiles: profileRows.filter((row) => !row.error).length };
|
|
1480
|
+
await ctx.artifacts.writeManifest(status2, counts, warnings, []);
|
|
1481
|
+
return { title: "Maps Comparison", summary, status: status2, counts, warnings, errors: [], reportPath };
|
|
1482
|
+
}
|
|
1483
|
+
};
|
|
1484
|
+
var serpComparisonWorkflowDefinition = {
|
|
1485
|
+
id: "serp-comparison",
|
|
1486
|
+
title: "SERP Comparison",
|
|
1487
|
+
description: "Compare ranking pages, SERP features, PAA evidence, AI Overview citations, and page-level content gaps.",
|
|
1488
|
+
inputSchema: SerpComparisonInputSchema,
|
|
1489
|
+
async run(input, ctx) {
|
|
1490
|
+
await ctx.artifacts.writeManifest("running", {}, [], []);
|
|
1491
|
+
const warnings = [];
|
|
1492
|
+
const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
|
|
1493
|
+
const serp = await ctx.client.post("/harvest/sync", {
|
|
1494
|
+
query: input.keyword,
|
|
1495
|
+
location: input.location,
|
|
1496
|
+
serpOnly: true,
|
|
1497
|
+
maxQuestions: 1,
|
|
1498
|
+
format: "json"
|
|
1499
|
+
}, 18e4);
|
|
1500
|
+
await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
|
|
1501
|
+
let paa = null;
|
|
1502
|
+
if (input.includePaa) {
|
|
1503
|
+
try {
|
|
1504
|
+
paa = await ctx.client.post("/harvest/sync", {
|
|
1505
|
+
query: input.keyword,
|
|
1506
|
+
location: input.location,
|
|
1507
|
+
maxQuestions: input.maxQuestions,
|
|
1508
|
+
format: "json"
|
|
1509
|
+
}, 28e4);
|
|
1510
|
+
await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
|
|
1511
|
+
} catch (err) {
|
|
1512
|
+
warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
|
|
1513
|
+
}
|
|
1514
|
+
}
|
|
1515
|
+
const organic = organicRows(serp, targetDomain).slice(0, input.maxResults);
|
|
1516
|
+
const organicSources = (serp.organicResults ?? []).slice(0, input.maxResults);
|
|
1517
|
+
const targetSource = input.url ? { position: 0, title: "Target page", url: input.url, domain: domainFromUrl2(input.url) } : organicSources.find((result) => targetDomain && (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) === targetDomain);
|
|
1518
|
+
const competitorSources = organicSources.filter((result) => result.url && result.url !== targetSource?.url).filter((result) => !targetDomain || (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) !== targetDomain).slice(0, input.extractTop);
|
|
1519
|
+
const targetPages = targetSource ? await extractPages(ctx, [targetSource], warnings, "Target") : [];
|
|
1520
|
+
const competitorPages = input.extractTop > 0 ? await extractPages(ctx, competitorSources, warnings, "Competitor") : [];
|
|
1521
|
+
const targetPage = targetPages[0]?.page ?? null;
|
|
1522
|
+
const pageRows = [
|
|
1523
|
+
...targetPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title })),
|
|
1524
|
+
...competitorPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }))
|
|
1525
|
+
];
|
|
1526
|
+
const gaps = pageGapRows(targetPage, competitorPages);
|
|
1527
|
+
const questions = questionRows(paa);
|
|
1528
|
+
const aiRows = (serp.aiOverview?.citations ?? []).map((citation, index) => ({
|
|
1529
|
+
citation_position: index + 1,
|
|
1530
|
+
citation_text: citation.text ?? "",
|
|
1531
|
+
url: citation.href ?? "",
|
|
1532
|
+
domain: domainFromUrl2(citation.href ?? ""),
|
|
1533
|
+
is_target: targetDomain ? domainFromUrl2(citation.href ?? "") === targetDomain : false
|
|
1534
|
+
}));
|
|
1535
|
+
await ctx.artifacts.writeCsv("Organic results CSV", "organic-results.csv", ["position", "title", "url", "domain", "snippet", "is_target"], organic);
|
|
1536
|
+
await ctx.artifacts.writeCsv("Page comparison CSV", "page-comparison.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], pageRows);
|
|
1537
|
+
await ctx.artifacts.writeCsv("Content gaps CSV", "content-gaps.csv", ["source_position", "source_domain", "source_url", "heading_level", "competitor_heading", "target_coverage", "terms"], gaps);
|
|
1538
|
+
await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
|
|
1539
|
+
await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], aiRows);
|
|
1540
|
+
await ctx.artifacts.writeJson("SERP comparison evidence", "evidence.json", { input, serp, paa, organic, pageRows, gaps, questions, aiRows, warnings });
|
|
1541
|
+
await ctx.artifacts.writeText("Writer brief", "brief.md", [
|
|
1542
|
+
`# SERP Comparison Brief: ${input.keyword}`,
|
|
1543
|
+
"",
|
|
1544
|
+
`Target: ${targetDomain ?? input.url ?? "not specified"}`,
|
|
1545
|
+
`Location: ${input.location ?? "not specified"}`,
|
|
1546
|
+
"",
|
|
1547
|
+
"## Recommended Actions",
|
|
1548
|
+
"- Use `content-gaps.csv` to decide which missing sections deserve coverage.",
|
|
1549
|
+
"- Use `paa-questions.csv` for FAQ and answer-block candidates.",
|
|
1550
|
+
"- Use `ai-overview-citations.csv` to see whether the target is cited in AI Overview evidence.",
|
|
1551
|
+
"- Treat extracted page headings as evidence, not a complete semantic analysis."
|
|
1552
|
+
].join("\n"));
|
|
1553
|
+
const summary = `${organic.length} organic results, ${pageRows.length} extracted pages, ${gaps.length} heading gaps, ${questions.length} PAA questions.`;
|
|
1554
|
+
const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
|
|
1555
|
+
title: "SERP Comparison",
|
|
1556
|
+
subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
|
|
1557
|
+
summary,
|
|
1558
|
+
warnings,
|
|
1559
|
+
tables: [
|
|
1560
|
+
{ title: "Organic Results", columns: ["position", "title", "domain", "is_target"], rows: organic },
|
|
1561
|
+
{ title: "Page Comparison", columns: ["position", "domain", "h1", "word_count", "heading_count", "schema_types"], rows: pageRows },
|
|
1562
|
+
{ title: "Content Gaps", columns: ["source_position", "source_domain", "competitor_heading", "target_coverage", "terms"], rows: gaps.slice(0, 80) },
|
|
1563
|
+
{ title: "PAA Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 80) }
|
|
1564
|
+
]
|
|
1565
|
+
}));
|
|
1566
|
+
const status2 = warnings.length ? "partial" : "succeeded";
|
|
1567
|
+
const counts = { organic: organic.length, pages: pageRows.length, gaps: gaps.length, questions: questions.length, aiCitations: aiRows.length };
|
|
1568
|
+
await ctx.artifacts.writeManifest(status2, counts, warnings, []);
|
|
1569
|
+
return { title: "SERP Comparison", summary, status: status2, counts, warnings, errors: [], reportPath };
|
|
1570
|
+
}
|
|
1571
|
+
};
|
|
1572
|
+
var paaExpansionBriefWorkflowDefinition = {
|
|
1573
|
+
id: "paa-expansion-brief",
|
|
1574
|
+
title: "PAA Expansion Brief",
|
|
1575
|
+
description: "Expand People Also Ask questions into an evidence-backed writer brief, section map, and source table.",
|
|
1576
|
+
inputSchema: PaaExpansionBriefInputSchema,
|
|
1577
|
+
async run(input, ctx) {
|
|
1578
|
+
await ctx.artifacts.writeManifest("running", {}, [], []);
|
|
1579
|
+
const warnings = [];
|
|
1580
|
+
const paa = await ctx.client.post("/harvest/sync", {
|
|
1581
|
+
query: input.keyword,
|
|
1582
|
+
location: input.location,
|
|
1583
|
+
maxQuestions: input.maxQuestions,
|
|
1584
|
+
depth: input.depth,
|
|
1585
|
+
format: "json"
|
|
1586
|
+
}, 3e5);
|
|
1587
|
+
await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
|
|
1588
|
+
const questions = questionRows(paa);
|
|
1589
|
+
const sourceRows2 = sourceDomainRows(questions);
|
|
1590
|
+
const byIntent = /* @__PURE__ */ new Map();
|
|
1591
|
+
for (const row of questions) {
|
|
1592
|
+
const list = byIntent.get(String(row.intent)) ?? [];
|
|
1593
|
+
list.push(row);
|
|
1594
|
+
byIntent.set(String(row.intent), list);
|
|
1595
|
+
}
|
|
1596
|
+
const sectionRows = [...byIntent.entries()].map(([intent, rows]) => ({
|
|
1597
|
+
recommended_section: intent,
|
|
1598
|
+
question_count: rows.length,
|
|
1599
|
+
sample_questions: rows.slice(0, 5).map((row) => row.question).join(" | "),
|
|
1600
|
+
source_domains: [...new Set(rows.map((row) => row.source_domain).filter(Boolean))].slice(0, 5).join("; "),
|
|
1601
|
+
terms: textTerms(rows.map((row) => `${row.question} ${row.answer_excerpt}`).join(" "), 10)
|
|
1602
|
+
})).sort((a, b) => Number(b.question_count) - Number(a.question_count));
|
|
1603
|
+
await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
|
|
1604
|
+
await ctx.artifacts.writeCsv("Source domains CSV", "source-domains.csv", ["domain", "question_mentions", "source_url_count", "top_intents"], sourceRows2);
|
|
1605
|
+
await ctx.artifacts.writeCsv("Section map CSV", "section-map.csv", ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], sectionRows);
|
|
1606
|
+
await ctx.artifacts.writeJson("PAA brief evidence", "evidence.json", { input, paa, questions, sourceRows: sourceRows2, sectionRows });
|
|
1607
|
+
await ctx.artifacts.writeText("Writer brief", "writer-brief.md", [
|
|
1608
|
+
`# PAA Expansion Brief: ${input.keyword}`,
|
|
1609
|
+
"",
|
|
1610
|
+
`Location: ${input.location ?? "not specified"}`,
|
|
1611
|
+
"",
|
|
1612
|
+
"## Suggested Page Structure",
|
|
1613
|
+
...sectionRows.map((row) => `- ${row.recommended_section}: answer ${row.question_count} related question(s). Sample: ${row.sample_questions}`),
|
|
1614
|
+
"",
|
|
1615
|
+
"## Writing Rules",
|
|
1616
|
+
"- Answer the highest-frequency question in the first 60 words of each section.",
|
|
1617
|
+
"- Use exact customer question language from `paa-questions.csv` for H2/H3 candidates.",
|
|
1618
|
+
"- Use `source-domains.csv` to identify which source types Google is already rewarding.",
|
|
1619
|
+
"- Do not invent citations; cite only rows that have a source URL."
|
|
1620
|
+
].join("\n"));
|
|
1621
|
+
const summary = `${questions.length} PAA questions grouped into ${sectionRows.length} writing sections.`;
|
|
1622
|
+
const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
|
|
1623
|
+
title: "PAA Expansion Brief",
|
|
1624
|
+
subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
|
|
1625
|
+
summary,
|
|
1626
|
+
warnings,
|
|
1627
|
+
tables: [
|
|
1628
|
+
{ title: "Section Map", columns: ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], rows: sectionRows },
|
|
1629
|
+
{ title: "Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 120) },
|
|
1630
|
+
{ title: "Source Domains", columns: ["domain", "question_mentions", "source_url_count", "top_intents"], rows: sourceRows2 }
|
|
1631
|
+
]
|
|
1632
|
+
}));
|
|
1633
|
+
const counts = { questions: questions.length, sections: sectionRows.length, sourceDomains: sourceRows2.length };
|
|
1634
|
+
await ctx.artifacts.writeManifest("succeeded", counts, warnings, []);
|
|
1635
|
+
return { title: "PAA Expansion Brief", summary, status: "succeeded", counts, warnings, errors: [], reportPath };
|
|
1636
|
+
}
|
|
1637
|
+
};
|
|
1638
|
+
var aiOverviewLanguageWorkflowDefinition = {
|
|
1639
|
+
id: "ai-overview-language",
|
|
1640
|
+
title: "AI Overview Language Brief",
|
|
1641
|
+
description: "Turn AI Overview, citation, PAA, and ranking-page evidence into answer-block and citation-hook guidance.",
|
|
1642
|
+
inputSchema: AiOverviewLanguageInputSchema,
|
|
1643
|
+
async run(input, ctx) {
|
|
1644
|
+
await ctx.artifacts.writeManifest("running", {}, [], []);
|
|
1645
|
+
const warnings = [];
|
|
1646
|
+
const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
|
|
1647
|
+
const serp = await ctx.client.post("/harvest/sync", {
|
|
1648
|
+
query: input.keyword,
|
|
1649
|
+
location: input.location,
|
|
1650
|
+
serpOnly: true,
|
|
1651
|
+
maxQuestions: 1,
|
|
1652
|
+
format: "json"
|
|
1653
|
+
}, 18e4);
|
|
1654
|
+
await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
|
|
1655
|
+
let paa = null;
|
|
1656
|
+
try {
|
|
1657
|
+
paa = await ctx.client.post("/harvest/sync", {
|
|
1658
|
+
query: input.keyword,
|
|
1659
|
+
location: input.location,
|
|
1660
|
+
maxQuestions: input.maxQuestions,
|
|
1661
|
+
format: "json"
|
|
1662
|
+
}, 28e4);
|
|
1663
|
+
await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
|
|
1664
|
+
} catch (err) {
|
|
1665
|
+
warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
|
|
1666
|
+
}
|
|
1667
|
+
const citations = (serp.aiOverview?.citations ?? []).map((citation, index) => {
|
|
1668
|
+
const domain = domainFromUrl2(citation.href ?? "");
|
|
1669
|
+
return {
|
|
1670
|
+
citation_position: index + 1,
|
|
1671
|
+
citation_text: citation.text ?? "",
|
|
1672
|
+
url: citation.href ?? "",
|
|
1673
|
+
domain,
|
|
1674
|
+
is_target: targetDomain ? domain === targetDomain : false
|
|
1675
|
+
};
|
|
1676
|
+
});
|
|
1677
|
+
const aioSentences = splitSentences(serp.aiOverview?.text);
|
|
1678
|
+
const claimRows = aioSentences.map((sentence, index) => ({
|
|
1679
|
+
position: index + 1,
|
|
1680
|
+
claim_type: classifySentence(sentence),
|
|
1681
|
+
sentence,
|
|
1682
|
+
reusable_pattern: sentence.length > 140 ? `${sentence.slice(0, 140)}...` : sentence
|
|
1683
|
+
}));
|
|
1684
|
+
const questions = questionRows(paa);
|
|
1685
|
+
const citationSources = (serp.aiOverview?.citations ?? []).filter((citation) => citation.href).map((citation, index) => ({ position: index + 1, title: citation.text, url: citation.href, domain: domainFromUrl2(citation.href) })).slice(0, input.extractTop);
|
|
1686
|
+
const extractedCitations = input.extractTop > 0 ? await extractPages(ctx, citationSources, warnings, "Citation") : [];
|
|
1687
|
+
const extractedRows = extractedCitations.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }));
|
|
1688
|
+
const languageRows = [
|
|
1689
|
+
{
|
|
1690
|
+
block: "direct_answer",
|
|
1691
|
+
guidance: "Open with a 40-70 word answer that directly resolves the query before adding context.",
|
|
1692
|
+
evidence_basis: questions[0]?.question ?? input.keyword
|
|
1693
|
+
},
|
|
1694
|
+
{
|
|
1695
|
+
block: "criteria_or_steps",
|
|
1696
|
+
guidance: "List the criteria, steps, or decision factors Google is already compressing into AI Overview language.",
|
|
1697
|
+
evidence_basis: claimRows.filter((row) => ["criteria", "process"].includes(String(row.claim_type))).map((row) => row.sentence).slice(0, 3).join(" | ")
|
|
1698
|
+
},
|
|
1699
|
+
{
|
|
1700
|
+
block: "citation_hook",
|
|
1701
|
+
guidance: "Add source-worthy details competitors can cite: definitions, numbers, examples, process details, and named entity relationships.",
|
|
1702
|
+
evidence_basis: citations.map((row) => `${row.domain}: ${row.citation_text}`).slice(0, 5).join(" | ")
|
|
1703
|
+
},
|
|
1704
|
+
{
|
|
1705
|
+
block: "faq_followups",
|
|
1706
|
+
guidance: "Use PAA phrasing for follow-up sections so the page answers adjacent questions in Google language.",
|
|
1707
|
+
evidence_basis: questions.slice(0, 5).map((row) => row.question).join(" | ")
|
|
1708
|
+
}
|
|
1709
|
+
];
|
|
1710
|
+
await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], citations);
|
|
1711
|
+
await ctx.artifacts.writeCsv("AI Overview claim patterns CSV", "claim-patterns.csv", ["position", "claim_type", "sentence", "reusable_pattern"], claimRows);
|
|
1712
|
+
await ctx.artifacts.writeCsv("Language guidance CSV", "language-guidance.csv", ["block", "guidance", "evidence_basis"], languageRows);
|
|
1713
|
+
await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
|
|
1714
|
+
await ctx.artifacts.writeCsv("Extracted citation pages CSV", "citation-pages.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], extractedRows);
|
|
1715
|
+
await ctx.artifacts.writeJson("AI Overview language evidence", "evidence.json", { input, serp, paa, citations, claimRows, languageRows, extractedRows, warnings });
|
|
1716
|
+
await ctx.artifacts.writeText("Answer block template", "answer-block-template.md", [
|
|
1717
|
+
`# AI Overview Language Brief: ${input.keyword}`,
|
|
1718
|
+
"",
|
|
1719
|
+
`AI Overview detected: ${serp.aiOverview?.detected ? "yes" : "no"}`,
|
|
1720
|
+
`Target cited: ${targetDomain ? citations.some((row) => row.is_target) ? "yes" : "no" : "target not specified"}`,
|
|
1721
|
+
"",
|
|
1722
|
+
"## Direct Answer Block",
|
|
1723
|
+
"Write one compact answer block that starts with the answer, not background. Keep it clear enough that Google could lift it as a standalone summary.",
|
|
1724
|
+
"",
|
|
1725
|
+
"## Suggested Follow-Up Blocks",
|
|
1726
|
+
...languageRows.map((row) => `- ${row.block}: ${row.guidance}`),
|
|
1727
|
+
"",
|
|
1728
|
+
"## Evidence to Mirror",
|
|
1729
|
+
...claimRows.slice(0, 8).map((row) => `- ${row.claim_type}: ${row.sentence}`),
|
|
1730
|
+
"",
|
|
1731
|
+
"## Citation Hooks",
|
|
1732
|
+
...citations.slice(0, 8).map((row) => `- ${row.domain}: ${row.citation_text}`)
|
|
1733
|
+
].join("\n"));
|
|
1734
|
+
const summary = `${citations.length} AI Overview citations, ${claimRows.length} claim patterns, ${questions.length} PAA questions, ${extractedRows.length} citation pages extracted.`;
|
|
1735
|
+
const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
|
|
1736
|
+
title: "AI Overview Language Brief",
|
|
1737
|
+
subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
|
|
1738
|
+
summary,
|
|
1739
|
+
warnings,
|
|
1740
|
+
tables: [
|
|
1741
|
+
{ title: "Language Guidance", columns: ["block", "guidance", "evidence_basis"], rows: languageRows },
|
|
1742
|
+
{ title: "AI Overview Citations", columns: ["citation_position", "citation_text", "domain", "is_target"], rows: citations },
|
|
1743
|
+
{ title: "Claim Patterns", columns: ["position", "claim_type", "sentence"], rows: claimRows },
|
|
1744
|
+
{ title: "PAA Follow-Ups", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 60) }
|
|
1745
|
+
]
|
|
1746
|
+
}));
|
|
1747
|
+
const status2 = warnings.length || !serp.aiOverview?.detected ? "partial" : "succeeded";
|
|
1748
|
+
const counts = { citations: citations.length, claimPatterns: claimRows.length, questions: questions.length, extractedCitationPages: extractedRows.length };
|
|
1749
|
+
await ctx.artifacts.writeManifest(status2, counts, warnings, []);
|
|
1750
|
+
return { title: "AI Overview Language Brief", summary, status: status2, counts, warnings, errors: [], reportPath };
|
|
1751
|
+
}
|
|
1752
|
+
};
|
|
1753
|
+
|
|
1023
1754
|
// src/workflows/registry.ts
|
|
1024
1755
|
var DEFINITIONS = [
|
|
1025
1756
|
directoryWorkflowDefinition,
|
|
1026
1757
|
agentPacketWorkflowDefinition,
|
|
1027
|
-
localCompetitiveAuditWorkflowDefinition
|
|
1758
|
+
localCompetitiveAuditWorkflowDefinition,
|
|
1759
|
+
mapComparisonWorkflowDefinition,
|
|
1760
|
+
serpComparisonWorkflowDefinition,
|
|
1761
|
+
paaExpansionBriefWorkflowDefinition,
|
|
1762
|
+
aiOverviewLanguageWorkflowDefinition
|
|
1028
1763
|
];
|
|
1029
1764
|
function listWorkflowDefinitions() {
|
|
1030
1765
|
return DEFINITIONS.map(({ id, title, description }) => ({ id, title, description }));
|
|
@@ -1041,7 +1776,7 @@ async function runWorkflow(id, rawInput, options = {}) {
|
|
|
1041
1776
|
if (!apiKey) throw new Error("MCP_SCRAPER_API_KEY is required for workflow runs. Pass --api-key or set the environment variable.");
|
|
1042
1777
|
const apiUrl = options.apiUrl?.trim() || process.env.MCP_SCRAPER_API_URL?.trim() || "https://mcpscraper.dev";
|
|
1043
1778
|
const artifacts = await ArtifactWriter.create(definition.id, definition.title, input, options.outputDir, options.runId);
|
|
1044
|
-
const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl);
|
|
1779
|
+
const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl, options.headers);
|
|
1045
1780
|
try {
|
|
1046
1781
|
return await definition.run(input, {
|
|
1047
1782
|
runId: artifacts.runId,
|
|
@@ -1078,29 +1813,42 @@ function compactInput(input) {
|
|
|
1078
1813
|
return Object.fromEntries(Object.entries(input).filter(([, value]) => value !== void 0));
|
|
1079
1814
|
}
|
|
1080
1815
|
function workflowInput(id, opts) {
|
|
1081
|
-
if (id === "agent-packet") {
|
|
1816
|
+
if (id === "agent-packet" || id === "serp-comparison" || id === "ai-overview-language") {
|
|
1082
1817
|
return compactInput({
|
|
1083
1818
|
keyword: opts.keyword,
|
|
1084
1819
|
domain: opts.domain,
|
|
1820
|
+
url: opts.url,
|
|
1085
1821
|
location: opts.location,
|
|
1822
|
+
maxResults: numberOpt(opts.maxResults),
|
|
1086
1823
|
maxQuestions: numberOpt(opts.maxQuestions),
|
|
1824
|
+
extractTop: numberOpt(opts.extractTop),
|
|
1087
1825
|
includeSerp: opts.serp === false ? false : void 0,
|
|
1088
1826
|
includePaa: opts.paa === false ? false : void 0,
|
|
1089
1827
|
includeAiOverview: booleanOpt(opts.includeAiOverview),
|
|
1090
1828
|
returnPartial: opts.returnPartial === false ? false : void 0
|
|
1091
1829
|
});
|
|
1092
1830
|
}
|
|
1093
|
-
if (id === "
|
|
1831
|
+
if (id === "paa-expansion-brief") {
|
|
1832
|
+
return compactInput({
|
|
1833
|
+
keyword: opts.keyword,
|
|
1834
|
+
location: opts.location,
|
|
1835
|
+
maxQuestions: numberOpt(opts.maxQuestions),
|
|
1836
|
+
depth: numberOpt(opts.depth),
|
|
1837
|
+
returnPartial: opts.returnPartial === false ? false : void 0
|
|
1838
|
+
});
|
|
1839
|
+
}
|
|
1840
|
+
if (id === "directory" || id === "local-competitive-audit" || id === "map-comparison") {
|
|
1094
1841
|
return compactInput({
|
|
1095
1842
|
query: opts.query,
|
|
1843
|
+
location: opts.location,
|
|
1096
1844
|
state: opts.state,
|
|
1097
1845
|
minPopulation: numberOpt(opts.minPop ?? opts.minPopulation),
|
|
1098
1846
|
maxCities: numberOpt(opts.maxCities),
|
|
1099
|
-
maxResultsPerCity: numberOpt(opts.perCity ?? opts.maxResultsPerCity),
|
|
1847
|
+
maxResultsPerCity: numberOpt(opts.perCity ?? opts.maxResultsPerCity ?? opts.maxResults),
|
|
1100
1848
|
concurrency: numberOpt(opts.concurrency),
|
|
1101
1849
|
proxyMode: opts.proxyMode,
|
|
1102
|
-
hydrateTop: id === "local-competitive-audit" ? numberOpt(opts.hydrateTop) : void 0,
|
|
1103
|
-
maxReviews: id === "local-competitive-audit" ? numberOpt(opts.reviews ?? opts.maxReviews) : void 0,
|
|
1850
|
+
hydrateTop: id === "local-competitive-audit" || id === "map-comparison" ? numberOpt(opts.hydrateTop) : void 0,
|
|
1851
|
+
maxReviews: id === "local-competitive-audit" || id === "map-comparison" ? numberOpt(opts.reviews ?? opts.maxReviews) : void 0,
|
|
1104
1852
|
returnPartial: opts.returnPartial === false ? false : void 0
|
|
1105
1853
|
});
|
|
1106
1854
|
}
|
|
@@ -1131,6 +1879,17 @@ function apiOptions(opts) {
|
|
|
1131
1879
|
apiKey
|
|
1132
1880
|
};
|
|
1133
1881
|
}
|
|
1882
|
+
function openExternalUrl(url) {
|
|
1883
|
+
const command = process.platform === "darwin" ? "open" : process.platform === "win32" ? "cmd" : "xdg-open";
|
|
1884
|
+
const args = process.platform === "win32" ? ["/c", "start", "", url] : [url];
|
|
1885
|
+
try {
|
|
1886
|
+
const child = (0, import_node_child_process2.spawn)(command, args, { detached: true, stdio: "ignore" });
|
|
1887
|
+
child.unref();
|
|
1888
|
+
return true;
|
|
1889
|
+
} catch {
|
|
1890
|
+
return false;
|
|
1891
|
+
}
|
|
1892
|
+
}
|
|
1134
1893
|
async function apiRequest(path, method, opts, body) {
|
|
1135
1894
|
const { apiUrl, apiKey } = apiOptions(opts);
|
|
1136
1895
|
const res = await fetch(`${apiUrl}${path}`, {
|
|
@@ -1149,7 +1908,7 @@ async function apiRequest(path, method, opts, body) {
|
|
|
1149
1908
|
return data;
|
|
1150
1909
|
}
|
|
1151
1910
|
function addWorkflowInputOptions(command) {
|
|
1152
|
-
return command.option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--location <location>", "Target location").option("--max-questions <n>", "Maximum PAA questions").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts");
|
|
1911
|
+
return command.option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--url <url>", "Target page URL").option("--location <location>", "Target location").option("--max-results <n>", "Maximum SERP or Maps results").option("--max-questions <n>", "Maximum PAA questions").option("--extract-top <n>", "Top result pages or citation pages to extract").option("--depth <n>", "PAA expansion depth").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts");
|
|
1153
1912
|
}
|
|
1154
1913
|
function buildHumanCli() {
|
|
1155
1914
|
const program = new import_commander.Command();
|
|
@@ -1159,6 +1918,34 @@ function buildHumanCli() {
|
|
|
1159
1918
|
writeOutput(opts.json ? output : renderDoctor(output), opts.json);
|
|
1160
1919
|
if (!output.ok) process.exitCode = 1;
|
|
1161
1920
|
});
|
|
1921
|
+
const billing = program.command("billing").description("Inspect billing and start checkout flows.");
|
|
1922
|
+
const billingConcurrency = billing.command("concurrency").description("Manage MCP Scraper concurrency slots.");
|
|
1923
|
+
billingConcurrency.command("info").description("Show current concurrency limit and extra-slot price.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (opts) => {
|
|
1924
|
+
const result = await apiRequest("/billing/credits", "POST", opts, {});
|
|
1925
|
+
if (opts.json) writeOutput(result.concurrency, true);
|
|
1926
|
+
else {
|
|
1927
|
+
const upgrade = result.concurrency?.upgrade;
|
|
1928
|
+
writeOutput([
|
|
1929
|
+
`Current limit: ${result.concurrency?.current_limit ?? "unknown"} concurrent operations`,
|
|
1930
|
+
`Extra slots: ${result.concurrency?.current_extra_slots ?? "unknown"}`,
|
|
1931
|
+
`Extra concurrency slot: ${upgrade?.price_label ?? "$5/month"}`,
|
|
1932
|
+
`Upgrade command: ${upgrade?.terminal_command ?? "mcp-scraper-cli billing concurrency checkout"}`
|
|
1933
|
+
].join("\n"), false);
|
|
1934
|
+
}
|
|
1935
|
+
});
|
|
1936
|
+
billingConcurrency.command("checkout").description("Create a hosted Stripe checkout for one extra concurrency slot.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").option("--no-open", "Print the checkout URL without opening a browser").action(async (opts) => {
|
|
1937
|
+
const result = await apiRequest("/billing/concurrency/terminal-checkout", "POST", opts, {});
|
|
1938
|
+
if (opts.json) {
|
|
1939
|
+
writeOutput(result, true);
|
|
1940
|
+
return;
|
|
1941
|
+
}
|
|
1942
|
+
const opened = opts.open !== false && openExternalUrl(result.checkout_url);
|
|
1943
|
+
writeOutput([
|
|
1944
|
+
`Extra concurrency slot: ${result.price?.price_label ?? "$5/month"}`,
|
|
1945
|
+
opened ? `Opened checkout: ${result.checkout_url}` : `Checkout URL: ${result.checkout_url}`,
|
|
1946
|
+
result.next_step ?? "Complete checkout, then retry the MCP request."
|
|
1947
|
+
].join("\n"), false);
|
|
1948
|
+
});
|
|
1162
1949
|
const agent = program.command("agent").description("Generate AI-agent install configs and workflow prompts.");
|
|
1163
1950
|
agent.command("install <host>").description("Print install/config instructions for codex, claude, or claude-desktop.").option("--api-key <key>", "API key to place in generated config").option("--package <spec>", "npm package spec", "mcp-scraper@latest").option("--json", "Print machine-readable JSON").action((host, opts) => {
|
|
1164
1951
|
const valid = ["codex", "claude", "claude-desktop"];
|
|
@@ -1181,7 +1968,7 @@ function buildHumanCli() {
|
|
|
1181
1968
|
else writeOutput(rows.map((row) => `${row.id} ${row.title}
|
|
1182
1969
|
${row.description}`).join("\n"), false);
|
|
1183
1970
|
});
|
|
1184
|
-
workflow.command("run <id>").description("Run a workflow and save local artifacts.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--location <location>", "Target location").option("--max-questions <n>", "Maximum PAA questions").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts").action(async (id, opts) => {
|
|
1971
|
+
workflow.command("run <id>").description("Run a workflow and save local artifacts.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--url <url>", "Target page URL").option("--location <location>", "Target location").option("--max-results <n>", "Maximum SERP or Maps results").option("--max-questions <n>", "Maximum PAA questions").option("--extract-top <n>", "Top result pages or citation pages to extract").option("--depth <n>", "PAA expansion depth").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts").action(async (id, opts) => {
|
|
1185
1972
|
const summary = await runWorkflow(id, workflowInput(id, opts), {
|
|
1186
1973
|
apiKey: opts.apiKey,
|
|
1187
1974
|
apiUrl: opts.apiUrl,
|