mcp-scraper 0.2.12 → 0.2.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +15 -2
  2. package/dist/bin/api-server.cjs +1836 -633
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +3 -3
  5. package/dist/bin/browser-agent-stdio-server.cjs +1 -1
  6. package/dist/bin/browser-agent-stdio-server.cjs.map +1 -1
  7. package/dist/bin/browser-agent-stdio-server.js +2 -2
  8. package/dist/bin/mcp-scraper-cli.cjs +799 -12
  9. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  10. package/dist/bin/mcp-scraper-cli.js +90 -9
  11. package/dist/bin/mcp-scraper-cli.js.map +1 -1
  12. package/dist/bin/mcp-scraper-combined-stdio-server.cjs +47 -5
  13. package/dist/bin/mcp-scraper-combined-stdio-server.cjs.map +1 -1
  14. package/dist/bin/mcp-scraper-combined-stdio-server.js +3 -3
  15. package/dist/bin/mcp-scraper-install.cjs +4 -3
  16. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  17. package/dist/bin/mcp-scraper-install.js +4 -3
  18. package/dist/bin/mcp-scraper-install.js.map +1 -1
  19. package/dist/bin/mcp-stdio-server.cjs +47 -5
  20. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  21. package/dist/bin/mcp-stdio-server.js +2 -2
  22. package/dist/{chunk-ROS67BNV.js → chunk-BSYPATSM.js} +21 -2
  23. package/dist/chunk-BSYPATSM.js.map +1 -0
  24. package/dist/chunk-F44RBOJ5.js +1575 -0
  25. package/dist/chunk-F44RBOJ5.js.map +1 -0
  26. package/dist/{chunk-P5RYXUV7.js → chunk-IWQTNY2E.js} +48 -6
  27. package/dist/chunk-IWQTNY2E.js.map +1 -0
  28. package/dist/chunk-ROCXJOVY.js +7 -0
  29. package/dist/chunk-ROCXJOVY.js.map +1 -0
  30. package/dist/{chunk-7GCCOT3M.js → chunk-TL7YTFLH.js} +15 -1
  31. package/dist/chunk-TL7YTFLH.js.map +1 -0
  32. package/dist/{chunk-PB7ENA3G.js → chunk-ZV2XXYR7.js} +2 -2
  33. package/dist/{db-BVHYI57K.js → db-P5X6UQ3E.js} +2 -2
  34. package/dist/{server-I2H5WG5I.js → server-7QP6HQJJ.js} +556 -164
  35. package/dist/server-7QP6HQJJ.js.map +1 -0
  36. package/dist/{worker-TDJQ6TH3.js → worker-OZSWIS3F.js} +3 -3
  37. package/docs/specs/cli-agent-wiring-spec.md +4 -1
  38. package/docs/specs/seo-cli-growth-roadmap-spec.md +4 -2
  39. package/package.json +1 -1
  40. package/dist/chunk-7GCCOT3M.js.map +0 -1
  41. package/dist/chunk-L6IS63WS.js +0 -869
  42. package/dist/chunk-L6IS63WS.js.map +0 -1
  43. package/dist/chunk-LK27MYGD.js +0 -7
  44. package/dist/chunk-LK27MYGD.js.map +0 -1
  45. package/dist/chunk-P5RYXUV7.js.map +0 -1
  46. package/dist/chunk-ROS67BNV.js.map +0 -1
  47. package/dist/server-I2H5WG5I.js.map +0 -1
  48. /package/dist/{chunk-PB7ENA3G.js.map → chunk-ZV2XXYR7.js.map} +0 -0
  49. /package/dist/{db-BVHYI57K.js.map → db-P5X6UQ3E.js.map} +0 -0
  50. /package/dist/{worker-TDJQ6TH3.js.map → worker-OZSWIS3F.js.map} +0 -0
@@ -3,11 +3,12 @@
3
3
 
4
4
  // src/cli/human-cli.ts
5
5
  var import_commander = require("commander");
6
+ var import_node_child_process2 = require("child_process");
6
7
  var import_promises3 = require("fs/promises");
7
8
  var import_node_path3 = require("path");
8
9
 
9
10
  // src/version.ts
10
- var PACKAGE_VERSION = "0.2.12";
11
+ var PACKAGE_VERSION = "0.2.14";
11
12
 
12
13
  // src/cli/agent-config.ts
13
14
  function apiKeyValue(options) {
@@ -234,6 +235,34 @@ var AGENT_PROMPTS = {
234
235
  "",
235
236
  "Use directory_workflow when the user wants cities selected by population and Google Maps candidates per city. Keep query and location separate. Preserve result_position, source_location, review stars/count, categories, and profile URLs for downstream CSV or directory use."
236
237
  ].join("\n"),
238
+ "map-comparison": [
239
+ "# MCP Scraper Maps Comparison Prompt",
240
+ "",
241
+ "Run the map-comparison workflow when the user wants to compare local Maps competitors in a city or across selected markets. Use `maps-results.csv`, `map-comparison.csv`, and `profile-insights.csv` as source of truth.",
242
+ "",
243
+ "Ground recommendations in result_position, review_stars, review_count, category, website presence, review topics, and profile attributes. Treat review gaps and missing websites as opportunity signals, not guaranteed ranking factors."
244
+ ].join("\n"),
245
+ "serp-comparison": [
246
+ "# MCP Scraper SERP Comparison Prompt",
247
+ "",
248
+ "Run the serp-comparison workflow when the user wants to know why competitors outrank a page or what the SERP rewards. Use organic results, extracted page headings, PAA questions, and AI Overview citations before making recommendations.",
249
+ "",
250
+ "Tie each recommendation to `content-gaps.csv`, `page-comparison.csv`, `paa-questions.csv`, or `ai-overview-citations.csv`. Do not invent missing sections or citations."
251
+ ].join("\n"),
252
+ "paa-expansion-brief": [
253
+ "# MCP Scraper PAA Expansion Brief Prompt",
254
+ "",
255
+ "Run the paa-expansion-brief workflow when the user wants to figure out what to write from People Also Ask expansion. Use `section-map.csv` to structure the brief and `paa-questions.csv` for exact customer-language headings.",
256
+ "",
257
+ "Answer the highest-priority questions directly, preserve source URLs when present, and separate evidence-backed sections from assumptions."
258
+ ].join("\n"),
259
+ "ai-overview-language": [
260
+ "# MCP Scraper AI Overview Language Prompt",
261
+ "",
262
+ "Run the ai-overview-language workflow when the user wants to know how to phrase content for AI Overview inclusion. Use `claim-patterns.csv`, `language-guidance.csv`, `ai-overview-citations.csv`, and PAA follow-ups as evidence.",
263
+ "",
264
+ "Recommend concise answer blocks, criteria/step language, citation hooks, and follow-up sections. Do not claim the target will be cited; frame the output as evidence-based language guidance."
265
+ ].join("\n"),
237
266
  "ai-citation-monitor": [
238
267
  "# MCP Scraper AI Citation Monitor Prompt",
239
268
  "",
@@ -412,22 +441,25 @@ async function openWorkflowReport(id, outputDir) {
412
441
  }
413
442
 
414
443
  // src/workflows/registry.ts
415
- var import_zod4 = require("zod");
444
+ var import_zod5 = require("zod");
416
445
 
417
446
  // src/workflows/http-client.ts
418
447
  var WorkflowHttpClient = class {
419
- constructor(apiUrl, apiKey, fetchImpl = fetch) {
448
+ constructor(apiUrl, apiKey, fetchImpl = fetch, extraHeaders = {}) {
420
449
  this.apiUrl = apiUrl;
421
450
  this.apiKey = apiKey;
422
451
  this.fetchImpl = fetchImpl;
452
+ this.extraHeaders = extraHeaders;
423
453
  }
424
454
  apiUrl;
425
455
  apiKey;
426
456
  fetchImpl;
457
+ extraHeaders;
427
458
  async post(path, body, timeoutMs = 18e4) {
428
459
  const res = await this.fetchImpl(`${this.apiUrl.replace(/\/$/, "")}${path}`, {
429
460
  method: "POST",
430
461
  headers: {
462
+ ...this.extraHeaders,
431
463
  "Content-Type": "application/json",
432
464
  "x-api-key": this.apiKey
433
465
  },
@@ -1020,11 +1052,714 @@ ${summary}
1020
1052
  }
1021
1053
  };
1022
1054
 
1055
+ // src/workflows/workflows/comparison-briefs.ts
1056
+ var import_zod4 = require("zod");
1057
+
1058
+ // src/workflows/workflows/seo-workflow-utils.ts
1059
+ var STOP_WORDS = /* @__PURE__ */ new Set([
1060
+ "about",
1061
+ "after",
1062
+ "also",
1063
+ "because",
1064
+ "been",
1065
+ "best",
1066
+ "both",
1067
+ "from",
1068
+ "have",
1069
+ "into",
1070
+ "more",
1071
+ "most",
1072
+ "near",
1073
+ "only",
1074
+ "over",
1075
+ "than",
1076
+ "that",
1077
+ "their",
1078
+ "them",
1079
+ "then",
1080
+ "there",
1081
+ "these",
1082
+ "they",
1083
+ "this",
1084
+ "what",
1085
+ "when",
1086
+ "where",
1087
+ "which",
1088
+ "while",
1089
+ "with",
1090
+ "your",
1091
+ "will",
1092
+ "would",
1093
+ "should",
1094
+ "could",
1095
+ "does",
1096
+ "were",
1097
+ "cost",
1098
+ "costs"
1099
+ ]);
1100
+ function normalizeDomain2(value) {
1101
+ if (!value) return null;
1102
+ try {
1103
+ const url = new URL(value.includes("://") ? value : `https://${value}`);
1104
+ return url.hostname.replace(/^www\./, "").toLowerCase();
1105
+ } catch {
1106
+ return value.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0]?.toLowerCase() || null;
1107
+ }
1108
+ }
1109
+ function domainFromUrl2(url) {
1110
+ return normalizeDomain2(url) ?? "";
1111
+ }
1112
+ function numberFrom2(value) {
1113
+ if (value === null || value === void 0 || value === "") return null;
1114
+ const parsed = Number(String(value).replace(/[^\d.]/g, ""));
1115
+ return Number.isFinite(parsed) ? parsed : null;
1116
+ }
1117
+ function median2(values) {
1118
+ const nums = values.filter((v) => v !== null).sort((a, b) => a - b);
1119
+ if (!nums.length) return null;
1120
+ return nums[Math.floor(nums.length / 2)] ?? null;
1121
+ }
1122
+ function textTerms(text, limit = 12) {
1123
+ const counts = /* @__PURE__ */ new Map();
1124
+ for (const token of text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/)) {
1125
+ if (token.length < 4 || STOP_WORDS.has(token)) continue;
1126
+ counts.set(token, (counts.get(token) ?? 0) + 1);
1127
+ }
1128
+ return [...counts.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, limit).map(([term, count]) => `${term} (${count})`).join("; ");
1129
+ }
1130
+ function classifyQuestion(question) {
1131
+ const q = question.toLowerCase();
1132
+ if (/\b(cost|price|pricing|charge|expensive|cheap)\b/.test(q)) return "cost";
1133
+ if (/\b(best|top|recommended|reviews?|compare|versus|vs)\b/.test(q)) return "comparison";
1134
+ if (/\b(how|steps?|process|way to)\b/.test(q)) return "process";
1135
+ if (/\b(why|worth|important|benefit)\b/.test(q)) return "why";
1136
+ if (/\b(near me|city|local|nearby)\b/.test(q)) return "local";
1137
+ if (/\b(can|does|do|is|are|should|will)\b/.test(q)) return "decision";
1138
+ return "definition";
1139
+ }
1140
+ function questionRows(paa) {
1141
+ return (paa?.flat ?? []).filter((row) => row.question).map((row, index) => {
1142
+ const url = row.source_cite ?? "";
1143
+ const domain = normalizeDomain2(row.source_site ?? "") ?? domainFromUrl2(url);
1144
+ return {
1145
+ position: index + 1,
1146
+ intent: classifyQuestion(row.question ?? ""),
1147
+ question: row.question ?? "",
1148
+ answer_excerpt: (row.answer ?? "").slice(0, 320),
1149
+ source_title: row.source_title ?? "",
1150
+ source_domain: domain,
1151
+ source_url: url
1152
+ };
1153
+ });
1154
+ }
1155
+ function sourceDomainRows(rows) {
1156
+ const byDomain = /* @__PURE__ */ new Map();
1157
+ for (const row of rows) {
1158
+ const domain = String(row.source_domain ?? row.domain ?? "");
1159
+ if (!domain) continue;
1160
+ const entry = byDomain.get(domain) ?? { questions: 0, urls: /* @__PURE__ */ new Set(), intents: /* @__PURE__ */ new Map() };
1161
+ entry.questions += row.question ? 1 : 0;
1162
+ if (row.source_url || row.url) entry.urls.add(String(row.source_url ?? row.url));
1163
+ if (row.intent) entry.intents.set(String(row.intent), (entry.intents.get(String(row.intent)) ?? 0) + 1);
1164
+ byDomain.set(domain, entry);
1165
+ }
1166
+ return [...byDomain.entries()].map(([domain, entry]) => ({
1167
+ domain,
1168
+ question_mentions: entry.questions,
1169
+ source_url_count: entry.urls.size,
1170
+ top_intents: [...entry.intents.entries()].sort((a, b) => b[1] - a[1]).map(([intent, count]) => `${intent} (${count})`).join("; ")
1171
+ })).sort((a, b) => Number(b.question_mentions) - Number(a.question_mentions) || String(a.domain).localeCompare(String(b.domain)));
1172
+ }
1173
+ function splitSentences(text) {
1174
+ return (text ?? "").replace(/\s+/g, " ").split(/(?<=[.!?])\s+/).map((sentence) => sentence.trim()).filter((sentence) => sentence.length > 20).slice(0, 20);
1175
+ }
1176
+ function classifySentence(sentence) {
1177
+ const s = sentence.toLowerCase();
1178
+ if (/\bis\b|\bare\b|\bmeans\b|\brefers to\b/.test(s)) return "definition";
1179
+ if (/\binclude\b|\bconsider\b|\bfactors?\b|\bcriteria\b/.test(s)) return "criteria";
1180
+ if (/\bfirst\b|\bthen\b|\bsteps?\b|\bprocess\b/.test(s)) return "process";
1181
+ if (/\bvs\b|\bthan\b|\bcompare\b|\bdifference\b/.test(s)) return "comparison";
1182
+ if (/\bcost\b|\bprice\b|\baverage\b|\brange\b/.test(s)) return "cost";
1183
+ return "claim";
1184
+ }
1185
+ function pageSummaryRow(page, source) {
1186
+ return {
1187
+ position: source.position ?? "",
1188
+ domain: source.domain,
1189
+ url: source.url,
1190
+ serp_title: source.title ?? "",
1191
+ page_title: page.title ?? "",
1192
+ h1: page.h1 ?? "",
1193
+ meta_description: page.metaDescription ?? "",
1194
+ word_count: page.wordCount ?? "",
1195
+ heading_count: page.headings?.length ?? 0,
1196
+ schema_types: (page.schemaTypes ?? []).join("; ")
1197
+ };
1198
+ }
1199
+ async function mapLimit2(items, limit, fn) {
1200
+ const out = new Array(items.length);
1201
+ let next = 0;
1202
+ async function worker() {
1203
+ while (next < items.length) {
1204
+ const index = next;
1205
+ next += 1;
1206
+ out[index] = await fn(items[index], index);
1207
+ }
1208
+ }
1209
+ await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, () => worker()));
1210
+ return out;
1211
+ }
1212
+
1213
+ // src/workflows/workflows/comparison-briefs.ts
1214
+ var ProxyModeSchema = import_zod4.z.enum(["location", "configured", "none"]);
1215
+ var MapComparisonInputSchema = import_zod4.z.object({
1216
+ query: import_zod4.z.string().min(1),
1217
+ location: import_zod4.z.string().optional(),
1218
+ state: import_zod4.z.string().optional(),
1219
+ minPopulation: import_zod4.z.number().int().min(0).default(1e5),
1220
+ maxCities: import_zod4.z.number().int().min(1).max(100).default(5),
1221
+ maxResultsPerCity: import_zod4.z.number().int().min(1).max(50).default(20),
1222
+ hydrateTop: import_zod4.z.number().int().min(0).max(10).default(5),
1223
+ maxReviews: import_zod4.z.number().int().min(0).max(500).default(25),
1224
+ concurrency: import_zod4.z.number().int().min(1).max(5).default(5),
1225
+ proxyMode: ProxyModeSchema.default("location"),
1226
+ returnPartial: import_zod4.z.boolean().default(true)
1227
+ }).refine((input) => input.location || input.state, {
1228
+ message: "Either location or state is required for map-comparison"
1229
+ });
1230
+ var SerpComparisonInputSchema = import_zod4.z.object({
1231
+ keyword: import_zod4.z.string().min(1),
1232
+ domain: import_zod4.z.string().optional(),
1233
+ url: import_zod4.z.string().url().optional(),
1234
+ location: import_zod4.z.string().optional(),
1235
+ maxResults: import_zod4.z.number().int().min(1).max(20).default(10),
1236
+ maxQuestions: import_zod4.z.number().int().min(1).max(200).default(40),
1237
+ extractTop: import_zod4.z.number().int().min(0).max(10).default(5),
1238
+ includePaa: import_zod4.z.boolean().default(true),
1239
+ includeAiOverview: import_zod4.z.boolean().default(true),
1240
+ returnPartial: import_zod4.z.boolean().default(true)
1241
+ });
1242
+ var PaaExpansionBriefInputSchema = import_zod4.z.object({
1243
+ keyword: import_zod4.z.string().min(1),
1244
+ location: import_zod4.z.string().optional(),
1245
+ maxQuestions: import_zod4.z.number().int().min(1).max(300).default(80),
1246
+ depth: import_zod4.z.number().int().min(1).max(6).default(3),
1247
+ returnPartial: import_zod4.z.boolean().default(true)
1248
+ });
1249
+ var AiOverviewLanguageInputSchema = import_zod4.z.object({
1250
+ keyword: import_zod4.z.string().min(1),
1251
+ domain: import_zod4.z.string().optional(),
1252
+ url: import_zod4.z.string().url().optional(),
1253
+ location: import_zod4.z.string().optional(),
1254
+ maxQuestions: import_zod4.z.number().int().min(1).max(200).default(40),
1255
+ extractTop: import_zod4.z.number().int().min(0).max(8).default(3),
1256
+ returnPartial: import_zod4.z.boolean().default(true)
1257
+ });
1258
+ function businessRowsFromMaps(location, query, results) {
1259
+ return results.map((result) => ({
1260
+ source_query: query,
1261
+ source_location: location,
1262
+ city: location.split(",")[0]?.trim() ?? location,
1263
+ state: location.split(",")[1]?.trim() ?? "",
1264
+ population: "",
1265
+ result_position: result.position,
1266
+ business_name: result.name,
1267
+ review_stars: result.rating ?? "",
1268
+ review_count: result.reviewCount ?? "",
1269
+ category: result.category ?? "",
1270
+ address: result.address ?? "",
1271
+ phone: result.phone ?? "",
1272
+ website_url: result.websiteUrl ?? "",
1273
+ place_url: result.placeUrl ?? "",
1274
+ cid: result.cid ?? "",
1275
+ cid_decimal: result.cidDecimal ?? "",
1276
+ result_status: "ok",
1277
+ error: ""
1278
+ }));
1279
+ }
1280
+ function marketRows(rows) {
1281
+ const byLocation = /* @__PURE__ */ new Map();
1282
+ for (const row of rows) {
1283
+ const key = String(row.source_location ?? row.location ?? "");
1284
+ if (!key) continue;
1285
+ const list = byLocation.get(key) ?? [];
1286
+ list.push(row);
1287
+ byLocation.set(key, list);
1288
+ }
1289
+ return [...byLocation.entries()].map(([location, list]) => {
1290
+ const counts = list.map((row) => numberFrom2(row.review_count));
1291
+ const ratings = list.map((row) => numberFrom2(row.review_stars));
1292
+ const topThree = list.slice(0, 3).map((row) => numberFrom2(row.review_count)).filter((v) => v !== null);
1293
+ const categories = /* @__PURE__ */ new Map();
1294
+ for (const row of list) {
1295
+ const category = String(row.category ?? "");
1296
+ if (category) categories.set(category, (categories.get(category) ?? 0) + 1);
1297
+ }
1298
+ return {
1299
+ source_location: location,
1300
+ result_count: list.filter((row) => row.business_name).length,
1301
+ median_review_count: median2(counts) ?? "",
1302
+ median_rating: median2(ratings) ?? "",
1303
+ top_three_average_review_count: topThree.length ? Math.round(topThree.reduce((a, b) => a + b, 0) / topThree.length) : "",
1304
+ top_categories: [...categories.entries()].sort((a, b) => b[1] - a[1]).slice(0, 5).map(([cat, count]) => `${cat} (${count})`).join("; "),
1305
+ websites_present: list.filter((row) => row.website_url).length
1306
+ };
1307
+ });
1308
+ }
1309
+ function comparisonRows(rows) {
1310
+ const benchmarkByLocation = /* @__PURE__ */ new Map();
1311
+ for (const market of marketRows(rows)) {
1312
+ benchmarkByLocation.set(String(market.source_location), numberFrom2(market.top_three_average_review_count) ?? 0);
1313
+ }
1314
+ return rows.filter((row) => row.business_name).map((row) => {
1315
+ const reviews = numberFrom2(row.review_count) ?? 0;
1316
+ const benchmark = benchmarkByLocation.get(String(row.source_location)) ?? 0;
1317
+ const websiteMissing = !row.website_url;
1318
+ const rank = numberFrom2(row.result_position) ?? 999;
1319
+ return {
1320
+ source_location: row.source_location,
1321
+ result_position: row.result_position,
1322
+ business_name: row.business_name,
1323
+ category: row.category,
1324
+ review_stars: row.review_stars,
1325
+ review_count: row.review_count,
1326
+ review_gap_to_top3_average: benchmark ? Math.max(0, benchmark - reviews) : "",
1327
+ website_url: row.website_url,
1328
+ place_url: row.place_url,
1329
+ comparison_note: rank <= 3 ? "visible leader" : websiteMissing ? "ranking without website" : reviews < benchmark ? "review-light competitor" : "visible competitor"
1330
+ };
1331
+ });
1332
+ }
1333
+ function organicRows(serp, targetDomain) {
1334
+ return (serp?.organicResults ?? []).map((result) => {
1335
+ const url = result.url ?? "";
1336
+ const domain = normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(url);
1337
+ return {
1338
+ position: result.position ?? "",
1339
+ title: result.title ?? "",
1340
+ url,
1341
+ domain,
1342
+ snippet: result.snippet ?? "",
1343
+ is_target: targetDomain ? domain === targetDomain : false
1344
+ };
1345
+ });
1346
+ }
1347
+ function pageGapRows(targetPage, competitorPages) {
1348
+ const targetHeadingText = new Set((targetPage?.headings ?? []).map((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim()));
1349
+ const targetTerms = new Set((targetPage?.headings ?? []).flatMap((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/).filter(Boolean)));
1350
+ const rows = [];
1351
+ for (const { source, page } of competitorPages) {
1352
+ for (const heading of page.headings ?? []) {
1353
+ if (heading.level > 3 || !heading.text) continue;
1354
+ const normalized = heading.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim();
1355
+ const terms = normalized.split(/\s+/).filter((term) => term.length > 3);
1356
+ const overlap = terms.filter((term) => targetTerms.has(term)).length;
1357
+ const covered = targetHeadingText.has(normalized) || overlap >= Math.max(2, Math.ceil(terms.length / 2));
1358
+ if (covered && targetPage) continue;
1359
+ rows.push({
1360
+ source_position: source.position ?? "",
1361
+ source_domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url),
1362
+ source_url: source.url ?? "",
1363
+ heading_level: heading.level,
1364
+ competitor_heading: heading.text,
1365
+ target_coverage: targetPage ? "not found in target headings" : "no target page extracted",
1366
+ terms: textTerms(heading.text, 6)
1367
+ });
1368
+ }
1369
+ }
1370
+ return rows.slice(0, 150);
1371
+ }
1372
+ async function extractPages(ctx, sources, warnings, labelPrefix) {
1373
+ return mapLimit2(sources.filter((source) => source.url), 2, async (source, index) => {
1374
+ try {
1375
+ const page = await ctx.client.post("/extract-url", { url: source.url }, 18e4);
1376
+ await ctx.artifacts.writeJson(`${labelPrefix} ${index + 1}`, `raw/extract-url/${labelPrefix.toLowerCase()}-${index + 1}.json`, page);
1377
+ return { source, page };
1378
+ } catch (err) {
1379
+ warnings.push(`Page extraction failed for ${source.url}: ${err instanceof Error ? err.message : String(err)}`);
1380
+ return null;
1381
+ }
1382
+ }).then((items) => items.filter((item) => item !== null));
1383
+ }
1384
+ var mapComparisonWorkflowDefinition = {
1385
+ id: "map-comparison",
1386
+ title: "Maps Comparison",
1387
+ description: "Compare Google Maps competitors by rank, reviews, stars, categories, websites, and profile/review signals.",
1388
+ inputSchema: MapComparisonInputSchema,
1389
+ async run(input, ctx) {
1390
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1391
+ const warnings = [];
1392
+ let rows;
1393
+ let directory = null;
1394
+ let mapsSearch = null;
1395
+ if (input.location) {
1396
+ mapsSearch = await ctx.client.post("/maps/search", {
1397
+ query: input.query,
1398
+ location: input.location,
1399
+ maxResults: input.maxResultsPerCity,
1400
+ proxyMode: input.proxyMode
1401
+ }, 24e4);
1402
+ rows = businessRowsFromMaps(input.location, input.query, mapsSearch.results);
1403
+ await ctx.artifacts.writeJson("Maps search raw JSON", "raw/maps-search.json", mapsSearch);
1404
+ } else {
1405
+ directory = await ctx.client.post("/directory/run", {
1406
+ query: input.query,
1407
+ state: input.state,
1408
+ minPopulation: input.minPopulation,
1409
+ maxCities: input.maxCities,
1410
+ maxResultsPerCity: input.maxResultsPerCity,
1411
+ concurrency: input.concurrency,
1412
+ proxyMode: input.proxyMode,
1413
+ saveCsv: true
1414
+ }, 9e5);
1415
+ rows = directoryRows(directory);
1416
+ await ctx.artifacts.writeJson("Directory raw JSON", "raw/directory-workflow.json", directory);
1417
+ await ctx.artifacts.writeCsv("Directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, rows);
1418
+ warnings.push(...directory.warnings);
1419
+ }
1420
+ const compareRows = comparisonRows(rows);
1421
+ const selected = compareRows.slice(0, input.hydrateTop * Math.max(1, input.location ? 1 : input.maxCities));
1422
+ const hydrated = await mapLimit2(selected, 3, async (row, index) => {
1423
+ try {
1424
+ const detail = await ctx.client.post("/maps/place", {
1425
+ businessName: row.business_name,
1426
+ location: row.source_location,
1427
+ includeReviews: input.maxReviews > 0,
1428
+ maxReviews: Math.max(1, input.maxReviews)
1429
+ }, 18e4);
1430
+ await ctx.artifacts.writeJson(`${row.business_name} profile`, `raw/maps-place-intel/${index + 1}-${String(row.business_name).toLowerCase().replace(/[^a-z0-9]+/g, "-")}.json`, detail);
1431
+ return { row, detail, error: "" };
1432
+ } catch (err) {
1433
+ const message = err instanceof Error ? err.message : String(err);
1434
+ warnings.push(`Profile hydration failed for ${row.business_name}: ${message}`);
1435
+ return { row, detail: null, error: message };
1436
+ }
1437
+ });
1438
+ const profileRows = hydrated.map(({ row, detail, error }) => ({
1439
+ source_location: row.source_location,
1440
+ result_position: row.result_position,
1441
+ business_name: row.business_name,
1442
+ category: detail?.category ?? row.category,
1443
+ review_stars: detail?.rating ?? row.review_stars,
1444
+ review_count: detail?.reviewCount ?? row.review_count,
1445
+ website_url: detail?.website ?? row.website_url,
1446
+ review_topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1447
+ about_attributes: (detail?.aboutAttributes ?? []).map((a) => `${a.section}: ${a.attribute}`).join("; "),
1448
+ reviews_status: detail?.reviewsStatus ?? "",
1449
+ error
1450
+ }));
1451
+ const markets = marketRows(rows);
1452
+ await ctx.artifacts.writeCsv("Maps results CSV", "maps-results.csv", ["source_query", "source_location", "city", "state", "population", "result_position", "business_name", "review_stars", "review_count", "category", "address", "phone", "website_url", "place_url", "cid", "cid_decimal", "result_status", "error"], rows);
1453
+ await ctx.artifacts.writeCsv("Comparison CSV", "map-comparison.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "website_url", "place_url", "comparison_note"], compareRows);
1454
+ await ctx.artifacts.writeCsv("Profile insights CSV", "profile-insights.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "website_url", "review_topics", "about_attributes", "reviews_status", "error"], profileRows);
1455
+ await ctx.artifacts.writeJson("Maps comparison evidence", "evidence.json", { input, directory, mapsSearch, rows, compareRows, profileRows, markets, warnings });
1456
+ await ctx.artifacts.writeText("Brief", "brief.md", [
1457
+ `# Maps Comparison: ${input.query}`,
1458
+ "",
1459
+ `Markets: ${markets.map((row) => row.source_location).join(", ")}`,
1460
+ "",
1461
+ "## How to Use",
1462
+ "- Compare rank position against review count and category patterns.",
1463
+ "- Treat review gaps and missing websites as opportunity signals, not guarantees.",
1464
+ "- Use profile topics and attributes as evidence for local content and GBP improvements."
1465
+ ].join("\n"));
1466
+ const summary = `${compareRows.length} Maps competitors compared across ${markets.length} market(s); ${profileRows.filter((row) => !row.error).length} profiles hydrated.`;
1467
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1468
+ title: "Maps Comparison",
1469
+ subtitle: `${input.query}${input.location ? ` \xB7 ${input.location}` : input.state ? ` \xB7 ${input.state}` : ""}`,
1470
+ summary,
1471
+ warnings,
1472
+ tables: [
1473
+ { title: "Market Benchmarks", columns: ["source_location", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "websites_present"], rows: markets },
1474
+ { title: "Competitor Comparison", columns: ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "comparison_note"], rows: compareRows.slice(0, 120) },
1475
+ { title: "Profile Insights", columns: ["source_location", "business_name", "review_topics", "about_attributes", "error"], rows: profileRows }
1476
+ ]
1477
+ }));
1478
+ const status2 = warnings.length ? "partial" : "succeeded";
1479
+ const counts = { markets: markets.length, competitors: compareRows.length, hydratedProfiles: profileRows.filter((row) => !row.error).length };
1480
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
1481
+ return { title: "Maps Comparison", summary, status: status2, counts, warnings, errors: [], reportPath };
1482
+ }
1483
+ };
1484
+ var serpComparisonWorkflowDefinition = {
1485
+ id: "serp-comparison",
1486
+ title: "SERP Comparison",
1487
+ description: "Compare ranking pages, SERP features, PAA evidence, AI Overview citations, and page-level content gaps.",
1488
+ inputSchema: SerpComparisonInputSchema,
1489
+ async run(input, ctx) {
1490
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1491
+ const warnings = [];
1492
+ const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
1493
+ const serp = await ctx.client.post("/harvest/sync", {
1494
+ query: input.keyword,
1495
+ location: input.location,
1496
+ serpOnly: true,
1497
+ maxQuestions: 1,
1498
+ format: "json"
1499
+ }, 18e4);
1500
+ await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
1501
+ let paa = null;
1502
+ if (input.includePaa) {
1503
+ try {
1504
+ paa = await ctx.client.post("/harvest/sync", {
1505
+ query: input.keyword,
1506
+ location: input.location,
1507
+ maxQuestions: input.maxQuestions,
1508
+ format: "json"
1509
+ }, 28e4);
1510
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1511
+ } catch (err) {
1512
+ warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
1513
+ }
1514
+ }
1515
+ const organic = organicRows(serp, targetDomain).slice(0, input.maxResults);
1516
+ const organicSources = (serp.organicResults ?? []).slice(0, input.maxResults);
1517
+ const targetSource = input.url ? { position: 0, title: "Target page", url: input.url, domain: domainFromUrl2(input.url) } : organicSources.find((result) => targetDomain && (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) === targetDomain);
1518
+ const competitorSources = organicSources.filter((result) => result.url && result.url !== targetSource?.url).filter((result) => !targetDomain || (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) !== targetDomain).slice(0, input.extractTop);
1519
+ const targetPages = targetSource ? await extractPages(ctx, [targetSource], warnings, "Target") : [];
1520
+ const competitorPages = input.extractTop > 0 ? await extractPages(ctx, competitorSources, warnings, "Competitor") : [];
1521
+ const targetPage = targetPages[0]?.page ?? null;
1522
+ const pageRows = [
1523
+ ...targetPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title })),
1524
+ ...competitorPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }))
1525
+ ];
1526
+ const gaps = pageGapRows(targetPage, competitorPages);
1527
+ const questions = questionRows(paa);
1528
+ const aiRows = (serp.aiOverview?.citations ?? []).map((citation, index) => ({
1529
+ citation_position: index + 1,
1530
+ citation_text: citation.text ?? "",
1531
+ url: citation.href ?? "",
1532
+ domain: domainFromUrl2(citation.href ?? ""),
1533
+ is_target: targetDomain ? domainFromUrl2(citation.href ?? "") === targetDomain : false
1534
+ }));
1535
+ await ctx.artifacts.writeCsv("Organic results CSV", "organic-results.csv", ["position", "title", "url", "domain", "snippet", "is_target"], organic);
1536
+ await ctx.artifacts.writeCsv("Page comparison CSV", "page-comparison.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], pageRows);
1537
+ await ctx.artifacts.writeCsv("Content gaps CSV", "content-gaps.csv", ["source_position", "source_domain", "source_url", "heading_level", "competitor_heading", "target_coverage", "terms"], gaps);
1538
+ await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
1539
+ await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], aiRows);
1540
+ await ctx.artifacts.writeJson("SERP comparison evidence", "evidence.json", { input, serp, paa, organic, pageRows, gaps, questions, aiRows, warnings });
1541
+ await ctx.artifacts.writeText("Writer brief", "brief.md", [
1542
+ `# SERP Comparison Brief: ${input.keyword}`,
1543
+ "",
1544
+ `Target: ${targetDomain ?? input.url ?? "not specified"}`,
1545
+ `Location: ${input.location ?? "not specified"}`,
1546
+ "",
1547
+ "## Recommended Actions",
1548
+ "- Use `content-gaps.csv` to decide which missing sections deserve coverage.",
1549
+ "- Use `paa-questions.csv` for FAQ and answer-block candidates.",
1550
+ "- Use `ai-overview-citations.csv` to see whether the target is cited in AI Overview evidence.",
1551
+ "- Treat extracted page headings as evidence, not a complete semantic analysis."
1552
+ ].join("\n"));
1553
+ const summary = `${organic.length} organic results, ${pageRows.length} extracted pages, ${gaps.length} heading gaps, ${questions.length} PAA questions.`;
1554
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1555
+ title: "SERP Comparison",
1556
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
1557
+ summary,
1558
+ warnings,
1559
+ tables: [
1560
+ { title: "Organic Results", columns: ["position", "title", "domain", "is_target"], rows: organic },
1561
+ { title: "Page Comparison", columns: ["position", "domain", "h1", "word_count", "heading_count", "schema_types"], rows: pageRows },
1562
+ { title: "Content Gaps", columns: ["source_position", "source_domain", "competitor_heading", "target_coverage", "terms"], rows: gaps.slice(0, 80) },
1563
+ { title: "PAA Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 80) }
1564
+ ]
1565
+ }));
1566
+ const status2 = warnings.length ? "partial" : "succeeded";
1567
+ const counts = { organic: organic.length, pages: pageRows.length, gaps: gaps.length, questions: questions.length, aiCitations: aiRows.length };
1568
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
1569
+ return { title: "SERP Comparison", summary, status: status2, counts, warnings, errors: [], reportPath };
1570
+ }
1571
+ };
1572
+ var paaExpansionBriefWorkflowDefinition = {
1573
+ id: "paa-expansion-brief",
1574
+ title: "PAA Expansion Brief",
1575
+ description: "Expand People Also Ask questions into an evidence-backed writer brief, section map, and source table.",
1576
+ inputSchema: PaaExpansionBriefInputSchema,
1577
+ async run(input, ctx) {
1578
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1579
+ const warnings = [];
1580
+ const paa = await ctx.client.post("/harvest/sync", {
1581
+ query: input.keyword,
1582
+ location: input.location,
1583
+ maxQuestions: input.maxQuestions,
1584
+ depth: input.depth,
1585
+ format: "json"
1586
+ }, 3e5);
1587
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1588
+ const questions = questionRows(paa);
1589
+ const sourceRows2 = sourceDomainRows(questions);
1590
+ const byIntent = /* @__PURE__ */ new Map();
1591
+ for (const row of questions) {
1592
+ const list = byIntent.get(String(row.intent)) ?? [];
1593
+ list.push(row);
1594
+ byIntent.set(String(row.intent), list);
1595
+ }
1596
+ const sectionRows = [...byIntent.entries()].map(([intent, rows]) => ({
1597
+ recommended_section: intent,
1598
+ question_count: rows.length,
1599
+ sample_questions: rows.slice(0, 5).map((row) => row.question).join(" | "),
1600
+ source_domains: [...new Set(rows.map((row) => row.source_domain).filter(Boolean))].slice(0, 5).join("; "),
1601
+ terms: textTerms(rows.map((row) => `${row.question} ${row.answer_excerpt}`).join(" "), 10)
1602
+ })).sort((a, b) => Number(b.question_count) - Number(a.question_count));
1603
+ await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
1604
+ await ctx.artifacts.writeCsv("Source domains CSV", "source-domains.csv", ["domain", "question_mentions", "source_url_count", "top_intents"], sourceRows2);
1605
+ await ctx.artifacts.writeCsv("Section map CSV", "section-map.csv", ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], sectionRows);
1606
+ await ctx.artifacts.writeJson("PAA brief evidence", "evidence.json", { input, paa, questions, sourceRows: sourceRows2, sectionRows });
1607
+ await ctx.artifacts.writeText("Writer brief", "writer-brief.md", [
1608
+ `# PAA Expansion Brief: ${input.keyword}`,
1609
+ "",
1610
+ `Location: ${input.location ?? "not specified"}`,
1611
+ "",
1612
+ "## Suggested Page Structure",
1613
+ ...sectionRows.map((row) => `- ${row.recommended_section}: answer ${row.question_count} related question(s). Sample: ${row.sample_questions}`),
1614
+ "",
1615
+ "## Writing Rules",
1616
+ "- Answer the highest-frequency question in the first 60 words of each section.",
1617
+ "- Use exact customer question language from `paa-questions.csv` for H2/H3 candidates.",
1618
+ "- Use `source-domains.csv` to identify which source types Google is already rewarding.",
1619
+ "- Do not invent citations; cite only rows that have a source URL."
1620
+ ].join("\n"));
1621
+ const summary = `${questions.length} PAA questions grouped into ${sectionRows.length} writing sections.`;
1622
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1623
+ title: "PAA Expansion Brief",
1624
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
1625
+ summary,
1626
+ warnings,
1627
+ tables: [
1628
+ { title: "Section Map", columns: ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], rows: sectionRows },
1629
+ { title: "Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 120) },
1630
+ { title: "Source Domains", columns: ["domain", "question_mentions", "source_url_count", "top_intents"], rows: sourceRows2 }
1631
+ ]
1632
+ }));
1633
+ const counts = { questions: questions.length, sections: sectionRows.length, sourceDomains: sourceRows2.length };
1634
+ await ctx.artifacts.writeManifest("succeeded", counts, warnings, []);
1635
+ return { title: "PAA Expansion Brief", summary, status: "succeeded", counts, warnings, errors: [], reportPath };
1636
+ }
1637
+ };
1638
+ var aiOverviewLanguageWorkflowDefinition = {
1639
+ id: "ai-overview-language",
1640
+ title: "AI Overview Language Brief",
1641
+ description: "Turn AI Overview, citation, PAA, and ranking-page evidence into answer-block and citation-hook guidance.",
1642
+ inputSchema: AiOverviewLanguageInputSchema,
1643
+ async run(input, ctx) {
1644
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1645
+ const warnings = [];
1646
+ const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
1647
+ const serp = await ctx.client.post("/harvest/sync", {
1648
+ query: input.keyword,
1649
+ location: input.location,
1650
+ serpOnly: true,
1651
+ maxQuestions: 1,
1652
+ format: "json"
1653
+ }, 18e4);
1654
+ await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
1655
+ let paa = null;
1656
+ try {
1657
+ paa = await ctx.client.post("/harvest/sync", {
1658
+ query: input.keyword,
1659
+ location: input.location,
1660
+ maxQuestions: input.maxQuestions,
1661
+ format: "json"
1662
+ }, 28e4);
1663
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1664
+ } catch (err) {
1665
+ warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
1666
+ }
1667
+ const citations = (serp.aiOverview?.citations ?? []).map((citation, index) => {
1668
+ const domain = domainFromUrl2(citation.href ?? "");
1669
+ return {
1670
+ citation_position: index + 1,
1671
+ citation_text: citation.text ?? "",
1672
+ url: citation.href ?? "",
1673
+ domain,
1674
+ is_target: targetDomain ? domain === targetDomain : false
1675
+ };
1676
+ });
1677
+ const aioSentences = splitSentences(serp.aiOverview?.text);
1678
+ const claimRows = aioSentences.map((sentence, index) => ({
1679
+ position: index + 1,
1680
+ claim_type: classifySentence(sentence),
1681
+ sentence,
1682
+ reusable_pattern: sentence.length > 140 ? `${sentence.slice(0, 140)}...` : sentence
1683
+ }));
1684
+ const questions = questionRows(paa);
1685
+ const citationSources = (serp.aiOverview?.citations ?? []).filter((citation) => citation.href).map((citation, index) => ({ position: index + 1, title: citation.text, url: citation.href, domain: domainFromUrl2(citation.href) })).slice(0, input.extractTop);
1686
+ const extractedCitations = input.extractTop > 0 ? await extractPages(ctx, citationSources, warnings, "Citation") : [];
1687
+ const extractedRows = extractedCitations.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }));
1688
+ const languageRows = [
1689
+ {
1690
+ block: "direct_answer",
1691
+ guidance: "Open with a 40-70 word answer that directly resolves the query before adding context.",
1692
+ evidence_basis: questions[0]?.question ?? input.keyword
1693
+ },
1694
+ {
1695
+ block: "criteria_or_steps",
1696
+ guidance: "List the criteria, steps, or decision factors Google is already compressing into AI Overview language.",
1697
+ evidence_basis: claimRows.filter((row) => ["criteria", "process"].includes(String(row.claim_type))).map((row) => row.sentence).slice(0, 3).join(" | ")
1698
+ },
1699
+ {
1700
+ block: "citation_hook",
1701
+ guidance: "Add source-worthy details competitors can cite: definitions, numbers, examples, process details, and named entity relationships.",
1702
+ evidence_basis: citations.map((row) => `${row.domain}: ${row.citation_text}`).slice(0, 5).join(" | ")
1703
+ },
1704
+ {
1705
+ block: "faq_followups",
1706
+ guidance: "Use PAA phrasing for follow-up sections so the page answers adjacent questions in Google language.",
1707
+ evidence_basis: questions.slice(0, 5).map((row) => row.question).join(" | ")
1708
+ }
1709
+ ];
1710
+ await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], citations);
1711
+ await ctx.artifacts.writeCsv("AI Overview claim patterns CSV", "claim-patterns.csv", ["position", "claim_type", "sentence", "reusable_pattern"], claimRows);
1712
+ await ctx.artifacts.writeCsv("Language guidance CSV", "language-guidance.csv", ["block", "guidance", "evidence_basis"], languageRows);
1713
+ await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
1714
+ await ctx.artifacts.writeCsv("Extracted citation pages CSV", "citation-pages.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], extractedRows);
1715
+ await ctx.artifacts.writeJson("AI Overview language evidence", "evidence.json", { input, serp, paa, citations, claimRows, languageRows, extractedRows, warnings });
1716
+ await ctx.artifacts.writeText("Answer block template", "answer-block-template.md", [
1717
+ `# AI Overview Language Brief: ${input.keyword}`,
1718
+ "",
1719
+ `AI Overview detected: ${serp.aiOverview?.detected ? "yes" : "no"}`,
1720
+ `Target cited: ${targetDomain ? citations.some((row) => row.is_target) ? "yes" : "no" : "target not specified"}`,
1721
+ "",
1722
+ "## Direct Answer Block",
1723
+ "Write one compact answer block that starts with the answer, not background. Keep it clear enough that Google could lift it as a standalone summary.",
1724
+ "",
1725
+ "## Suggested Follow-Up Blocks",
1726
+ ...languageRows.map((row) => `- ${row.block}: ${row.guidance}`),
1727
+ "",
1728
+ "## Evidence to Mirror",
1729
+ ...claimRows.slice(0, 8).map((row) => `- ${row.claim_type}: ${row.sentence}`),
1730
+ "",
1731
+ "## Citation Hooks",
1732
+ ...citations.slice(0, 8).map((row) => `- ${row.domain}: ${row.citation_text}`)
1733
+ ].join("\n"));
1734
+ const summary = `${citations.length} AI Overview citations, ${claimRows.length} claim patterns, ${questions.length} PAA questions, ${extractedRows.length} citation pages extracted.`;
1735
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1736
+ title: "AI Overview Language Brief",
1737
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
1738
+ summary,
1739
+ warnings,
1740
+ tables: [
1741
+ { title: "Language Guidance", columns: ["block", "guidance", "evidence_basis"], rows: languageRows },
1742
+ { title: "AI Overview Citations", columns: ["citation_position", "citation_text", "domain", "is_target"], rows: citations },
1743
+ { title: "Claim Patterns", columns: ["position", "claim_type", "sentence"], rows: claimRows },
1744
+ { title: "PAA Follow-Ups", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 60) }
1745
+ ]
1746
+ }));
1747
+ const status2 = warnings.length || !serp.aiOverview?.detected ? "partial" : "succeeded";
1748
+ const counts = { citations: citations.length, claimPatterns: claimRows.length, questions: questions.length, extractedCitationPages: extractedRows.length };
1749
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
1750
+ return { title: "AI Overview Language Brief", summary, status: status2, counts, warnings, errors: [], reportPath };
1751
+ }
1752
+ };
1753
+
1023
1754
  // src/workflows/registry.ts
1024
1755
  var DEFINITIONS = [
1025
1756
  directoryWorkflowDefinition,
1026
1757
  agentPacketWorkflowDefinition,
1027
- localCompetitiveAuditWorkflowDefinition
1758
+ localCompetitiveAuditWorkflowDefinition,
1759
+ mapComparisonWorkflowDefinition,
1760
+ serpComparisonWorkflowDefinition,
1761
+ paaExpansionBriefWorkflowDefinition,
1762
+ aiOverviewLanguageWorkflowDefinition
1028
1763
  ];
1029
1764
  function listWorkflowDefinitions() {
1030
1765
  return DEFINITIONS.map(({ id, title, description }) => ({ id, title, description }));
@@ -1041,7 +1776,7 @@ async function runWorkflow(id, rawInput, options = {}) {
1041
1776
  if (!apiKey) throw new Error("MCP_SCRAPER_API_KEY is required for workflow runs. Pass --api-key or set the environment variable.");
1042
1777
  const apiUrl = options.apiUrl?.trim() || process.env.MCP_SCRAPER_API_URL?.trim() || "https://mcpscraper.dev";
1043
1778
  const artifacts = await ArtifactWriter.create(definition.id, definition.title, input, options.outputDir, options.runId);
1044
- const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl);
1779
+ const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl, options.headers);
1045
1780
  try {
1046
1781
  return await definition.run(input, {
1047
1782
  runId: artifacts.runId,
@@ -1078,29 +1813,42 @@ function compactInput(input) {
1078
1813
  return Object.fromEntries(Object.entries(input).filter(([, value]) => value !== void 0));
1079
1814
  }
1080
1815
  function workflowInput(id, opts) {
1081
- if (id === "agent-packet") {
1816
+ if (id === "agent-packet" || id === "serp-comparison" || id === "ai-overview-language") {
1082
1817
  return compactInput({
1083
1818
  keyword: opts.keyword,
1084
1819
  domain: opts.domain,
1820
+ url: opts.url,
1085
1821
  location: opts.location,
1822
+ maxResults: numberOpt(opts.maxResults),
1086
1823
  maxQuestions: numberOpt(opts.maxQuestions),
1824
+ extractTop: numberOpt(opts.extractTop),
1087
1825
  includeSerp: opts.serp === false ? false : void 0,
1088
1826
  includePaa: opts.paa === false ? false : void 0,
1089
1827
  includeAiOverview: booleanOpt(opts.includeAiOverview),
1090
1828
  returnPartial: opts.returnPartial === false ? false : void 0
1091
1829
  });
1092
1830
  }
1093
- if (id === "directory" || id === "local-competitive-audit") {
1831
+ if (id === "paa-expansion-brief") {
1832
+ return compactInput({
1833
+ keyword: opts.keyword,
1834
+ location: opts.location,
1835
+ maxQuestions: numberOpt(opts.maxQuestions),
1836
+ depth: numberOpt(opts.depth),
1837
+ returnPartial: opts.returnPartial === false ? false : void 0
1838
+ });
1839
+ }
1840
+ if (id === "directory" || id === "local-competitive-audit" || id === "map-comparison") {
1094
1841
  return compactInput({
1095
1842
  query: opts.query,
1843
+ location: opts.location,
1096
1844
  state: opts.state,
1097
1845
  minPopulation: numberOpt(opts.minPop ?? opts.minPopulation),
1098
1846
  maxCities: numberOpt(opts.maxCities),
1099
- maxResultsPerCity: numberOpt(opts.perCity ?? opts.maxResultsPerCity),
1847
+ maxResultsPerCity: numberOpt(opts.perCity ?? opts.maxResultsPerCity ?? opts.maxResults),
1100
1848
  concurrency: numberOpt(opts.concurrency),
1101
1849
  proxyMode: opts.proxyMode,
1102
- hydrateTop: id === "local-competitive-audit" ? numberOpt(opts.hydrateTop) : void 0,
1103
- maxReviews: id === "local-competitive-audit" ? numberOpt(opts.reviews ?? opts.maxReviews) : void 0,
1850
+ hydrateTop: id === "local-competitive-audit" || id === "map-comparison" ? numberOpt(opts.hydrateTop) : void 0,
1851
+ maxReviews: id === "local-competitive-audit" || id === "map-comparison" ? numberOpt(opts.reviews ?? opts.maxReviews) : void 0,
1104
1852
  returnPartial: opts.returnPartial === false ? false : void 0
1105
1853
  });
1106
1854
  }
@@ -1131,6 +1879,17 @@ function apiOptions(opts) {
1131
1879
  apiKey
1132
1880
  };
1133
1881
  }
1882
+ function openExternalUrl(url) {
1883
+ const command = process.platform === "darwin" ? "open" : process.platform === "win32" ? "cmd" : "xdg-open";
1884
+ const args = process.platform === "win32" ? ["/c", "start", "", url] : [url];
1885
+ try {
1886
+ const child = (0, import_node_child_process2.spawn)(command, args, { detached: true, stdio: "ignore" });
1887
+ child.unref();
1888
+ return true;
1889
+ } catch {
1890
+ return false;
1891
+ }
1892
+ }
1134
1893
  async function apiRequest(path, method, opts, body) {
1135
1894
  const { apiUrl, apiKey } = apiOptions(opts);
1136
1895
  const res = await fetch(`${apiUrl}${path}`, {
@@ -1149,7 +1908,7 @@ async function apiRequest(path, method, opts, body) {
1149
1908
  return data;
1150
1909
  }
1151
1910
  function addWorkflowInputOptions(command) {
1152
- return command.option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--location <location>", "Target location").option("--max-questions <n>", "Maximum PAA questions").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts");
1911
+ return command.option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--url <url>", "Target page URL").option("--location <location>", "Target location").option("--max-results <n>", "Maximum SERP or Maps results").option("--max-questions <n>", "Maximum PAA questions").option("--extract-top <n>", "Top result pages or citation pages to extract").option("--depth <n>", "PAA expansion depth").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts");
1153
1912
  }
1154
1913
  function buildHumanCli() {
1155
1914
  const program = new import_commander.Command();
@@ -1159,6 +1918,34 @@ function buildHumanCli() {
1159
1918
  writeOutput(opts.json ? output : renderDoctor(output), opts.json);
1160
1919
  if (!output.ok) process.exitCode = 1;
1161
1920
  });
1921
+ const billing = program.command("billing").description("Inspect billing and start checkout flows.");
1922
+ const billingConcurrency = billing.command("concurrency").description("Manage MCP Scraper concurrency slots.");
1923
+ billingConcurrency.command("info").description("Show current concurrency limit and extra-slot price.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (opts) => {
1924
+ const result = await apiRequest("/billing/credits", "POST", opts, {});
1925
+ if (opts.json) writeOutput(result.concurrency, true);
1926
+ else {
1927
+ const upgrade = result.concurrency?.upgrade;
1928
+ writeOutput([
1929
+ `Current limit: ${result.concurrency?.current_limit ?? "unknown"} concurrent operations`,
1930
+ `Extra slots: ${result.concurrency?.current_extra_slots ?? "unknown"}`,
1931
+ `Extra concurrency slot: ${upgrade?.price_label ?? "$5/month"}`,
1932
+ `Upgrade command: ${upgrade?.terminal_command ?? "mcp-scraper-cli billing concurrency checkout"}`
1933
+ ].join("\n"), false);
1934
+ }
1935
+ });
1936
+ billingConcurrency.command("checkout").description("Create a hosted Stripe checkout for one extra concurrency slot.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").option("--no-open", "Print the checkout URL without opening a browser").action(async (opts) => {
1937
+ const result = await apiRequest("/billing/concurrency/terminal-checkout", "POST", opts, {});
1938
+ if (opts.json) {
1939
+ writeOutput(result, true);
1940
+ return;
1941
+ }
1942
+ const opened = opts.open !== false && openExternalUrl(result.checkout_url);
1943
+ writeOutput([
1944
+ `Extra concurrency slot: ${result.price?.price_label ?? "$5/month"}`,
1945
+ opened ? `Opened checkout: ${result.checkout_url}` : `Checkout URL: ${result.checkout_url}`,
1946
+ result.next_step ?? "Complete checkout, then retry the MCP request."
1947
+ ].join("\n"), false);
1948
+ });
1162
1949
  const agent = program.command("agent").description("Generate AI-agent install configs and workflow prompts.");
1163
1950
  agent.command("install <host>").description("Print install/config instructions for codex, claude, or claude-desktop.").option("--api-key <key>", "API key to place in generated config").option("--package <spec>", "npm package spec", "mcp-scraper@latest").option("--json", "Print machine-readable JSON").action((host, opts) => {
1164
1951
  const valid = ["codex", "claude", "claude-desktop"];
@@ -1181,7 +1968,7 @@ function buildHumanCli() {
1181
1968
  else writeOutput(rows.map((row) => `${row.id} ${row.title}
1182
1969
  ${row.description}`).join("\n"), false);
1183
1970
  });
1184
- workflow.command("run <id>").description("Run a workflow and save local artifacts.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--location <location>", "Target location").option("--max-questions <n>", "Maximum PAA questions").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts").action(async (id, opts) => {
1971
+ workflow.command("run <id>").description("Run a workflow and save local artifacts.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--url <url>", "Target page URL").option("--location <location>", "Target location").option("--max-results <n>", "Maximum SERP or Maps results").option("--max-questions <n>", "Maximum PAA questions").option("--extract-top <n>", "Top result pages or citation pages to extract").option("--depth <n>", "PAA expansion depth").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts").action(async (id, opts) => {
1185
1972
  const summary = await runWorkflow(id, workflowInput(id, opts), {
1186
1973
  apiKey: opts.apiKey,
1187
1974
  apiUrl: opts.apiUrl,