mcp-scraper 0.40.1 → 0.40.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/README.md +1 -1
  2. package/dist/bin/api-server.cjs +60220 -0
  3. package/dist/bin/api-server.cjs.map +1 -0
  4. package/dist/bin/api-server.d.cts +1 -0
  5. package/dist/bin/api-server.d.ts +1 -0
  6. package/dist/bin/api-server.js +38 -0
  7. package/dist/bin/api-server.js.map +1 -0
  8. package/dist/bin/mcp-scraper-cli.cjs +2671 -0
  9. package/dist/bin/mcp-scraper-cli.cjs.map +1 -0
  10. package/dist/bin/mcp-scraper-cli.d.cts +1 -0
  11. package/dist/bin/mcp-scraper-cli.d.ts +1 -0
  12. package/dist/bin/mcp-scraper-cli.js +742 -0
  13. package/dist/bin/mcp-scraper-cli.js.map +1 -0
  14. package/dist/bin/mcp-scraper-install.cjs +129 -0
  15. package/dist/bin/mcp-scraper-install.cjs.map +1 -0
  16. package/dist/bin/mcp-scraper-install.d.cts +1 -0
  17. package/dist/bin/mcp-scraper-install.d.ts +1 -0
  18. package/dist/bin/mcp-scraper-install.js +27 -0
  19. package/dist/bin/mcp-scraper-install.js.map +1 -0
  20. package/dist/bin/mcp-stdio-server.cjs +12532 -0
  21. package/dist/bin/mcp-stdio-server.cjs.map +1 -0
  22. package/dist/bin/mcp-stdio-server.d.cts +1 -0
  23. package/dist/bin/mcp-stdio-server.d.ts +1 -0
  24. package/dist/bin/mcp-stdio-server.js +135 -0
  25. package/dist/bin/mcp-stdio-server.js.map +1 -0
  26. package/dist/bin/paa-harvest.cjs +3808 -0
  27. package/dist/bin/paa-harvest.cjs.map +1 -0
  28. package/dist/bin/paa-harvest.d.cts +1 -0
  29. package/dist/bin/paa-harvest.d.ts +1 -0
  30. package/dist/bin/paa-harvest.js +44 -0
  31. package/dist/bin/paa-harvest.js.map +1 -0
  32. package/dist/chunk-345BQXZH.js +712 -0
  33. package/dist/chunk-345BQXZH.js.map +1 -0
  34. package/dist/chunk-3PIWJS6Y.js +276 -0
  35. package/dist/chunk-3PIWJS6Y.js.map +1 -0
  36. package/dist/chunk-44HZLHDV.js +52 -0
  37. package/dist/chunk-44HZLHDV.js.map +1 -0
  38. package/dist/chunk-7XBBFBYY.js +3410 -0
  39. package/dist/chunk-7XBBFBYY.js.map +1 -0
  40. package/dist/chunk-CB5C3BPB.js +135 -0
  41. package/dist/chunk-CB5C3BPB.js.map +1 -0
  42. package/dist/chunk-FQI5PFE7.js +1866 -0
  43. package/dist/chunk-FQI5PFE7.js.map +1 -0
  44. package/dist/chunk-G3P3ZDB4.js +69 -0
  45. package/dist/chunk-G3P3ZDB4.js.map +1 -0
  46. package/dist/chunk-GUVKHCKE.js +3050 -0
  47. package/dist/chunk-GUVKHCKE.js.map +1 -0
  48. package/dist/chunk-K443GQY5.js +24 -0
  49. package/dist/chunk-K443GQY5.js.map +1 -0
  50. package/dist/chunk-LP6E462I.js +11555 -0
  51. package/dist/chunk-LP6E462I.js.map +1 -0
  52. package/dist/chunk-NVXNEOUQ.js +158 -0
  53. package/dist/chunk-NVXNEOUQ.js.map +1 -0
  54. package/dist/chunk-QZXKQB7Y.js +414 -0
  55. package/dist/chunk-QZXKQB7Y.js.map +1 -0
  56. package/dist/chunk-SBLGBZZB.js +647 -0
  57. package/dist/chunk-SBLGBZZB.js.map +1 -0
  58. package/dist/chunk-X2LKCX6H.js +499 -0
  59. package/dist/chunk-X2LKCX6H.js.map +1 -0
  60. package/dist/chunk-XELSA2MS.js +108 -0
  61. package/dist/chunk-XELSA2MS.js.map +1 -0
  62. package/dist/chunk-XORPNO3Z.js +684 -0
  63. package/dist/chunk-XORPNO3Z.js.map +1 -0
  64. package/dist/chunk-ZUJLSICT.js +7 -0
  65. package/dist/chunk-ZUJLSICT.js.map +1 -0
  66. package/dist/db-W3CP562I.js +239 -0
  67. package/dist/db-W3CP562I.js.map +1 -0
  68. package/dist/editorial-reading-room/assets/app.js +335 -0
  69. package/dist/editorial-reading-room/assets/index.html +131 -0
  70. package/dist/editorial-reading-room/assets/styles.css +1052 -0
  71. package/dist/extract-bundle-K4PG3RZJ.js +568 -0
  72. package/dist/extract-bundle-K4PG3RZJ.js.map +1 -0
  73. package/dist/index.cjs +4160 -0
  74. package/dist/index.cjs.map +1 -0
  75. package/dist/index.d.cts +413 -0
  76. package/dist/index.d.ts +413 -0
  77. package/dist/index.js +338 -0
  78. package/dist/index.js.map +1 -0
  79. package/dist/location-data-repository-RLQX6SNM.js +35 -0
  80. package/dist/location-data-repository-RLQX6SNM.js.map +1 -0
  81. package/dist/server-KM3CFWCF.js +34674 -0
  82. package/dist/server-KM3CFWCF.js.map +1 -0
  83. package/dist/site-extract-repository-2SMMFKKL.js +62 -0
  84. package/dist/site-extract-repository-2SMMFKKL.js.map +1 -0
  85. package/dist/worker-FXAGFYOE.js +142 -0
  86. package/dist/worker-FXAGFYOE.js.map +1 -0
  87. package/package.json +4 -2
@@ -0,0 +1,2671 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+
4
+ // src/cli/human-cli.ts
5
+ var import_commander = require("commander");
6
+ var import_node_child_process2 = require("child_process");
7
+ var import_promises4 = require("fs/promises");
8
+ var import_node_path3 = require("path");
9
+
10
+ // src/version.ts
11
+ var PACKAGE_VERSION = "0.40.2";
12
+
13
+ // src/cli/agent-config.ts
14
+ function apiKeyValue(options) {
15
+ return options.apiKey?.trim() || "sk_live_your_key";
16
+ }
17
+ function packageSpec(options) {
18
+ return options.packageSpec?.trim() || "mcp-scraper@latest";
19
+ }
20
+ function combinedNpxArgs(options = {}) {
21
+ return ["-y", "--package", packageSpec(options), "mcp-scraper"];
22
+ }
23
+ function envConfig(options) {
24
+ const env = { MCP_SCRAPER_API_KEY: apiKeyValue(options) };
25
+ const profileName = options.browserProfileName?.trim();
26
+ if (profileName) env.BROWSER_AGENT_PROFILE_NAME = profileName;
27
+ if (options.browserProfileSaveChanges === true) env.BROWSER_AGENT_PROFILE_SAVE_CHANGES = "true";
28
+ return env;
29
+ }
30
+ function tomlInlineTable(value) {
31
+ const entries = Object.entries(value).map(([key, item]) => `${key} = ${JSON.stringify(item)}`);
32
+ return `{ ${entries.join(", ")} }`;
33
+ }
34
+ function renderCodexConfig(options = {}) {
35
+ return [
36
+ "[mcp_servers.mcp-scraper]",
37
+ 'command = "npx"',
38
+ `args = ${JSON.stringify(combinedNpxArgs(options))}`,
39
+ `env = ${tomlInlineTable(envConfig(options))}`
40
+ ].join("\n");
41
+ }
42
+ function renderClaudeCommand(options = {}) {
43
+ const env = Object.entries(envConfig(options)).map(([key, value]) => ` --env ${key}=${value}`);
44
+ return [
45
+ "claude mcp add mcp-scraper --scope user",
46
+ ...env,
47
+ ` -- npx ${combinedNpxArgs(options).join(" ")}`
48
+ ].join(" \\\n");
49
+ }
50
+ function claudeMcpRemoveArgs() {
51
+ return ["mcp", "remove", "mcp-scraper", "-s", "user"];
52
+ }
53
+ function claudeMcpGetArgs() {
54
+ return ["mcp", "get", "mcp-scraper"];
55
+ }
56
+ function parseClaudeMcpGet(stdout) {
57
+ const command = stdout.match(/^\s*Command:\s*(.+?)\s*$/m)?.[1];
58
+ if (!command) return null;
59
+ const argLine = stdout.match(/^\s*Args:\s*(.*?)\s*$/m)?.[1] ?? "";
60
+ const args = argLine.length ? argLine.split(/\s+/) : [];
61
+ const env = {};
62
+ const envBlock = stdout.split(/^\s*Environment:\s*$/m)[1];
63
+ if (envBlock) {
64
+ for (const line of envBlock.split("\n")) {
65
+ const pair = line.match(/^\s{2,}([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);
66
+ if (!pair) {
67
+ if (line.trim().length && !/^\s{2,}/.test(line)) break;
68
+ continue;
69
+ }
70
+ env[pair[1]] = pair[2];
71
+ }
72
+ }
73
+ return { command, args, env };
74
+ }
75
+ function claudeMcpRestoreArgs(snapshot) {
76
+ const args = ["mcp", "add", "mcp-scraper", "--scope", "user"];
77
+ for (const [key, value] of Object.entries(snapshot.env)) {
78
+ args.push("--env", `${key}=${value}`);
79
+ }
80
+ args.push("--", snapshot.command, ...snapshot.args);
81
+ return args;
82
+ }
83
+ function claudeMcpAddArgs(options = {}) {
84
+ const args = ["mcp", "add", "mcp-scraper", "--scope", "user"];
85
+ for (const [key, value] of Object.entries(envConfig(options))) {
86
+ args.push("--env", `${key}=${value}`);
87
+ }
88
+ args.push("--", "npx", ...combinedNpxArgs(options));
89
+ return args;
90
+ }
91
+ function normalizeAgentHost(host) {
92
+ if (host === "claude-code") return "claude";
93
+ if (host === "claude" || host === "codex" || host === "claude-desktop") return host;
94
+ throw new Error('Unknown host "' + host + '". Use: codex, claude, claude-code, or claude-desktop');
95
+ }
96
+ function renderClaudeDesktopConfig(options = {}) {
97
+ return JSON.stringify({
98
+ mcpServers: {
99
+ "mcp-scraper": {
100
+ command: "npx",
101
+ args: combinedNpxArgs(options),
102
+ env: envConfig(options)
103
+ }
104
+ }
105
+ }, null, 2);
106
+ }
107
+ function renderAgentInstall(host, options = {}) {
108
+ const normalizedHost = normalizeAgentHost(host);
109
+ const restart = "Restart the MCP client so it starts a fresh npx process.";
110
+ const applyCommand = `MCP_SCRAPER_API_KEY=${apiKeyValue(options)} npx -y -p mcp-scraper@latest mcp-scraper-cli agent install claude --apply`;
111
+ if (normalizedHost === "codex") {
112
+ return [
113
+ "# Codex MCP config",
114
+ renderCodexConfig(options),
115
+ "",
116
+ restart
117
+ ].join("\n");
118
+ }
119
+ if (normalizedHost === "claude") {
120
+ return [
121
+ "# Claude Code command",
122
+ renderClaudeCommand(options),
123
+ "",
124
+ "# One-command Claude Code setup",
125
+ applyCommand,
126
+ "",
127
+ restart
128
+ ].join("\n");
129
+ }
130
+ return [
131
+ "# Claude Desktop config",
132
+ renderClaudeDesktopConfig(options),
133
+ "",
134
+ "Desktop Extension: https://mcpscraper.dev/downloads/mcp-scraper.mcpb",
135
+ restart
136
+ ].join("\n");
137
+ }
138
+
139
+ // src/cli/doctor.ts
140
+ var import_promises = require("fs/promises");
141
+ var import_node_os = require("os");
142
+ var import_node_path = require("path");
143
+ function status(ok, warn = false) {
144
+ if (ok) return "pass";
145
+ return warn ? "warn" : "fail";
146
+ }
147
+ function parseMajor(version) {
148
+ return Number(version.replace(/^v/, "").split(".")[0] ?? 0);
149
+ }
150
+ async function readKeyFile() {
151
+ const path = process.env.MCP_SCRAPER_KEY_PATH?.trim() || (0, import_node_path.join)((0, import_node_os.homedir)(), ".mcp-scraper-key");
152
+ try {
153
+ const value = (await (0, import_promises.readFile)(path, "utf8")).trim();
154
+ return value || null;
155
+ } catch {
156
+ return null;
157
+ }
158
+ }
159
+ async function npmLatest(fetchImpl) {
160
+ try {
161
+ const res = await fetchImpl("https://registry.npmjs.org/mcp-scraper/latest", {
162
+ signal: AbortSignal.timeout(5e3)
163
+ });
164
+ if (!res.ok) return null;
165
+ const data = await res.json();
166
+ return data.version ?? null;
167
+ } catch {
168
+ return null;
169
+ }
170
+ }
171
+ async function runDoctor(options = {}) {
172
+ const fetchImpl = options.fetchImpl ?? fetch;
173
+ const apiUrl = (options.apiUrl ?? process.env.MCP_SCRAPER_API_URL ?? "https://mcpscraper.dev").replace(/\/$/, "");
174
+ const outputDir = options.outputDir ?? process.env.MCP_SCRAPER_OUTPUT_DIR ?? (0, import_node_path.join)((0, import_node_os.homedir)(), "Downloads", "mcp-scraper");
175
+ const configuredKey = options.apiKey?.trim() || process.env.MCP_SCRAPER_API_KEY?.trim() || await readKeyFile();
176
+ const latest = await npmLatest(fetchImpl);
177
+ const checks = [];
178
+ const nodeMajor = parseMajor(process.version);
179
+ checks.push({
180
+ id: "node",
181
+ label: "Node.js",
182
+ status: status(nodeMajor >= 20),
183
+ detail: process.version,
184
+ fix: nodeMajor >= 20 ? void 0 : "Install Node.js 20 or newer."
185
+ });
186
+ checks.push({
187
+ id: "package_version",
188
+ label: "Local package",
189
+ status: latest && latest !== PACKAGE_VERSION ? "warn" : "pass",
190
+ detail: latest ? `local ${PACKAGE_VERSION}, npm latest ${latest}` : `local ${PACKAGE_VERSION}, npm latest unavailable`,
191
+ fix: latest && latest !== PACKAGE_VERSION ? "Use mcp-scraper@latest and restart the MCP client." : void 0
192
+ });
193
+ checks.push({
194
+ id: "api_key",
195
+ label: "API key",
196
+ status: configuredKey ? "pass" : "warn",
197
+ detail: configuredKey ? "configured" : "not configured",
198
+ fix: configuredKey ? void 0 : "Set MCP_SCRAPER_API_KEY or pass --api-key."
199
+ });
200
+ try {
201
+ await (0, import_promises.mkdir)(outputDir, { recursive: true });
202
+ await (0, import_promises.access)(outputDir);
203
+ checks.push({ id: "output_dir", label: "Output directory", status: "pass", detail: outputDir });
204
+ } catch (err) {
205
+ checks.push({
206
+ id: "output_dir",
207
+ label: "Output directory",
208
+ status: "fail",
209
+ detail: outputDir,
210
+ fix: err instanceof Error ? err.message : "Create a writable output directory."
211
+ });
212
+ }
213
+ if (configuredKey) {
214
+ try {
215
+ const res = await fetchImpl(`${apiUrl}/me`, {
216
+ headers: { "x-api-key": configuredKey },
217
+ signal: AbortSignal.timeout(8e3)
218
+ });
219
+ checks.push({
220
+ id: "api_reachability",
221
+ label: "Hosted API",
222
+ status: res.ok ? "pass" : "fail",
223
+ detail: `${apiUrl}/me returned ${res.status}`,
224
+ fix: res.ok ? void 0 : "Verify the API key and account status."
225
+ });
226
+ } catch (err) {
227
+ checks.push({
228
+ id: "api_reachability",
229
+ label: "Hosted API",
230
+ status: "fail",
231
+ detail: err instanceof Error ? err.message : String(err),
232
+ fix: "Check network access and MCP_SCRAPER_API_URL."
233
+ });
234
+ }
235
+ } else {
236
+ checks.push({
237
+ id: "api_reachability",
238
+ label: "Hosted API",
239
+ status: "skip",
240
+ detail: "skipped because no API key is configured"
241
+ });
242
+ }
243
+ checks.push({
244
+ id: "mcp_config",
245
+ label: "MCP command",
246
+ status: "pass",
247
+ detail: `npx ${combinedNpxArgs().join(" ")}`,
248
+ fix: "Restart the MCP client after package updates."
249
+ });
250
+ return {
251
+ ok: checks.every((check) => check.status === "pass" || check.status === "skip"),
252
+ version: PACKAGE_VERSION,
253
+ npmLatest: latest,
254
+ checks,
255
+ recommendedConfig: {
256
+ command: "npx",
257
+ args: combinedNpxArgs(),
258
+ env: { MCP_SCRAPER_API_KEY: configuredKey ? "$MCP_SCRAPER_API_KEY" : "sk_live_your_key" }
259
+ }
260
+ };
261
+ }
262
+ function renderDoctor(output) {
263
+ const icon = { pass: "PASS", warn: "WARN", fail: "FAIL", skip: "SKIP" };
264
+ const lines = [
265
+ `mcp-scraper doctor v${output.version}`,
266
+ "",
267
+ ...output.checks.map((check) => {
268
+ const fix = check.fix ? `
269
+ Fix: ${check.fix}` : "";
270
+ return `[${icon[check.status]}] ${check.label}: ${check.detail}${fix}`;
271
+ }),
272
+ "",
273
+ `Recommended MCP command: npx ${output.recommendedConfig.args.join(" ")}`
274
+ ];
275
+ return lines.join("\n");
276
+ }
277
+
278
+ // src/cli/prompts.ts
279
+ var AGENT_PROMPTS = {
280
+ "agent-packet": [
281
+ "# MCP Scraper Agent Packet Prompt",
282
+ "",
283
+ "Use MCP Scraper as the evidence layer. Run the agent-packet workflow for the target keyword/domain, then treat `evidence.json`, `sources.csv`, and `competitors.csv` as source of truth.",
284
+ "",
285
+ "Do not invent citations. If a recommendation is not supported by the packet, mark it as an assumption. Turn the evidence into a concise SEO brief, an implementation task list, and content recommendations tied to source rows."
286
+ ].join("\n"),
287
+ "local-competitive-audit": [
288
+ "# MCP Scraper Local Competitive Audit Prompt",
289
+ "",
290
+ "Run the local-competitive-audit workflow for the niche and markets. Use the city summary, competitor CSV, and review-insight CSV to identify market difficulty, review themes, GBP category patterns, and weak competitors.",
291
+ "",
292
+ "Ground recommendations in Maps rank position, review count, star rating, categories, profile details, and review topics. Do not overstate heuristic opportunity scores."
293
+ ].join("\n"),
294
+ "directory-workflow": [
295
+ "# MCP Scraper Directory Workflow Prompt",
296
+ "",
297
+ "Use directory_workflow when the user wants cities selected by population and Google Maps candidates per city. Keep query and location separate. Preserve result_position, source_location, review stars/count, categories, and profile URLs for downstream CSV or directory use."
298
+ ].join("\n"),
299
+ "map-comparison": [
300
+ "# MCP Scraper Maps Comparison Prompt",
301
+ "",
302
+ "Run the map-comparison workflow when the user wants to compare local Maps competitors in a city or across selected markets. Use `maps-results.csv`, `map-comparison.csv`, and `profile-insights.csv` as source of truth.",
303
+ "",
304
+ "Ground recommendations in result_position, review_stars, review_count, category, website presence, review topics, and profile attributes. Treat review gaps and missing websites as opportunity signals, not guaranteed ranking factors."
305
+ ].join("\n"),
306
+ "serp-comparison": [
307
+ "# MCP Scraper SERP Comparison Prompt",
308
+ "",
309
+ "Run the serp-comparison workflow when the user wants to know why competitors outrank a page or what the SERP rewards. Use organic results, extracted page headings, PAA questions, and AI Overview citations before making recommendations.",
310
+ "",
311
+ "Tie each recommendation to `content-gaps.csv`, `page-comparison.csv`, `paa-questions.csv`, or `ai-overview-citations.csv`. Do not invent missing sections or citations."
312
+ ].join("\n"),
313
+ "paa-expansion-brief": [
314
+ "# MCP Scraper PAA Expansion Brief Prompt",
315
+ "",
316
+ "Run the paa-expansion-brief workflow when the user wants to figure out what to write from People Also Ask expansion. Use `section-map.csv` to structure the brief and `paa-questions.csv` for exact customer-language headings.",
317
+ "",
318
+ "Answer the highest-priority questions directly, preserve source URLs when present, and separate evidence-backed sections from assumptions."
319
+ ].join("\n"),
320
+ "ai-overview-language": [
321
+ "# MCP Scraper AI Overview Language Prompt",
322
+ "",
323
+ "Run the ai-overview-language workflow when the user wants to know how to phrase content for AI Overview inclusion. Use `claim-patterns.csv`, `language-guidance.csv`, `ai-overview-citations.csv`, and PAA follow-ups as evidence.",
324
+ "",
325
+ "Recommend concise answer blocks, criteria/step language, citation hooks, and follow-up sections. Do not claim the target will be cited; frame the output as evidence-based language guidance."
326
+ ].join("\n"),
327
+ "ai-citation-monitor": [
328
+ "# MCP Scraper AI Citation Monitor Prompt",
329
+ "",
330
+ "Use search_serp and harvest_paa to check whether the target brand/domain appears in AI Overview citations, organic results, People Also Ask sources, local pack, forums, and videos. Save evidence before summarizing trends or gaps."
331
+ ].join("\n"),
332
+ "serp-brief": [
333
+ "# MCP Scraper SERP Brief Prompt",
334
+ "",
335
+ "Use live SERP, PAA, AI Overview, forum, video, and source evidence to create a writer brief. Tie each recommended section to evidence rows and identify missing proof, entities, comparisons, and customer questions."
336
+ ].join("\n")
337
+ };
338
+ function listPrompts() {
339
+ return Object.keys(AGENT_PROMPTS).sort();
340
+ }
341
+ function renderPrompt(name) {
342
+ const prompt = AGENT_PROMPTS[name];
343
+ if (!prompt) throw new Error(`Unknown prompt "${name}". Available: ${listPrompts().join(", ")}`);
344
+ return prompt;
345
+ }
346
+
347
+ // src/workflows/artifact-writer.ts
348
+ var import_promises2 = require("fs/promises");
349
+ var import_node_fs = require("fs");
350
+ var import_node_os2 = require("os");
351
+ var import_node_path2 = require("path");
352
+ var import_node_child_process = require("child_process");
353
+
354
+ // src/directory/csv.ts
355
+ function csvCell(value) {
356
+ if (value === null || value === void 0) return "";
357
+ const text = String(value);
358
+ return /[",\n\r]/.test(text) ? `"${text.replace(/"/g, '""')}"` : text;
359
+ }
360
+ function rowsToCsv(headers, rows) {
361
+ return [
362
+ headers.join(","),
363
+ ...rows.map((row) => headers.map((header) => csvCell(row[header])).join(","))
364
+ ].join("\n") + "\n";
365
+ }
366
+
367
+ // src/lib/slugify.ts
368
+ function slugify(s) {
369
+ return s.toLowerCase().replace(/\s+/g, "-").replace(/[^a-z0-9-]/g, "");
370
+ }
371
+
372
+ // src/output/memory-library-sink.ts
373
+ async function postToMemoryLibrary(args) {
374
+ const url = process.env.MCP_MEMORY_INGEST_URL;
375
+ const key = process.env.MCP_MEMORY_KEY;
376
+ if (!url || !key) return;
377
+ try {
378
+ const res = await fetch(url, {
379
+ method: "POST",
380
+ headers: { "content-type": "application/json" },
381
+ body: JSON.stringify({ data: { apiKey: key, title: args.title, content: args.content, source: args.source } })
382
+ });
383
+ const body = await res.json().catch(() => null);
384
+ if (!res.ok || body?.result?.error) {
385
+ console.warn("[memory-library-sink] ingest not accepted:", res.status, body?.result?.message ?? "");
386
+ }
387
+ } catch (err) {
388
+ console.warn("[memory-library-sink] ingest failed:", err?.message);
389
+ }
390
+ }
391
+
392
+ // src/workflows/artifact-writer.ts
393
+ function workflowOutputBaseDir(outputDir) {
394
+ return outputDir?.trim() || process.env.MCP_SCRAPER_OUTPUT_DIR?.trim() || (0, import_node_path2.join)((0, import_node_os2.homedir)(), "Downloads", "mcp-scraper");
395
+ }
396
+ function timestamp() {
397
+ return (/* @__PURE__ */ new Date()).toISOString().replace(/[:.]/g, "-");
398
+ }
399
+ function safeSlug(value) {
400
+ return slugify(value).replace(/^-+|-+$/g, "").slice(0, 80) || "run";
401
+ }
402
+ var ArtifactWriter = class _ArtifactWriter {
403
+ constructor(workflowId, title, runId, baseDir, runDir, startedAt, input) {
404
+ this.workflowId = workflowId;
405
+ this.title = title;
406
+ this.runId = runId;
407
+ this.baseDir = baseDir;
408
+ this.runDir = runDir;
409
+ this.startedAt = startedAt;
410
+ this.input = input;
411
+ }
412
+ workflowId;
413
+ title;
414
+ runId;
415
+ baseDir;
416
+ runDir;
417
+ startedAt;
418
+ input;
419
+ artifacts = [];
420
+ static async create(workflowId, title, input, outputDir, forcedRunId) {
421
+ const startedAt = (/* @__PURE__ */ new Date()).toISOString();
422
+ const runId = forcedRunId?.trim() || `${timestamp()}-${safeSlug(workflowId)}-${Math.random().toString(36).slice(2, 8)}`;
423
+ const nameSource = String(input.keyword ?? input.query ?? input.domain ?? input.state ?? workflowId);
424
+ const baseDir = workflowOutputBaseDir(outputDir);
425
+ const runDir = (0, import_node_path2.join)(baseDir, "workflows", workflowId, `${timestamp()}-${safeSlug(nameSource)}-${safeSlug(runId).slice(0, 20)}`);
426
+ await (0, import_promises2.mkdir)(runDir, { recursive: true });
427
+ return new _ArtifactWriter(workflowId, title, runId, baseDir, runDir, startedAt, input);
428
+ }
429
+ async remember(kind, label, path, rows) {
430
+ const size = await (0, import_promises2.stat)(path).then((s) => s.size).catch(() => void 0);
431
+ this.artifacts.push({ kind, label, path, bytes: size, rows });
432
+ await postToMemoryLibrary({ title: `${this.title} ${label}`, content: `artifact: ${path}`, source: `mcp-scraper-workflow:${this.workflowId}:${this.runId}` });
433
+ return path;
434
+ }
435
+ async writeJson(label, relativePath, data) {
436
+ const path = (0, import_node_path2.join)(this.runDir, relativePath);
437
+ await (0, import_promises2.mkdir)((0, import_node_path2.dirname)(path), { recursive: true });
438
+ await (0, import_promises2.writeFile)(path, JSON.stringify(data, null, 2), "utf8");
439
+ return this.remember("json", label, path);
440
+ }
441
+ async writeText(label, relativePath, text, kind = "markdown") {
442
+ const path = (0, import_node_path2.join)(this.runDir, relativePath);
443
+ await (0, import_promises2.mkdir)((0, import_node_path2.dirname)(path), { recursive: true });
444
+ await (0, import_promises2.writeFile)(path, text, "utf8");
445
+ return this.remember(kind, label, path);
446
+ }
447
+ async writeCsv(label, relativePath, headers, rows) {
448
+ const path = (0, import_node_path2.join)(this.runDir, relativePath);
449
+ await (0, import_promises2.mkdir)((0, import_node_path2.dirname)(path), { recursive: true });
450
+ await (0, import_promises2.writeFile)(path, rowsToCsv(headers, rows), "utf8");
451
+ return this.remember("csv", label, path, rows.length);
452
+ }
453
+ async writeHtml(label, relativePath, html) {
454
+ return this.writeText(label, relativePath, html, "html_report");
455
+ }
456
+ async writeManifest(status2, counts, warnings, errors) {
457
+ const manifest = {
458
+ workflow: this.workflowId,
459
+ title: this.title,
460
+ runId: this.runId,
461
+ status: status2,
462
+ startedAt: this.startedAt,
463
+ completedAt: status2 === "running" ? null : (/* @__PURE__ */ new Date()).toISOString(),
464
+ input: this.input,
465
+ artifacts: this.artifacts,
466
+ warnings,
467
+ errors,
468
+ counts
469
+ };
470
+ const path = (0, import_node_path2.join)(this.runDir, "manifest.json");
471
+ await (0, import_promises2.writeFile)(path, JSON.stringify(manifest, null, 2), "utf8");
472
+ await updateWorkflowIndex(manifest, path, this.baseDir);
473
+ return path;
474
+ }
475
+ };
476
+ function indexPath(baseDir = workflowOutputBaseDir()) {
477
+ return (0, import_node_path2.join)(baseDir, "workflows", "index.json");
478
+ }
479
+ async function readIndex(baseDir) {
480
+ const path = indexPath(baseDir);
481
+ if (!(0, import_node_fs.existsSync)(path)) return { runs: [] };
482
+ try {
483
+ return JSON.parse(await (0, import_promises2.readFile)(path, "utf8"));
484
+ } catch {
485
+ return { runs: [] };
486
+ }
487
+ }
488
+ async function updateWorkflowIndex(manifest, manifestPath, baseDir) {
489
+ if (manifest.status === "running") return;
490
+ const path = indexPath(baseDir);
491
+ const index = await readIndex(baseDir);
492
+ const report = manifest.artifacts.find((a) => a.kind === "html_report")?.path ?? null;
493
+ const entry = {
494
+ workflow: manifest.workflow,
495
+ runId: manifest.runId,
496
+ status: manifest.status,
497
+ startedAt: manifest.startedAt,
498
+ reportPath: report,
499
+ manifestPath,
500
+ summary: `${manifest.title} (${manifest.status})`
501
+ };
502
+ index.runs = [entry, ...index.runs.filter((r) => r.runId !== manifest.runId)].slice(0, 200);
503
+ await (0, import_promises2.mkdir)((0, import_node_path2.dirname)(path), { recursive: true });
504
+ await (0, import_promises2.writeFile)(path, JSON.stringify(index, null, 2), "utf8");
505
+ }
506
+ async function listWorkflowReports(outputDir) {
507
+ return (await readIndex(workflowOutputBaseDir(outputDir))).runs;
508
+ }
509
+ async function findWorkflowReport(id, outputDir) {
510
+ const runs = await listWorkflowReports(outputDir);
511
+ if (id === "last") return runs[0] ?? null;
512
+ return runs.find((run) => run.runId === id) ?? null;
513
+ }
514
+ async function openWorkflowReport(id, outputDir) {
515
+ const run = await findWorkflowReport(id, outputDir);
516
+ if (!run?.reportPath) throw new Error(`No report found for "${id}"`);
517
+ if ((0, import_node_os2.platform)() === "darwin") {
518
+ await new Promise((resolve, reject) => {
519
+ (0, import_node_child_process.execFile)("open", [run.reportPath], (err) => err ? reject(err) : resolve());
520
+ });
521
+ }
522
+ return run.reportPath;
523
+ }
524
+
525
+ // src/workflows/registry.ts
526
+ var import_promises3 = require("fs/promises");
527
+ var import_zod7 = require("zod");
528
+
529
+ // src/workflows/http-client.ts
530
+ var WorkflowHttpClient = class {
531
+ constructor(apiUrl, apiKey, fetchImpl = fetch, extraHeaders = {}) {
532
+ this.apiUrl = apiUrl;
533
+ this.apiKey = apiKey;
534
+ this.fetchImpl = fetchImpl;
535
+ this.extraHeaders = extraHeaders;
536
+ }
537
+ apiUrl;
538
+ apiKey;
539
+ fetchImpl;
540
+ extraHeaders;
541
+ async post(path, body, timeoutMs = 18e4) {
542
+ const res = await this.fetchImpl(`${this.apiUrl.replace(/\/$/, "")}${path}`, {
543
+ method: "POST",
544
+ headers: {
545
+ ...this.extraHeaders,
546
+ "Content-Type": "application/json",
547
+ "x-api-key": this.apiKey
548
+ },
549
+ body: JSON.stringify(body),
550
+ signal: AbortSignal.timeout(timeoutMs)
551
+ });
552
+ const data = await res.json().catch(() => ({}));
553
+ if (!res.ok) {
554
+ const message = typeof data.error === "string" ? data.error : `Workflow API ${path} failed with ${res.status}`;
555
+ throw new Error(message);
556
+ }
557
+ return data;
558
+ }
559
+ };
560
+
561
+ // src/workflows/workflows/agent-packet.ts
562
+ var import_zod = require("zod");
563
+
564
+ // src/workflows/report-renderer.ts
565
+ function escapeHtml(value) {
566
+ return String(value ?? "").replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/"/g, "&quot;").replace(/'/g, "&#39;");
567
+ }
568
+ function renderTable(table) {
569
+ const rows = table.rows.map((row) => `<tr>${table.columns.map((col) => `<td>${escapeHtml(row[col])}</td>`).join("")}</tr>`).join("\n");
570
+ return [
571
+ `<section>`,
572
+ `<h2>${escapeHtml(table.title)}</h2>`,
573
+ `<div class="table-wrap"><table>`,
574
+ `<thead><tr>${table.columns.map((col) => `<th>${escapeHtml(col)}</th>`).join("")}</tr></thead>`,
575
+ `<tbody>${rows || `<tr><td colspan="${table.columns.length}">No rows</td></tr>`}</tbody>`,
576
+ `</table></div>`,
577
+ `</section>`
578
+ ].join("\n");
579
+ }
580
+ function renderWorkflowReport(input) {
581
+ const warnings = input.warnings?.length ? `<section class="warnings"><h2>Warnings</h2><ul>${input.warnings.map((w) => `<li>${escapeHtml(w)}</li>`).join("")}</ul></section>` : "";
582
+ const sections = (input.sections ?? []).map((section) => `<section><h2>${escapeHtml(section.title)}</h2><p>${escapeHtml(section.body)}</p></section>`).join("\n");
583
+ const tables = (input.tables ?? []).map(renderTable).join("\n");
584
+ return [
585
+ "<!doctype html>",
586
+ '<html lang="en">',
587
+ "<head>",
588
+ '<meta charset="utf-8">',
589
+ '<meta name="viewport" content="width=device-width, initial-scale=1">',
590
+ `<title>${escapeHtml(input.title)}</title>`,
591
+ "<style>",
592
+ ':root{font-family:Inter,ui-sans-serif,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;color:#172026;background:#f6f7f8}',
593
+ "body{margin:0;padding:32px}main{max-width:1180px;margin:0 auto;background:#fff;border:1px solid #d9dee3;border-radius:8px;padding:28px}",
594
+ "h1{font-size:30px;line-height:1.15;margin:0 0 8px}h2{font-size:18px;margin:28px 0 12px}p{line-height:1.55;color:#3d4952}.sub{color:#66737f;margin:0 0 20px}",
595
+ ".summary{background:#eef6f8;border-left:4px solid #12829a;padding:14px 16px;margin:20px 0}.warnings{background:#fff7e8;border-left:4px solid #c87b00;padding:12px 16px}",
596
+ ".table-wrap{overflow:auto;border:1px solid #dfe5ea;border-radius:8px}table{border-collapse:collapse;width:100%;font-size:13px}th,td{border-bottom:1px solid #edf0f2;padding:9px 10px;text-align:left;vertical-align:top}th{background:#f3f5f7;color:#31404c;position:sticky;top:0}tr:last-child td{border-bottom:0}",
597
+ "</style>",
598
+ "</head>",
599
+ "<body><main>",
600
+ `<h1>${escapeHtml(input.title)}</h1>`,
601
+ input.subtitle ? `<p class="sub">${escapeHtml(input.subtitle)}</p>` : "",
602
+ `<div class="summary">${escapeHtml(input.summary)}</div>`,
603
+ warnings,
604
+ sections,
605
+ tables,
606
+ "</main></body></html>"
607
+ ].join("\n");
608
+ }
609
+
610
+ // src/workflows/workflows/agent-packet.ts
611
+ var AgentPacketInputSchema = import_zod.z.object({
612
+ keyword: import_zod.z.string().min(1),
613
+ domain: import_zod.z.string().optional(),
614
+ location: import_zod.z.string().optional(),
615
+ maxQuestions: import_zod.z.number().int().min(1).max(200).default(40),
616
+ includeSerp: import_zod.z.boolean().default(true),
617
+ includePaa: import_zod.z.boolean().default(true),
618
+ includeAiOverview: import_zod.z.boolean().default(true),
619
+ returnPartial: import_zod.z.boolean().default(true)
620
+ });
621
+ function normalizeDomain(value) {
622
+ if (!value) return null;
623
+ try {
624
+ const url = new URL(value.includes("://") ? value : `https://${value}`);
625
+ return url.hostname.replace(/^www\./, "").toLowerCase();
626
+ } catch {
627
+ return value.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0]?.toLowerCase() || null;
628
+ }
629
+ }
630
+ function domainFromUrl(url) {
631
+ return normalizeDomain(url) ?? "";
632
+ }
633
+ function sourceRows(input, serp, paa) {
634
+ const target = normalizeDomain(input.domain);
635
+ const rows = [];
636
+ for (const result of serp?.organicResults ?? []) {
637
+ const domain = normalizeDomain(result.domain) ?? domainFromUrl(result.url);
638
+ rows.push({
639
+ surface: "organic",
640
+ position: result.position,
641
+ title: result.title,
642
+ url: result.url,
643
+ domain,
644
+ question: "",
645
+ answer_excerpt: result.snippet ?? "",
646
+ is_target: target ? domain === target : false,
647
+ is_competitor: target ? domain !== target : true
648
+ });
649
+ }
650
+ for (const row of paa?.flat ?? []) {
651
+ const url = row.source_cite ?? "";
652
+ const domain = normalizeDomain(row.source_site ?? "") ?? (url ? domainFromUrl(url) : "");
653
+ rows.push({
654
+ surface: "paa",
655
+ position: "",
656
+ title: row.source_title ?? row.source_site ?? "",
657
+ url,
658
+ domain,
659
+ question: row.question,
660
+ answer_excerpt: (row.answer ?? "").slice(0, 240),
661
+ is_target: target ? domain === target : false,
662
+ is_competitor: target ? Boolean(domain && domain !== target) : Boolean(domain)
663
+ });
664
+ }
665
+ for (const [i, cite] of (serp?.aiOverview?.citations ?? []).entries()) {
666
+ const url = cite.href ?? "";
667
+ const domain = url ? domainFromUrl(url) : "";
668
+ rows.push({
669
+ surface: "ai_overview",
670
+ position: i + 1,
671
+ title: cite.text ?? "",
672
+ url,
673
+ domain,
674
+ question: "",
675
+ answer_excerpt: "",
676
+ is_target: target ? domain === target : false,
677
+ is_competitor: target ? Boolean(domain && domain !== target) : Boolean(domain)
678
+ });
679
+ }
680
+ return rows;
681
+ }
682
+ function competitorRows(rows, targetDomain) {
683
+ const byDomain = /* @__PURE__ */ new Map();
684
+ for (const row of rows) {
685
+ const domain = String(row.domain ?? "");
686
+ if (!domain || domain === targetDomain) continue;
687
+ const entry = byDomain.get(domain) ?? { organic: [], paa: 0, ai: 0, urls: /* @__PURE__ */ new Set() };
688
+ if (row.surface === "organic" && row.position) entry.organic.push(Number(row.position));
689
+ if (row.surface === "paa") entry.paa += 1;
690
+ if (row.surface === "ai_overview") entry.ai += 1;
691
+ if (row.url) entry.urls.add(String(row.url));
692
+ byDomain.set(domain, entry);
693
+ }
694
+ return [...byDomain.entries()].map(([domain, entry]) => ({
695
+ domain,
696
+ organic_best_position: entry.organic.length ? Math.min(...entry.organic) : "",
697
+ organic_count: entry.organic.length,
698
+ paa_mentions: entry.paa,
699
+ ai_overview_citations: entry.ai,
700
+ source_url_count: entry.urls.size
701
+ })).sort((a, b) => Number(a.organic_best_position || 999) - Number(b.organic_best_position || 999));
702
+ }
703
+ var agentPacketWorkflowDefinition = {
704
+ id: "agent-packet",
705
+ title: "Agent-Ready SEO Packet",
706
+ description: "Create an evidence folder for AI agents from live SERP/PAA/AI search surfaces.",
707
+ inputSchema: AgentPacketInputSchema,
708
+ createState: () => ({ serp: null, paa: null, warnings: [] }),
709
+ steps: [
710
+ {
711
+ id: "harvest-serp",
712
+ title: "Harvest organic SERP + AI Overview",
713
+ async run({ input, state, ctx }) {
714
+ if (!input.includeSerp) {
715
+ return { state, output: { skipped: true, reason: "includeSerp is false" } };
716
+ }
717
+ try {
718
+ const serp = await ctx.client.post("/harvest/sync", {
719
+ query: input.keyword,
720
+ location: input.location,
721
+ serpOnly: true,
722
+ maxQuestions: 1,
723
+ format: "json"
724
+ }, 18e4);
725
+ await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
726
+ return {
727
+ state: { ...state, serp },
728
+ output: {
729
+ organicResults: serp.organicResults?.length ?? 0,
730
+ localPack: serp.localPack?.length ?? 0,
731
+ aiOverviewDetected: Boolean(serp.aiOverview?.detected),
732
+ aiOverviewCitations: serp.aiOverview?.citations?.length ?? 0
733
+ }
734
+ };
735
+ } catch (err) {
736
+ const message = `SERP evidence unavailable: ${err instanceof Error ? err.message : String(err)}`;
737
+ return { state: { ...state, warnings: [...state.warnings, message] }, output: { error: message }, warnings: [message] };
738
+ }
739
+ }
740
+ },
741
+ {
742
+ id: "harvest-paa",
743
+ title: "Harvest People Also Ask",
744
+ async run({ input, state, ctx }) {
745
+ if (!input.includePaa) {
746
+ return { state, output: { skipped: true, reason: "includePaa is false" } };
747
+ }
748
+ try {
749
+ const paa = await ctx.client.post("/harvest/sync", {
750
+ query: input.keyword,
751
+ location: input.location,
752
+ maxQuestions: input.maxQuestions,
753
+ format: "json"
754
+ }, 28e4);
755
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
756
+ return {
757
+ state: { ...state, paa },
758
+ output: { paaQuestions: paa.flat?.length ?? 0 }
759
+ };
760
+ } catch (err) {
761
+ const message = `PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`;
762
+ return { state: { ...state, warnings: [...state.warnings, message] }, output: { error: message }, warnings: [message] };
763
+ }
764
+ }
765
+ },
766
+ {
767
+ id: "assemble",
768
+ title: "Assemble evidence packet",
769
+ async run({ input, state, ctx }) {
770
+ const { serp, paa } = state;
771
+ const warnings = state.warnings;
772
+ if (!serp && !paa && !input.returnPartial) throw new Error("No SEO evidence was collected");
773
+ const rows = sourceRows(input, serp, paa);
774
+ const target = normalizeDomain(input.domain);
775
+ const competitors = competitorRows(rows, target);
776
+ const evidence = {
777
+ input,
778
+ serp: serp ?? { status: "skipped" },
779
+ paa: paa ?? { status: "skipped" },
780
+ target: {
781
+ domain: target,
782
+ organicPositions: rows.filter((r) => r.surface === "organic" && r.is_target).map((r) => Number(r.position)),
783
+ citedInAiOverview: rows.some((r) => r.surface === "ai_overview" && r.is_target),
784
+ citedInPaa: rows.some((r) => r.surface === "paa" && r.is_target)
785
+ },
786
+ competitors,
787
+ warnings
788
+ };
789
+ await ctx.artifacts.writeJson("Evidence JSON", "evidence.json", evidence);
790
+ await ctx.artifacts.writeCsv("Sources CSV", "sources.csv", ["surface", "position", "title", "url", "domain", "question", "answer_excerpt", "is_target", "is_competitor"], rows);
791
+ await ctx.artifacts.writeCsv("Competitors CSV", "competitors.csv", ["domain", "organic_best_position", "organic_count", "paa_mentions", "ai_overview_citations", "source_url_count"], competitors);
792
+ const brief = [
793
+ `# SEO Evidence Brief: ${input.keyword}`,
794
+ "",
795
+ `Location: ${input.location ?? "not specified"}`,
796
+ `Target domain: ${target ?? "not specified"}`,
797
+ "",
798
+ "## Evidence Summary",
799
+ `- Organic rows: ${rows.filter((r) => r.surface === "organic").length}`,
800
+ `- PAA rows: ${rows.filter((r) => r.surface === "paa").length}`,
801
+ `- AI Overview citations: ${rows.filter((r) => r.surface === "ai_overview").length}`,
802
+ `- Competitor domains: ${competitors.length}`,
803
+ "",
804
+ "## Recommended Use",
805
+ "Use the CSV files as source of truth. Tie content recommendations to evidence rows and mark unsupported ideas as assumptions."
806
+ ].join("\n");
807
+ await ctx.artifacts.writeText("Brief", "brief.md", brief);
808
+ await ctx.artifacts.writeText("Agent tasks", "tasks.md", [
809
+ "# Agent Tasks",
810
+ "",
811
+ "- [ ] Read `evidence.json` before writing recommendations.",
812
+ "- [ ] Compare the target domain against `competitors.csv`.",
813
+ "- [ ] Use `sources.csv` for citations and source-grounded page sections.",
814
+ "- [ ] Mark unsupported recommendations as assumptions."
815
+ ].join("\n"));
816
+ await ctx.artifacts.writeText("Agent instructions", "agent-instructions.md", [
817
+ "# Agent Instructions",
818
+ "",
819
+ "You are working from an MCP Scraper SEO evidence packet. Use `evidence.json` and CSV files as source of truth. Do not invent citations. If a recommendation is not supported by evidence, mark it as an assumption."
820
+ ].join("\n"));
821
+ const summary = `${rows.length} evidence rows and ${competitors.length} competitor domains collected.`;
822
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
823
+ title: "Agent-Ready SEO Packet",
824
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
825
+ summary,
826
+ warnings,
827
+ tables: [
828
+ { title: "Competitor Domains", columns: ["domain", "organic_best_position", "organic_count", "paa_mentions", "ai_overview_citations", "source_url_count"], rows: competitors.slice(0, 50) },
829
+ { title: "Evidence Sources", columns: ["surface", "position", "title", "domain", "question", "is_target"], rows: rows.slice(0, 100) }
830
+ ]
831
+ }));
832
+ const status2 = warnings.length ? "partial" : "succeeded";
833
+ const counts = { sources: rows.length, competitors: competitors.length };
834
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
835
+ return {
836
+ state,
837
+ output: { sources: rows.length, competitors: competitors.length, status: status2 },
838
+ summary: { title: "Agent-Ready SEO Packet", summary, status: status2, counts, warnings, errors: [], reportPath }
839
+ };
840
+ }
841
+ }
842
+ ]
843
+ };
844
+
845
+ // src/workflows/workflows/directory.ts
846
+ var import_zod3 = require("zod");
847
+
848
+ // src/schemas.ts
849
+ var import_zod2 = require("zod");
850
+ var DEFAULT_PROXY_MODE = "none";
851
+ var DEFAULT_MAPS_PROXY_MODE = "none";
852
+ var HarvestOptionsSchema = import_zod2.z.object({
853
+ query: import_zod2.z.string().min(1),
854
+ location: import_zod2.z.string().optional(),
855
+ gl: import_zod2.z.string().length(2).default("us"),
856
+ hl: import_zod2.z.string().length(2).default("en"),
857
+ device: import_zod2.z.enum(["desktop", "mobile"]).default("desktop"),
858
+ proxyMode: import_zod2.z.enum(["location", "configured", "none"]).default(DEFAULT_PROXY_MODE),
859
+ proxyZip: import_zod2.z.string().regex(/^\d{5}$/).optional(),
860
+ keepDefaultProxy: import_zod2.z.boolean().optional(),
861
+ requireUsEgress: import_zod2.z.boolean().optional(),
862
+ maxAttempts: import_zod2.z.number().int().min(1).max(12).optional(),
863
+ debug: import_zod2.z.boolean().default(false),
864
+ depth: import_zod2.z.number().int().min(1).max(30).default(3),
865
+ maxQuestions: import_zod2.z.number().int().min(1).max(1e3).default(100),
866
+ headless: import_zod2.z.boolean().default(false),
867
+ profileDir: import_zod2.z.string().optional(),
868
+ proxy: import_zod2.z.string().url().optional(),
869
+ kernelApiKey: import_zod2.z.string().optional(),
870
+ kernelProxyId: import_zod2.z.string().optional(),
871
+ kernelProxyResolution: import_zod2.z.unknown().optional(),
872
+ outputDir: import_zod2.z.string().default("./paa-output"),
873
+ format: import_zod2.z.enum(["json", "csv", "both"]).default("both"),
874
+ serpOnly: import_zod2.z.boolean().default(false),
875
+ pages: import_zod2.z.number().int().min(1).max(2).default(1),
876
+ recency: import_zod2.z.enum(["day", "week", "month", "year"]).optional(),
877
+ softDeadlineMs: import_zod2.z.number().optional()
878
+ });
879
+ var MapsPlaceOptionsSchema = import_zod2.z.object({
880
+ businessName: import_zod2.z.string().min(1),
881
+ location: import_zod2.z.string().min(1),
882
+ gl: import_zod2.z.string().length(2).default("us"),
883
+ hl: import_zod2.z.string().length(2).default("en"),
884
+ includeReviews: import_zod2.z.boolean().default(false),
885
+ maxReviews: import_zod2.z.number().int().min(1).max(500).default(50),
886
+ includeServices: import_zod2.z.boolean().default(false),
887
+ kernelApiKey: import_zod2.z.string().optional(),
888
+ kernelProxyId: import_zod2.z.string().optional(),
889
+ headless: import_zod2.z.boolean().default(true)
890
+ });
891
+ var MapsSearchOptionsSchema = import_zod2.z.object({
892
+ query: import_zod2.z.string().min(1),
893
+ location: import_zod2.z.string().optional(),
894
+ gl: import_zod2.z.string().length(2).default("us"),
895
+ hl: import_zod2.z.string().length(2).default("en"),
896
+ maxResults: import_zod2.z.number().int().min(1).max(50).default(10),
897
+ includeServices: import_zod2.z.boolean().default(false),
898
+ proxyMode: import_zod2.z.enum(["location", "configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE),
899
+ proxyZip: import_zod2.z.string().regex(/^\d{5}$/).optional(),
900
+ serpRedirect: import_zod2.z.boolean().optional(),
901
+ forceDirectEgress: import_zod2.z.boolean().optional(),
902
+ debug: import_zod2.z.boolean().default(false),
903
+ kernelApiKey: import_zod2.z.string().optional(),
904
+ kernelProxyId: import_zod2.z.string().optional(),
905
+ kernelProxyResolution: import_zod2.z.unknown().optional(),
906
+ headless: import_zod2.z.boolean().default(true)
907
+ });
908
+ var RawPAAItemSchema = import_zod2.z.object({
909
+ question: import_zod2.z.string().min(1),
910
+ answer: import_zod2.z.string().optional(),
911
+ sourceTitle: import_zod2.z.string().optional(),
912
+ sourceSite: import_zod2.z.string().optional(),
913
+ sourceCite: import_zod2.z.string().optional()
914
+ });
915
+ var RawMapsOverviewSchema = import_zod2.z.object({
916
+ name: import_zod2.z.string().nullable(),
917
+ rating: import_zod2.z.string().nullable(),
918
+ reviewCount: import_zod2.z.string().nullable(),
919
+ category: import_zod2.z.string().nullable(),
920
+ address: import_zod2.z.string().nullable(),
921
+ hoursSummary: import_zod2.z.string().nullable(),
922
+ phone: import_zod2.z.string().nullable(),
923
+ phoneDisplay: import_zod2.z.string().nullable(),
924
+ website: import_zod2.z.string().nullable(),
925
+ plusCode: import_zod2.z.string().nullable(),
926
+ bookingUrl: import_zod2.z.string().nullable()
927
+ });
928
+ var RawMapsHoursRowSchema = import_zod2.z.object({
929
+ day: import_zod2.z.string(),
930
+ hours: import_zod2.z.string()
931
+ });
932
+ var RawMapsReviewStatsSchema = import_zod2.z.object({
933
+ reviewHistogram: import_zod2.z.array(import_zod2.z.object({
934
+ stars: import_zod2.z.number(),
935
+ count: import_zod2.z.string()
936
+ })),
937
+ reviewTopics: import_zod2.z.array(import_zod2.z.object({
938
+ label: import_zod2.z.string(),
939
+ count: import_zod2.z.string()
940
+ }))
941
+ });
942
+ var RawMapsReviewCardSchema = import_zod2.z.object({
943
+ reviewId: import_zod2.z.string(),
944
+ author: import_zod2.z.string().nullable(),
945
+ stars: import_zod2.z.string().nullable(),
946
+ date: import_zod2.z.string().nullable(),
947
+ text: import_zod2.z.string().nullable(),
948
+ ownerResponse: import_zod2.z.string().nullable()
949
+ });
950
+ var RawMapsAboutAttributeSchema = import_zod2.z.object({
951
+ section: import_zod2.z.string(),
952
+ attribute: import_zod2.z.string()
953
+ });
954
+
955
+ // src/workflows/workflows/directory.ts
956
+ var DirectoryWorkflowCliInputSchema = import_zod3.z.object({
957
+ query: import_zod3.z.string().min(1),
958
+ state: import_zod3.z.string().min(2).default("TN"),
959
+ minPopulation: import_zod3.z.number().int().min(0).default(1e5),
960
+ maxCities: import_zod3.z.number().int().min(1).max(100).default(25),
961
+ maxResultsPerCity: import_zod3.z.number().int().min(1).max(50).default(20),
962
+ concurrency: import_zod3.z.number().int().min(1).max(5).default(5),
963
+ proxyMode: import_zod3.z.enum(["configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE),
964
+ saveCsv: import_zod3.z.boolean().default(true)
965
+ });
966
+ function directoryRows(result) {
967
+ const rows = [];
968
+ for (const city of result.cities) {
969
+ if (!city.results.length) {
970
+ rows.push({
971
+ source_query: result.query,
972
+ source_location: city.location,
973
+ city: city.city,
974
+ state: city.state,
975
+ population: city.population,
976
+ result_position: null,
977
+ business_name: null,
978
+ review_stars: null,
979
+ review_count: null,
980
+ category: null,
981
+ address: null,
982
+ phone: null,
983
+ website_url: null,
984
+ place_url: null,
985
+ cid: null,
986
+ cid_decimal: null,
987
+ result_status: city.status,
988
+ error: city.error
989
+ });
990
+ continue;
991
+ }
992
+ for (const business of city.results) {
993
+ rows.push({
994
+ source_query: result.query,
995
+ source_location: city.location,
996
+ city: city.city,
997
+ state: city.state,
998
+ population: city.population,
999
+ result_position: business.position,
1000
+ business_name: business.name,
1001
+ review_stars: business.rating,
1002
+ review_count: business.reviewCount,
1003
+ category: business.category,
1004
+ address: business.address,
1005
+ phone: business.phone,
1006
+ website_url: business.websiteUrl,
1007
+ place_url: business.placeUrl,
1008
+ cid: business.cid,
1009
+ cid_decimal: business.cidDecimal,
1010
+ result_status: city.status,
1011
+ error: city.error
1012
+ });
1013
+ }
1014
+ }
1015
+ return rows;
1016
+ }
1017
+ var DIRECTORY_CSV_HEADERS = [
1018
+ "source_query",
1019
+ "source_location",
1020
+ "city",
1021
+ "state",
1022
+ "population",
1023
+ "result_position",
1024
+ "business_name",
1025
+ "review_stars",
1026
+ "review_count",
1027
+ "category",
1028
+ "address",
1029
+ "phone",
1030
+ "website_url",
1031
+ "place_url",
1032
+ "cid",
1033
+ "cid_decimal",
1034
+ "result_status",
1035
+ "error"
1036
+ ];
1037
+ var directoryWorkflowDefinition = {
1038
+ id: "directory",
1039
+ title: "Directory Workflow",
1040
+ description: "Select city markets and export Google Maps business candidates.",
1041
+ inputSchema: DirectoryWorkflowCliInputSchema,
1042
+ async run(input, ctx) {
1043
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1044
+ const result = await ctx.client.post("/directory/run", { ...input, background: false }, 9e5);
1045
+ await ctx.artifacts.writeJson("Directory evidence", "evidence.json", result);
1046
+ const rows = directoryRows(result);
1047
+ await ctx.artifacts.writeCsv("Directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, rows);
1048
+ const cityRows = result.cities.map((city) => ({
1049
+ city: city.city,
1050
+ state: city.state,
1051
+ population: city.population,
1052
+ status: city.status,
1053
+ result_count: city.resultCount,
1054
+ error: city.error ?? ""
1055
+ }));
1056
+ const topRows = rows.filter((row) => row.business_name).slice(0, 100).map((row) => ({
1057
+ city: row.city,
1058
+ position: row.result_position,
1059
+ business: row.business_name,
1060
+ rating: row.review_stars,
1061
+ reviews: row.review_count,
1062
+ category: row.category,
1063
+ website: row.website_url
1064
+ }));
1065
+ const summary = `${result.selectedCityCount} cities processed with ${result.totalResultCount} Maps results.`;
1066
+ await ctx.artifacts.writeText("Summary", "summary.md", `# Directory Workflow
1067
+
1068
+ ${summary}
1069
+ `);
1070
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1071
+ title: "Directory Workflow",
1072
+ subtitle: `${input.query} \xB7 ${input.state}`,
1073
+ summary,
1074
+ warnings: result.warnings,
1075
+ tables: [
1076
+ { title: "Cities", columns: ["city", "state", "population", "status", "result_count", "error"], rows: cityRows },
1077
+ { title: "Top Results", columns: ["city", "position", "business", "rating", "reviews", "category", "website"], rows: topRows }
1078
+ ]
1079
+ }));
1080
+ const status2 = result.cities.some((city) => city.status === "failed") ? "partial" : "succeeded";
1081
+ const counts = { cities: result.selectedCityCount, results: result.totalResultCount, rows: rows.length };
1082
+ await ctx.artifacts.writeManifest(status2, counts, result.warnings, []);
1083
+ return { title: "Directory Workflow", summary, status: status2, counts, warnings: result.warnings, errors: [], reportPath };
1084
+ }
1085
+ };
1086
+
1087
+ // src/workflows/workflows/get-leads.ts
1088
+ var import_zod4 = require("zod");
1089
+ var GetLeadsInputSchema = import_zod4.z.object({
1090
+ query: import_zod4.z.string().min(1).describe('Business category, niche, or keyword to search on Google Maps, e.g. "roofers", "med spas", "dentists". Do not include the city here.'),
1091
+ location: import_zod4.z.string().min(1).describe('City / market to search, e.g. "Houston, TX" or "Austin, Texas".'),
1092
+ maxResults: import_zod4.z.number().int().min(1).max(50).default(25).describe("How many Maps businesses to collect for the market. Maximum 50."),
1093
+ enrichWebsites: import_zod4.z.boolean().default(true).describe("Visit each business website (home + contact pages) to harvest email and social links. Uses the proxy/browser-backed extractor so blocked sites still resolve."),
1094
+ hydrateReviewCounts: import_zod4.z.boolean().default(true).describe("Deep-dive each profile to confirm the review count and booking URL that the Maps search list omits."),
1095
+ concurrency: import_zod4.z.number().int().min(1).max(4).default(3).describe("How many businesses to enrich in parallel. Keep low to respect per-account concurrency limits."),
1096
+ proxyMode: import_zod4.z.enum(["configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE).describe("Proxy behavior for the Maps search. Leave unset for clean egress; country/region localization comes from gl/hl plus the city or region in the query.")
1097
+ });
1098
+ var LEADS_CSV_HEADERS = [
1099
+ "position",
1100
+ "business_name",
1101
+ "review_stars",
1102
+ "review_count",
1103
+ "category",
1104
+ "address",
1105
+ "phone",
1106
+ "website",
1107
+ "email",
1108
+ "email_domain_match",
1109
+ "email_status",
1110
+ "socials",
1111
+ "booking_url",
1112
+ "place_url",
1113
+ "cid",
1114
+ "source_location"
1115
+ ];
1116
+ var EMAIL_RE = /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/g;
1117
+ var MAILTO_RE = /mailto:([^"'?>\s]+)/gi;
1118
+ var SOCIAL_RE = /https?:\/\/(?:www\.)?(?:facebook|instagram|linkedin|twitter|youtube|tiktok)\.com\/[^\s"'<>)]+/gi;
1119
+ var EMAIL_BAD = [".png", ".jpg", ".jpeg", ".gif", ".svg", ".webp", ".ico", "sentry", "wix.com", "wixpress", "example.", "godaddy", "squarespace", "schema.org", "@2x", "u003e", "core-js"];
1120
+ var EMAIL_PREFIX = ["info@", "office@", "contact@", "sales@", "hello@", "admin@", "service@", "support@"];
1121
+ var SOCIAL_SKIP = ["/sharer", "/share?", "/intent/", "plugins/", "/dialog/", "platform.", "badge", "developers."];
1122
+ async function mapLimit(items, limit, fn) {
1123
+ const out = new Array(items.length);
1124
+ let next = 0;
1125
+ async function worker() {
1126
+ while (next < items.length) {
1127
+ const index = next;
1128
+ next += 1;
1129
+ out[index] = await fn(items[index], index);
1130
+ }
1131
+ }
1132
+ await Promise.all(Array.from({ length: Math.min(limit, items.length) }, () => worker()));
1133
+ return out;
1134
+ }
1135
+ function hostRoot(url) {
1136
+ try {
1137
+ return new URL(url).hostname.replace(/^www\./, "").split(".")[0].toLowerCase();
1138
+ } catch {
1139
+ return "";
1140
+ }
1141
+ }
1142
+ function cleanEmails(raw) {
1143
+ const out = /* @__PURE__ */ new Set();
1144
+ for (const value of raw) {
1145
+ const email = value.trim().toLowerCase().replace(/\.$/, "");
1146
+ if (email.split("@").length !== 2) continue;
1147
+ if (email.length > 60) continue;
1148
+ if (EMAIL_BAD.some((bad) => email.includes(bad))) continue;
1149
+ out.add(email);
1150
+ }
1151
+ return [...out];
1152
+ }
1153
+ function pickEmail(emails, host) {
1154
+ if (!emails.length) return "";
1155
+ const root = hostRoot(host);
1156
+ for (const prefix of EMAIL_PREFIX) {
1157
+ const hit = emails.find((e) => e.startsWith(prefix) && root && e.split("@")[1].includes(root));
1158
+ if (hit) return hit;
1159
+ }
1160
+ const domainHit = emails.find((e) => root && e.split("@")[1].includes(root));
1161
+ if (domainHit) return domainHit;
1162
+ for (const prefix of EMAIL_PREFIX) {
1163
+ const hit = emails.find((e) => e.startsWith(prefix));
1164
+ if (hit) return hit;
1165
+ }
1166
+ return emails[0];
1167
+ }
1168
+ function parseContacts(html) {
1169
+ const mailtos = [...html.matchAll(MAILTO_RE)].map((m) => decodeURIComponent(m[1]).split("?")[0]);
1170
+ const inline = html.match(EMAIL_RE) ?? [];
1171
+ const emails = cleanEmails([...mailtos, ...inline]);
1172
+ const socials = [...new Set((html.match(SOCIAL_RE) ?? []).map((s) => s.replace(/[)"'<>]+$/, "")).filter((s) => !SOCIAL_SKIP.some((skip) => s.toLowerCase().includes(skip))))].slice(0, 5);
1173
+ return { emails, socials };
1174
+ }
1175
+ function buildAttemptUrls(website) {
1176
+ const urls = /* @__PURE__ */ new Set();
1177
+ urls.add(website);
1178
+ try {
1179
+ const origin = new URL(website).origin;
1180
+ urls.add(`${origin}/contact`);
1181
+ urls.add(`${origin}/contact-us`);
1182
+ } catch {
1183
+ return [...urls];
1184
+ }
1185
+ return [...urls].slice(0, 3);
1186
+ }
1187
+ var getLeadsWorkflowDefinition = {
1188
+ id: "get-leads",
1189
+ title: "Get Leads",
1190
+ description: "Build an outreach-ready local lead list: Google Maps search for a niche in a market, confirm review counts + booking URLs, then visit each business website (home + contact) to harvest email and social links, with proxy/browser-backed extraction so blocked sites still resolve. Saves a leads CSV.",
1191
+ inputSchema: GetLeadsInputSchema,
1192
+ async run(input, ctx) {
1193
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1194
+ const warnings = [];
1195
+ const errors = [];
1196
+ const search = await ctx.client.post("/maps/search", {
1197
+ query: input.query,
1198
+ location: input.location,
1199
+ maxResults: input.maxResults,
1200
+ proxyMode: input.proxyMode
1201
+ }, 18e4);
1202
+ await ctx.artifacts.writeJson("Maps search raw", "raw/maps-search.json", search);
1203
+ const seen = /* @__PURE__ */ new Set();
1204
+ const deduped = search.results.filter((result) => {
1205
+ const key = result.cid ?? result.placeUrl ?? result.name.toLowerCase();
1206
+ if (seen.has(key)) return false;
1207
+ seen.add(key);
1208
+ return true;
1209
+ });
1210
+ const rows = await mapLimit(deduped, input.concurrency, async (result) => {
1211
+ let reviewCount = result.reviewCount ?? "";
1212
+ let bookingUrl = "";
1213
+ let website = result.websiteUrl ?? "";
1214
+ if (input.hydrateReviewCounts) {
1215
+ try {
1216
+ const place = await ctx.client.post("/maps/place", {
1217
+ businessName: result.name,
1218
+ location: input.location,
1219
+ includeReviews: false
1220
+ }, 18e4);
1221
+ reviewCount = place.reviewCount ?? reviewCount;
1222
+ bookingUrl = place.bookingUrl ?? "";
1223
+ website = website || (place.website ?? "");
1224
+ } catch (err) {
1225
+ warnings.push(`Review hydration failed for ${result.name}: ${err instanceof Error ? err.message : String(err)}`);
1226
+ }
1227
+ }
1228
+ let email = "";
1229
+ let socials = [];
1230
+ let emailStatus = website ? "no email found" : "no website";
1231
+ let domainMatch = "";
1232
+ if (input.enrichWebsites && website) {
1233
+ const attempts = buildAttemptUrls(website);
1234
+ let reached = false;
1235
+ const allEmails = [];
1236
+ const allSocials = /* @__PURE__ */ new Set();
1237
+ for (const attemptUrl of attempts) {
1238
+ try {
1239
+ const page = await ctx.client.post("/extract-url", { url: attemptUrl }, 12e4);
1240
+ reached = true;
1241
+ const parsed = parseContacts(page.bodyHtml ?? page.bodyMarkdown ?? "");
1242
+ allEmails.push(...parsed.emails);
1243
+ parsed.socials.forEach((s) => allSocials.add(s));
1244
+ if (parsed.emails.length) break;
1245
+ } catch {
1246
+ continue;
1247
+ }
1248
+ }
1249
+ socials = [...allSocials].slice(0, 5);
1250
+ email = pickEmail([...new Set(allEmails)], website);
1251
+ if (email) {
1252
+ domainMatch = hostRoot(website) && email.split("@")[1].includes(hostRoot(website)) ? "yes" : "NO-verify";
1253
+ emailStatus = domainMatch === "yes" ? "ok" : "ok (off-domain \u2014 verify)";
1254
+ } else if (!reached) {
1255
+ emailStatus = "site unreachable";
1256
+ } else if (socials.length) {
1257
+ emailStatus = "form/social only";
1258
+ }
1259
+ }
1260
+ return {
1261
+ position: result.position,
1262
+ business_name: result.name,
1263
+ review_stars: result.rating ?? "",
1264
+ review_count: reviewCount,
1265
+ category: result.category ?? "",
1266
+ address: result.address ?? "",
1267
+ phone: result.phone ?? "",
1268
+ website,
1269
+ email,
1270
+ email_domain_match: domainMatch,
1271
+ email_status: emailStatus,
1272
+ socials: socials.join(" ; "),
1273
+ booking_url: bookingUrl,
1274
+ place_url: result.placeUrl,
1275
+ cid: result.cid ?? "",
1276
+ source_location: input.location
1277
+ };
1278
+ });
1279
+ await ctx.artifacts.writeCsv("Leads CSV", "exports/leads.csv", LEADS_CSV_HEADERS, rows);
1280
+ await ctx.artifacts.writeJson("Leads evidence", "evidence.json", { input, search: { query: search.searchQuery, resultCount: search.resultCount }, rows });
1281
+ const withEmail = rows.filter((r) => r.email).length;
1282
+ const domainMatched = rows.filter((r) => r.email_domain_match === "yes").length;
1283
+ const withReviewCount = rows.filter((r) => r.review_count).length;
1284
+ const withSocials = rows.filter((r) => r.socials).length;
1285
+ const withWebsite = rows.filter((r) => r.website).length;
1286
+ const summary = `${rows.length} ${input.query} in ${input.location}: ${withEmail} emails (${domainMatched} domain-matched), ${withReviewCount} review counts, ${withSocials} with socials.`;
1287
+ await ctx.artifacts.writeText("Summary", "summary.md", `# Get Leads
1288
+
1289
+ ${summary}
1290
+ `);
1291
+ await ctx.artifacts.writeText("Agent tasks", "tasks.md", [
1292
+ "# Get Leads \u2014 next steps",
1293
+ "",
1294
+ "- [ ] Verify deliverability of the harvested emails before outreach (MX/SMTP check).",
1295
+ '- [ ] For "form/social only" rows, use the website contact form or the captured social profile.',
1296
+ '- [ ] Treat "off-domain \u2014 verify" emails with care; they may belong to a web developer or parent brand.',
1297
+ "- [ ] Respect CAN-SPAM / GDPR \u2014 scraped contact data is not consent to email."
1298
+ ].join("\n"));
1299
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1300
+ title: "Get Leads",
1301
+ subtitle: `${input.query} \xB7 ${input.location}`,
1302
+ summary,
1303
+ warnings,
1304
+ tables: [
1305
+ {
1306
+ title: "Leads",
1307
+ columns: ["position", "business_name", "review_stars", "review_count", "phone", "website", "email", "email_status", "socials", "booking_url"],
1308
+ rows
1309
+ }
1310
+ ]
1311
+ }));
1312
+ const status2 = errors.length ? "failed" : warnings.length ? "partial" : "succeeded";
1313
+ const counts = {
1314
+ businesses: rows.length,
1315
+ withWebsite,
1316
+ emails: withEmail,
1317
+ domainMatchedEmails: domainMatched,
1318
+ reviewCounts: withReviewCount,
1319
+ withSocials
1320
+ };
1321
+ await ctx.artifacts.writeManifest(status2, counts, warnings, errors);
1322
+ return { title: "Get Leads", summary, status: status2, counts, warnings, errors, reportPath };
1323
+ }
1324
+ };
1325
+
1326
+ // src/workflows/workflows/local-competitive-audit.ts
1327
+ var import_zod5 = require("zod");
1328
+ var LocalCompetitiveAuditInputSchema = import_zod5.z.object({
1329
+ query: import_zod5.z.string().min(1),
1330
+ state: import_zod5.z.string().min(2).default("TN"),
1331
+ minPopulation: import_zod5.z.number().int().min(0).default(1e5),
1332
+ maxCities: import_zod5.z.number().int().min(1).max(100).default(25),
1333
+ maxResultsPerCity: import_zod5.z.number().int().min(1).max(50).default(20),
1334
+ hydrateTop: import_zod5.z.number().int().min(0).max(10).default(5),
1335
+ maxReviews: import_zod5.z.number().int().min(0).max(500).default(50),
1336
+ concurrency: import_zod5.z.number().int().min(1).max(5).default(5),
1337
+ proxyMode: import_zod5.z.enum(["configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE),
1338
+ returnPartial: import_zod5.z.boolean().default(true)
1339
+ });
1340
+ async function mapLimit2(items, limit, fn) {
1341
+ const out = new Array(items.length);
1342
+ let next = 0;
1343
+ async function worker() {
1344
+ while (next < items.length) {
1345
+ const index = next;
1346
+ next += 1;
1347
+ out[index] = await fn(items[index], index);
1348
+ }
1349
+ }
1350
+ await Promise.all(Array.from({ length: Math.min(limit, items.length) }, () => worker()));
1351
+ return out;
1352
+ }
1353
+ function numberFrom(value) {
1354
+ if (value === null || value === void 0 || value === "") return null;
1355
+ const parsed = Number(String(value).replace(/[^\d.]/g, ""));
1356
+ return Number.isFinite(parsed) ? parsed : null;
1357
+ }
1358
+ function median(values) {
1359
+ const nums = values.filter((v) => v !== null).sort((a, b) => a - b);
1360
+ if (!nums.length) return null;
1361
+ return nums[Math.floor(nums.length / 2)] ?? null;
1362
+ }
1363
+ function termsFrom(texts) {
1364
+ const stop = /* @__PURE__ */ new Set(["the", "and", "for", "with", "that", "this", "was", "were", "are", "you", "our", "they", "had", "have", "not", "but", "from", "very", "great", "good"]);
1365
+ const counts = /* @__PURE__ */ new Map();
1366
+ for (const text of texts) {
1367
+ for (const token of text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/)) {
1368
+ if (token.length < 4 || stop.has(token)) continue;
1369
+ counts.set(token, (counts.get(token) ?? 0) + 1);
1370
+ }
1371
+ }
1372
+ return [...counts.entries()].sort((a, b) => b[1] - a[1]).slice(0, 8).map(([term, count]) => `${term} (${count})`).join("; ");
1373
+ }
1374
+ var localCompetitiveAuditWorkflowDefinition = {
1375
+ id: "local-competitive-audit",
1376
+ title: "Local Competitive Audit",
1377
+ description: "Audit local Maps competitors, categories, review counts, and review themes across city markets.",
1378
+ inputSchema: LocalCompetitiveAuditInputSchema,
1379
+ async run(input, ctx) {
1380
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1381
+ const warnings = [];
1382
+ const errors = [];
1383
+ const directory = await ctx.client.post("/directory/run", {
1384
+ query: input.query,
1385
+ state: input.state,
1386
+ minPopulation: input.minPopulation,
1387
+ maxCities: input.maxCities,
1388
+ maxResultsPerCity: input.maxResultsPerCity,
1389
+ concurrency: input.concurrency,
1390
+ proxyMode: input.proxyMode,
1391
+ saveCsv: true,
1392
+ background: false
1393
+ }, 9e5);
1394
+ await ctx.artifacts.writeJson("Directory raw JSON", "raw/directory-workflow.json", directory);
1395
+ const baseRows = directoryRows(directory);
1396
+ await ctx.artifacts.writeCsv("Base directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, baseRows);
1397
+ const selected = directory.cities.flatMap(
1398
+ (city) => city.results.slice(0, input.hydrateTop).map((result) => ({ city, result }))
1399
+ );
1400
+ const seen = /* @__PURE__ */ new Set();
1401
+ const deduped = selected.filter(({ city, result }) => {
1402
+ const key = result.cid ?? result.placeUrl ?? `${city.location}:${result.name.toLowerCase()}`;
1403
+ if (seen.has(key)) return false;
1404
+ seen.add(key);
1405
+ return true;
1406
+ });
1407
+ const hydrated = await mapLimit2(deduped, 3, async ({ city, result }, index) => {
1408
+ try {
1409
+ const detail = await ctx.client.post("/maps/place", {
1410
+ businessName: result.name,
1411
+ location: city.location,
1412
+ includeReviews: input.maxReviews > 0,
1413
+ maxReviews: Math.max(1, input.maxReviews)
1414
+ }, 18e4);
1415
+ await ctx.artifacts.writeJson(`${result.name} profile`, `raw/maps-place-intel/${index + 1}-${result.name.toLowerCase().replace(/[^a-z0-9]+/g, "-")}.json`, detail);
1416
+ return { city, result, detail, error: null };
1417
+ } catch (err) {
1418
+ const message = err instanceof Error ? err.message : String(err);
1419
+ warnings.push(`Profile hydration failed for ${result.name} (${city.location}): ${message}`);
1420
+ return { city, result, detail: null, error: message };
1421
+ }
1422
+ });
1423
+ const competitorRows2 = hydrated.map(({ city, result, detail, error }) => {
1424
+ const reviews = detail?.reviews ?? [];
1425
+ const ownerResponses = reviews.filter((r) => r.ownerResponse).length;
1426
+ return {
1427
+ city: city.city,
1428
+ state: city.state,
1429
+ source_location: city.location,
1430
+ result_position: result.position,
1431
+ business_name: result.name,
1432
+ category: detail?.category ?? result.category,
1433
+ review_stars: detail?.rating ?? result.rating,
1434
+ review_count: detail?.reviewCount ?? result.reviewCount,
1435
+ phone: result.phone,
1436
+ website_url: detail?.website ?? result.websiteUrl,
1437
+ place_url: result.placeUrl,
1438
+ cid: result.cid,
1439
+ cid_decimal: result.cidDecimal,
1440
+ hydrated: detail ? "true" : "false",
1441
+ reviews_status: detail?.reviewsStatus ?? "",
1442
+ review_topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1443
+ owner_response_rate: reviews.length ? (ownerResponses / reviews.length).toFixed(2) : "",
1444
+ error: error ?? ""
1445
+ };
1446
+ });
1447
+ const reviewInsightRows = hydrated.map(({ city, result, detail }) => {
1448
+ const reviews = detail?.reviews ?? [];
1449
+ const positive = reviews.filter((r) => Number(r.stars) >= 5 && r.text).map((r) => r.text);
1450
+ const negative = reviews.filter((r) => Number(r.stars) > 0 && Number(r.stars) <= 3 && r.text).map((r) => r.text);
1451
+ return {
1452
+ city: city.city,
1453
+ business_name: result.name,
1454
+ cid: result.cid,
1455
+ review_count: detail?.reviewCount ?? result.reviewCount,
1456
+ average_rating: detail?.rating ?? result.rating,
1457
+ topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1458
+ praise_terms: termsFrom(positive),
1459
+ complaint_terms: termsFrom(negative),
1460
+ positive_sample: positive[0]?.slice(0, 280) ?? "",
1461
+ negative_sample: negative[0]?.slice(0, 280) ?? ""
1462
+ };
1463
+ });
1464
+ const citySummaryRows = directory.cities.map((city) => {
1465
+ const resultReviewCounts = city.results.map((r) => numberFrom(r.reviewCount));
1466
+ const resultRatings = city.results.map((r) => numberFrom(r.rating));
1467
+ const topThreeReviews = city.results.slice(0, 3).map((r) => numberFrom(r.reviewCount)).filter((v) => v !== null);
1468
+ const categories = /* @__PURE__ */ new Map();
1469
+ for (const result of city.results) {
1470
+ if (!result.category) continue;
1471
+ categories.set(result.category, (categories.get(result.category) ?? 0) + 1);
1472
+ }
1473
+ const medReviews = median(resultReviewCounts);
1474
+ const medRating = median(resultRatings);
1475
+ const topThreeAvg = topThreeReviews.length ? Math.round(topThreeReviews.reduce((a, b) => a + b, 0) / topThreeReviews.length) : null;
1476
+ const difficulty = Math.min(100, Math.round((topThreeAvg ?? 0) / 20 + (medRating ?? 0) * 10 + city.results.filter((r) => r.websiteUrl).length));
1477
+ const opportunity = Math.max(0, 100 - difficulty);
1478
+ return {
1479
+ city: city.city,
1480
+ state: city.state,
1481
+ population: city.population,
1482
+ result_count: city.resultCount,
1483
+ median_review_count: medReviews ?? "",
1484
+ median_rating: medRating ?? "",
1485
+ top_three_average_review_count: topThreeAvg ?? "",
1486
+ top_categories: [...categories.entries()].sort((a, b) => b[1] - a[1]).slice(0, 5).map(([cat, count]) => `${cat} (${count})`).join("; "),
1487
+ opportunity_score: opportunity,
1488
+ difficulty_score: difficulty
1489
+ };
1490
+ });
1491
+ await ctx.artifacts.writeCsv("Competitors CSV", "competitors.csv", ["city", "state", "source_location", "result_position", "business_name", "category", "review_stars", "review_count", "phone", "website_url", "place_url", "cid", "cid_decimal", "hydrated", "reviews_status", "review_topics", "owner_response_rate", "error"], competitorRows2);
1492
+ await ctx.artifacts.writeCsv("Review insights CSV", "review-insights.csv", ["city", "business_name", "cid", "review_count", "average_rating", "topics", "praise_terms", "complaint_terms", "positive_sample", "negative_sample"], reviewInsightRows);
1493
+ await ctx.artifacts.writeCsv("City summary CSV", "city-summary.csv", ["city", "state", "population", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "opportunity_score", "difficulty_score"], citySummaryRows);
1494
+ await ctx.artifacts.writeJson("Audit evidence", "evidence.json", { input, directory, hydrated, citySummaryRows, competitorRows: competitorRows2, reviewInsightRows });
1495
+ await ctx.artifacts.writeText("Agent tasks", "tasks.md", [
1496
+ "# Local Competitive Audit Tasks",
1497
+ "",
1498
+ "- [ ] Review city opportunity and difficulty scores as heuristics, not absolute truth.",
1499
+ "- [ ] Use review topics and samples as customer-language evidence.",
1500
+ "- [ ] Compare target GBP category, review count, and website quality against top competitors.",
1501
+ "- [ ] Identify cities with low review-count leaders and weak website coverage."
1502
+ ].join("\n"));
1503
+ const summary = `${directory.selectedCityCount} cities, ${directory.totalResultCount} Maps results, ${hydrated.filter((h) => h.detail).length} hydrated profiles.`;
1504
+ await ctx.artifacts.writeText("Summary", "summary.md", `# Local Competitive Audit
1505
+
1506
+ ${summary}
1507
+ `);
1508
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1509
+ title: "Local Competitive Audit",
1510
+ subtitle: `${input.query} \xB7 ${input.state}`,
1511
+ summary,
1512
+ warnings,
1513
+ tables: [
1514
+ { title: "City Opportunity", columns: ["city", "state", "population", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "opportunity_score", "difficulty_score"], rows: citySummaryRows },
1515
+ { title: "Hydrated Competitors", columns: ["city", "result_position", "business_name", "category", "review_stars", "review_count", "hydrated", "review_topics"], rows: competitorRows2.slice(0, 100) },
1516
+ { title: "Review Insights", columns: ["city", "business_name", "topics", "praise_terms", "complaint_terms"], rows: reviewInsightRows.slice(0, 100) }
1517
+ ]
1518
+ }));
1519
+ const status2 = warnings.length || directory.cities.some((city) => city.status === "failed") ? "partial" : "succeeded";
1520
+ const counts = { cities: directory.selectedCityCount, mapsResults: directory.totalResultCount, hydratedProfiles: hydrated.filter((h) => h.detail).length };
1521
+ await ctx.artifacts.writeManifest(status2, counts, warnings, errors);
1522
+ return { title: "Local Competitive Audit", summary, status: status2, counts, warnings, errors, reportPath };
1523
+ }
1524
+ };
1525
+
1526
+ // src/workflows/workflows/comparison-briefs.ts
1527
+ var import_zod6 = require("zod");
1528
+
1529
+ // src/workflows/workflows/seo-workflow-utils.ts
1530
+ var STOP_WORDS = /* @__PURE__ */ new Set([
1531
+ "about",
1532
+ "after",
1533
+ "also",
1534
+ "because",
1535
+ "been",
1536
+ "best",
1537
+ "both",
1538
+ "from",
1539
+ "have",
1540
+ "into",
1541
+ "more",
1542
+ "most",
1543
+ "near",
1544
+ "only",
1545
+ "over",
1546
+ "than",
1547
+ "that",
1548
+ "their",
1549
+ "them",
1550
+ "then",
1551
+ "there",
1552
+ "these",
1553
+ "they",
1554
+ "this",
1555
+ "what",
1556
+ "when",
1557
+ "where",
1558
+ "which",
1559
+ "while",
1560
+ "with",
1561
+ "your",
1562
+ "will",
1563
+ "would",
1564
+ "should",
1565
+ "could",
1566
+ "does",
1567
+ "were",
1568
+ "cost",
1569
+ "costs"
1570
+ ]);
1571
+ function normalizeDomain2(value) {
1572
+ if (!value) return null;
1573
+ try {
1574
+ const url = new URL(value.includes("://") ? value : `https://${value}`);
1575
+ return url.hostname.replace(/^www\./, "").toLowerCase();
1576
+ } catch {
1577
+ return value.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0]?.toLowerCase() || null;
1578
+ }
1579
+ }
1580
+ function domainFromUrl2(url) {
1581
+ return normalizeDomain2(url) ?? "";
1582
+ }
1583
+ function numberFrom2(value) {
1584
+ if (value === null || value === void 0 || value === "") return null;
1585
+ const parsed = Number(String(value).replace(/[^\d.]/g, ""));
1586
+ return Number.isFinite(parsed) ? parsed : null;
1587
+ }
1588
+ function median2(values) {
1589
+ const nums = values.filter((v) => v !== null).sort((a, b) => a - b);
1590
+ if (!nums.length) return null;
1591
+ return nums[Math.floor(nums.length / 2)] ?? null;
1592
+ }
1593
+ function textTerms(text, limit = 12) {
1594
+ const counts = /* @__PURE__ */ new Map();
1595
+ for (const token of text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/)) {
1596
+ if (token.length < 4 || STOP_WORDS.has(token)) continue;
1597
+ counts.set(token, (counts.get(token) ?? 0) + 1);
1598
+ }
1599
+ return [...counts.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, limit).map(([term, count]) => `${term} (${count})`).join("; ");
1600
+ }
1601
+ function classifyQuestion(question) {
1602
+ const q = question.toLowerCase();
1603
+ if (/\b(cost|price|pricing|charge|expensive|cheap)\b/.test(q)) return "cost";
1604
+ if (/\b(best|top|recommended|reviews?|compare|versus|vs)\b/.test(q)) return "comparison";
1605
+ if (/\b(how|steps?|process|way to)\b/.test(q)) return "process";
1606
+ if (/\b(why|worth|important|benefit)\b/.test(q)) return "why";
1607
+ if (/\b(near me|city|local|nearby)\b/.test(q)) return "local";
1608
+ if (/\b(can|does|do|is|are|should|will)\b/.test(q)) return "decision";
1609
+ return "definition";
1610
+ }
1611
+ function questionRows(paa) {
1612
+ return (paa?.flat ?? []).filter((row) => row.question).map((row, index) => {
1613
+ const url = row.source_cite ?? "";
1614
+ const domain = normalizeDomain2(row.source_site ?? "") ?? domainFromUrl2(url);
1615
+ return {
1616
+ position: index + 1,
1617
+ intent: classifyQuestion(row.question ?? ""),
1618
+ question: row.question ?? "",
1619
+ answer_excerpt: (row.answer ?? "").slice(0, 320),
1620
+ source_title: row.source_title ?? "",
1621
+ source_domain: domain,
1622
+ source_url: url
1623
+ };
1624
+ });
1625
+ }
1626
+ function sourceDomainRows(rows) {
1627
+ const byDomain = /* @__PURE__ */ new Map();
1628
+ for (const row of rows) {
1629
+ const domain = String(row.source_domain ?? row.domain ?? "");
1630
+ if (!domain) continue;
1631
+ const entry = byDomain.get(domain) ?? { questions: 0, urls: /* @__PURE__ */ new Set(), intents: /* @__PURE__ */ new Map() };
1632
+ entry.questions += row.question ? 1 : 0;
1633
+ if (row.source_url || row.url) entry.urls.add(String(row.source_url ?? row.url));
1634
+ if (row.intent) entry.intents.set(String(row.intent), (entry.intents.get(String(row.intent)) ?? 0) + 1);
1635
+ byDomain.set(domain, entry);
1636
+ }
1637
+ return [...byDomain.entries()].map(([domain, entry]) => ({
1638
+ domain,
1639
+ question_mentions: entry.questions,
1640
+ source_url_count: entry.urls.size,
1641
+ top_intents: [...entry.intents.entries()].sort((a, b) => b[1] - a[1]).map(([intent, count]) => `${intent} (${count})`).join("; ")
1642
+ })).sort((a, b) => Number(b.question_mentions) - Number(a.question_mentions) || String(a.domain).localeCompare(String(b.domain)));
1643
+ }
1644
+ function splitSentences(text) {
1645
+ return (text ?? "").replace(/\s+/g, " ").split(/(?<=[.!?])\s+/).map((sentence) => sentence.trim()).filter((sentence) => sentence.length > 20).slice(0, 20);
1646
+ }
1647
+ function classifySentence(sentence) {
1648
+ const s = sentence.toLowerCase();
1649
+ if (/\bis\b|\bare\b|\bmeans\b|\brefers to\b/.test(s)) return "definition";
1650
+ if (/\binclude\b|\bconsider\b|\bfactors?\b|\bcriteria\b/.test(s)) return "criteria";
1651
+ if (/\bfirst\b|\bthen\b|\bsteps?\b|\bprocess\b/.test(s)) return "process";
1652
+ if (/\bvs\b|\bthan\b|\bcompare\b|\bdifference\b/.test(s)) return "comparison";
1653
+ if (/\bcost\b|\bprice\b|\baverage\b|\brange\b/.test(s)) return "cost";
1654
+ return "claim";
1655
+ }
1656
+ function pageSummaryRow(page, source) {
1657
+ return {
1658
+ position: source.position ?? "",
1659
+ domain: source.domain,
1660
+ url: source.url,
1661
+ serp_title: source.title ?? "",
1662
+ page_title: page.title ?? "",
1663
+ h1: page.h1 ?? "",
1664
+ meta_description: page.metaDescription ?? "",
1665
+ word_count: page.wordCount ?? "",
1666
+ heading_count: page.headings?.length ?? 0,
1667
+ schema_types: (page.schemaTypes ?? []).join("; ")
1668
+ };
1669
+ }
1670
+ async function mapLimit3(items, limit, fn) {
1671
+ const out = new Array(items.length);
1672
+ let next = 0;
1673
+ async function worker() {
1674
+ while (next < items.length) {
1675
+ const index = next;
1676
+ next += 1;
1677
+ out[index] = await fn(items[index], index);
1678
+ }
1679
+ }
1680
+ await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, () => worker()));
1681
+ return out;
1682
+ }
1683
+
1684
+ // src/workflows/workflows/comparison-briefs.ts
1685
+ var ProxyModeSchema = import_zod6.z.enum(["configured", "none"]);
1686
+ var MapComparisonInputSchema = import_zod6.z.object({
1687
+ query: import_zod6.z.string().min(1),
1688
+ location: import_zod6.z.string().optional(),
1689
+ state: import_zod6.z.string().optional(),
1690
+ minPopulation: import_zod6.z.number().int().min(0).default(1e5),
1691
+ maxCities: import_zod6.z.number().int().min(1).max(100).default(5),
1692
+ maxResultsPerCity: import_zod6.z.number().int().min(1).max(50).default(20),
1693
+ hydrateTop: import_zod6.z.number().int().min(0).max(10).default(5),
1694
+ maxReviews: import_zod6.z.number().int().min(0).max(500).default(25),
1695
+ concurrency: import_zod6.z.number().int().min(1).max(5).default(5),
1696
+ proxyMode: ProxyModeSchema.default(DEFAULT_MAPS_PROXY_MODE),
1697
+ returnPartial: import_zod6.z.boolean().default(true)
1698
+ }).refine((input) => input.location || input.state, {
1699
+ message: "Either location or state is required for map-comparison"
1700
+ });
1701
+ var SerpComparisonInputSchema = import_zod6.z.object({
1702
+ keyword: import_zod6.z.string().min(1),
1703
+ domain: import_zod6.z.string().optional(),
1704
+ url: import_zod6.z.string().url().optional(),
1705
+ location: import_zod6.z.string().optional(),
1706
+ maxResults: import_zod6.z.number().int().min(1).max(20).default(10),
1707
+ maxQuestions: import_zod6.z.number().int().min(1).max(200).default(40),
1708
+ extractTop: import_zod6.z.number().int().min(0).max(10).default(5),
1709
+ includePaa: import_zod6.z.boolean().default(true),
1710
+ includeAiOverview: import_zod6.z.boolean().default(true),
1711
+ returnPartial: import_zod6.z.boolean().default(true)
1712
+ });
1713
+ var PaaExpansionBriefInputSchema = import_zod6.z.object({
1714
+ keyword: import_zod6.z.string().min(1),
1715
+ location: import_zod6.z.string().optional(),
1716
+ maxQuestions: import_zod6.z.number().int().min(1).max(300).default(80),
1717
+ depth: import_zod6.z.number().int().min(1).max(6).default(3),
1718
+ returnPartial: import_zod6.z.boolean().default(true)
1719
+ });
1720
+ var AiOverviewLanguageInputSchema = import_zod6.z.object({
1721
+ keyword: import_zod6.z.string().min(1),
1722
+ domain: import_zod6.z.string().optional(),
1723
+ url: import_zod6.z.string().url().optional(),
1724
+ location: import_zod6.z.string().optional(),
1725
+ maxQuestions: import_zod6.z.number().int().min(1).max(200).default(40),
1726
+ extractTop: import_zod6.z.number().int().min(0).max(8).default(3),
1727
+ returnPartial: import_zod6.z.boolean().default(true)
1728
+ });
1729
+ function businessRowsFromMaps(location, query, results) {
1730
+ return results.map((result) => ({
1731
+ source_query: query,
1732
+ source_location: location,
1733
+ city: location.split(",")[0]?.trim() ?? location,
1734
+ state: location.split(",")[1]?.trim() ?? "",
1735
+ population: "",
1736
+ result_position: result.position,
1737
+ business_name: result.name,
1738
+ review_stars: result.rating ?? "",
1739
+ review_count: result.reviewCount ?? "",
1740
+ category: result.category ?? "",
1741
+ address: result.address ?? "",
1742
+ phone: result.phone ?? "",
1743
+ website_url: result.websiteUrl ?? "",
1744
+ place_url: result.placeUrl ?? "",
1745
+ cid: result.cid ?? "",
1746
+ cid_decimal: result.cidDecimal ?? "",
1747
+ result_status: "ok",
1748
+ error: ""
1749
+ }));
1750
+ }
1751
+ function marketRows(rows) {
1752
+ const byLocation = /* @__PURE__ */ new Map();
1753
+ for (const row of rows) {
1754
+ const key = String(row.source_location ?? row.location ?? "");
1755
+ if (!key) continue;
1756
+ const list = byLocation.get(key) ?? [];
1757
+ list.push(row);
1758
+ byLocation.set(key, list);
1759
+ }
1760
+ return [...byLocation.entries()].map(([location, list]) => {
1761
+ const counts = list.map((row) => numberFrom2(row.review_count));
1762
+ const ratings = list.map((row) => numberFrom2(row.review_stars));
1763
+ const topThree = list.slice(0, 3).map((row) => numberFrom2(row.review_count)).filter((v) => v !== null);
1764
+ const categories = /* @__PURE__ */ new Map();
1765
+ for (const row of list) {
1766
+ const category = String(row.category ?? "");
1767
+ if (category) categories.set(category, (categories.get(category) ?? 0) + 1);
1768
+ }
1769
+ return {
1770
+ source_location: location,
1771
+ result_count: list.filter((row) => row.business_name).length,
1772
+ median_review_count: median2(counts) ?? "",
1773
+ median_rating: median2(ratings) ?? "",
1774
+ top_three_average_review_count: topThree.length ? Math.round(topThree.reduce((a, b) => a + b, 0) / topThree.length) : "",
1775
+ top_categories: [...categories.entries()].sort((a, b) => b[1] - a[1]).slice(0, 5).map(([cat, count]) => `${cat} (${count})`).join("; "),
1776
+ websites_present: list.filter((row) => row.website_url).length
1777
+ };
1778
+ });
1779
+ }
1780
+ function comparisonRows(rows) {
1781
+ const benchmarkByLocation = /* @__PURE__ */ new Map();
1782
+ for (const market of marketRows(rows)) {
1783
+ benchmarkByLocation.set(String(market.source_location), numberFrom2(market.top_three_average_review_count) ?? 0);
1784
+ }
1785
+ return rows.filter((row) => row.business_name).map((row) => {
1786
+ const reviews = numberFrom2(row.review_count) ?? 0;
1787
+ const benchmark = benchmarkByLocation.get(String(row.source_location)) ?? 0;
1788
+ const websiteMissing = !row.website_url;
1789
+ const rank = numberFrom2(row.result_position) ?? 999;
1790
+ return {
1791
+ source_location: row.source_location,
1792
+ result_position: row.result_position,
1793
+ business_name: row.business_name,
1794
+ category: row.category,
1795
+ review_stars: row.review_stars,
1796
+ review_count: row.review_count,
1797
+ review_gap_to_top3_average: benchmark ? Math.max(0, benchmark - reviews) : "",
1798
+ website_url: row.website_url,
1799
+ place_url: row.place_url,
1800
+ comparison_note: rank <= 3 ? "visible leader" : websiteMissing ? "ranking without website" : reviews < benchmark ? "review-light competitor" : "visible competitor"
1801
+ };
1802
+ });
1803
+ }
1804
+ function organicRows(serp, targetDomain) {
1805
+ return (serp?.organicResults ?? []).map((result) => {
1806
+ const url = result.url ?? "";
1807
+ const domain = normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(url);
1808
+ return {
1809
+ position: result.position ?? "",
1810
+ title: result.title ?? "",
1811
+ url,
1812
+ domain,
1813
+ snippet: result.snippet ?? "",
1814
+ is_target: targetDomain ? domain === targetDomain : false
1815
+ };
1816
+ });
1817
+ }
1818
+ function pageGapRows(targetPage, competitorPages) {
1819
+ const targetHeadingText = new Set((targetPage?.headings ?? []).map((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim()));
1820
+ const targetTerms = new Set((targetPage?.headings ?? []).flatMap((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/).filter(Boolean)));
1821
+ const rows = [];
1822
+ for (const { source, page } of competitorPages) {
1823
+ for (const heading of page.headings ?? []) {
1824
+ if (heading.level > 3 || !heading.text) continue;
1825
+ const normalized = heading.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim();
1826
+ const terms = normalized.split(/\s+/).filter((term) => term.length > 3);
1827
+ const overlap = terms.filter((term) => targetTerms.has(term)).length;
1828
+ const covered = targetHeadingText.has(normalized) || overlap >= Math.max(2, Math.ceil(terms.length / 2));
1829
+ if (covered && targetPage) continue;
1830
+ rows.push({
1831
+ source_position: source.position ?? "",
1832
+ source_domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url),
1833
+ source_url: source.url ?? "",
1834
+ heading_level: heading.level,
1835
+ competitor_heading: heading.text,
1836
+ target_coverage: targetPage ? "not found in target headings" : "no target page extracted",
1837
+ terms: textTerms(heading.text, 6)
1838
+ });
1839
+ }
1840
+ }
1841
+ return rows.slice(0, 150);
1842
+ }
1843
+ async function extractPages(ctx, sources, warnings, labelPrefix) {
1844
+ return mapLimit3(sources.filter((source) => source.url), 2, async (source, index) => {
1845
+ try {
1846
+ const page = await ctx.client.post("/extract-url", { url: source.url }, 18e4);
1847
+ await ctx.artifacts.writeJson(`${labelPrefix} ${index + 1}`, `raw/extract-url/${labelPrefix.toLowerCase()}-${index + 1}.json`, page);
1848
+ return { source, page };
1849
+ } catch (err) {
1850
+ warnings.push(`Page extraction failed for ${source.url}: ${err instanceof Error ? err.message : String(err)}`);
1851
+ return null;
1852
+ }
1853
+ }).then((items) => items.filter((item) => item !== null));
1854
+ }
1855
+ var mapComparisonWorkflowDefinition = {
1856
+ id: "map-comparison",
1857
+ title: "Maps Comparison",
1858
+ description: "Compare Google Maps competitors by rank, reviews, stars, categories, websites, and profile/review signals.",
1859
+ inputSchema: MapComparisonInputSchema,
1860
+ async run(input, ctx) {
1861
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1862
+ const warnings = [];
1863
+ let rows;
1864
+ let directory = null;
1865
+ let mapsSearch = null;
1866
+ if (input.location) {
1867
+ mapsSearch = await ctx.client.post("/maps/search", {
1868
+ query: input.query,
1869
+ location: input.location,
1870
+ maxResults: input.maxResultsPerCity,
1871
+ proxyMode: input.proxyMode
1872
+ }, 24e4);
1873
+ rows = businessRowsFromMaps(input.location, input.query, mapsSearch.results);
1874
+ await ctx.artifacts.writeJson("Maps search raw JSON", "raw/maps-search.json", mapsSearch);
1875
+ } else {
1876
+ directory = await ctx.client.post("/directory/run", {
1877
+ query: input.query,
1878
+ state: input.state,
1879
+ minPopulation: input.minPopulation,
1880
+ maxCities: input.maxCities,
1881
+ maxResultsPerCity: input.maxResultsPerCity,
1882
+ concurrency: input.concurrency,
1883
+ proxyMode: input.proxyMode,
1884
+ saveCsv: true,
1885
+ background: false
1886
+ }, 9e5);
1887
+ rows = directoryRows(directory);
1888
+ await ctx.artifacts.writeJson("Directory raw JSON", "raw/directory-workflow.json", directory);
1889
+ await ctx.artifacts.writeCsv("Directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, rows);
1890
+ warnings.push(...directory.warnings);
1891
+ }
1892
+ const compareRows = comparisonRows(rows);
1893
+ const selected = compareRows.slice(0, input.hydrateTop * Math.max(1, input.location ? 1 : input.maxCities));
1894
+ const hydrated = await mapLimit3(selected, 3, async (row, index) => {
1895
+ try {
1896
+ const detail = await ctx.client.post("/maps/place", {
1897
+ businessName: row.business_name,
1898
+ location: row.source_location,
1899
+ includeReviews: input.maxReviews > 0,
1900
+ maxReviews: Math.max(1, input.maxReviews)
1901
+ }, 18e4);
1902
+ await ctx.artifacts.writeJson(`${row.business_name} profile`, `raw/maps-place-intel/${index + 1}-${String(row.business_name).toLowerCase().replace(/[^a-z0-9]+/g, "-")}.json`, detail);
1903
+ return { row, detail, error: "" };
1904
+ } catch (err) {
1905
+ const message = err instanceof Error ? err.message : String(err);
1906
+ warnings.push(`Profile hydration failed for ${row.business_name}: ${message}`);
1907
+ return { row, detail: null, error: message };
1908
+ }
1909
+ });
1910
+ const profileRows = hydrated.map(({ row, detail, error }) => ({
1911
+ source_location: row.source_location,
1912
+ result_position: row.result_position,
1913
+ business_name: row.business_name,
1914
+ category: detail?.category ?? row.category,
1915
+ review_stars: detail?.rating ?? row.review_stars,
1916
+ review_count: detail?.reviewCount ?? row.review_count,
1917
+ website_url: detail?.website ?? row.website_url,
1918
+ review_topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1919
+ about_attributes: (detail?.aboutAttributes ?? []).map((a) => `${a.section}: ${a.attribute}`).join("; "),
1920
+ reviews_status: detail?.reviewsStatus ?? "",
1921
+ error
1922
+ }));
1923
+ const markets = marketRows(rows);
1924
+ await ctx.artifacts.writeCsv("Maps results CSV", "maps-results.csv", ["source_query", "source_location", "city", "state", "population", "result_position", "business_name", "review_stars", "review_count", "category", "address", "phone", "website_url", "place_url", "cid", "cid_decimal", "result_status", "error"], rows);
1925
+ await ctx.artifacts.writeCsv("Comparison CSV", "map-comparison.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "website_url", "place_url", "comparison_note"], compareRows);
1926
+ await ctx.artifacts.writeCsv("Profile insights CSV", "profile-insights.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "website_url", "review_topics", "about_attributes", "reviews_status", "error"], profileRows);
1927
+ await ctx.artifacts.writeJson("Maps comparison evidence", "evidence.json", { input, directory, mapsSearch, rows, compareRows, profileRows, markets, warnings });
1928
+ await ctx.artifacts.writeText("Brief", "brief.md", [
1929
+ `# Maps Comparison: ${input.query}`,
1930
+ "",
1931
+ `Markets: ${markets.map((row) => row.source_location).join(", ")}`,
1932
+ "",
1933
+ "## How to Use",
1934
+ "- Compare rank position against review count and category patterns.",
1935
+ "- Treat review gaps and missing websites as opportunity signals, not guarantees.",
1936
+ "- Use profile topics and attributes as evidence for local content and GBP improvements."
1937
+ ].join("\n"));
1938
+ const summary = `${compareRows.length} Maps competitors compared across ${markets.length} market(s); ${profileRows.filter((row) => !row.error).length} profiles hydrated.`;
1939
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1940
+ title: "Maps Comparison",
1941
+ subtitle: `${input.query}${input.location ? ` \xB7 ${input.location}` : input.state ? ` \xB7 ${input.state}` : ""}`,
1942
+ summary,
1943
+ warnings,
1944
+ tables: [
1945
+ { title: "Market Benchmarks", columns: ["source_location", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "websites_present"], rows: markets },
1946
+ { title: "Competitor Comparison", columns: ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "comparison_note"], rows: compareRows.slice(0, 120) },
1947
+ { title: "Profile Insights", columns: ["source_location", "business_name", "review_topics", "about_attributes", "error"], rows: profileRows }
1948
+ ]
1949
+ }));
1950
+ const status2 = warnings.length ? "partial" : "succeeded";
1951
+ const counts = { markets: markets.length, competitors: compareRows.length, hydratedProfiles: profileRows.filter((row) => !row.error).length };
1952
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
1953
+ return { title: "Maps Comparison", summary, status: status2, counts, warnings, errors: [], reportPath };
1954
+ }
1955
+ };
1956
+ var serpComparisonWorkflowDefinition = {
1957
+ id: "serp-comparison",
1958
+ title: "SERP Comparison",
1959
+ description: "Compare ranking pages, SERP features, PAA evidence, AI Overview citations, and page-level content gaps.",
1960
+ inputSchema: SerpComparisonInputSchema,
1961
+ async run(input, ctx) {
1962
+ await ctx.artifacts.writeManifest("running", {}, [], []);
1963
+ const warnings = [];
1964
+ const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
1965
+ const serp = await ctx.client.post("/harvest/sync", {
1966
+ query: input.keyword,
1967
+ location: input.location,
1968
+ serpOnly: true,
1969
+ maxQuestions: 1,
1970
+ format: "json"
1971
+ }, 18e4);
1972
+ await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
1973
+ let paa = null;
1974
+ if (input.includePaa) {
1975
+ try {
1976
+ paa = await ctx.client.post("/harvest/sync", {
1977
+ query: input.keyword,
1978
+ location: input.location,
1979
+ maxQuestions: input.maxQuestions,
1980
+ format: "json"
1981
+ }, 28e4);
1982
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1983
+ } catch (err) {
1984
+ warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
1985
+ }
1986
+ }
1987
+ const organic = organicRows(serp, targetDomain).slice(0, input.maxResults);
1988
+ const organicSources = (serp.organicResults ?? []).slice(0, input.maxResults);
1989
+ const targetSource = input.url ? { position: 0, title: "Target page", url: input.url, domain: domainFromUrl2(input.url) } : organicSources.find((result) => targetDomain && (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) === targetDomain);
1990
+ const competitorSources = organicSources.filter((result) => result.url && result.url !== targetSource?.url).filter((result) => !targetDomain || (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) !== targetDomain).slice(0, input.extractTop);
1991
+ const targetPages = targetSource ? await extractPages(ctx, [targetSource], warnings, "Target") : [];
1992
+ const competitorPages = input.extractTop > 0 ? await extractPages(ctx, competitorSources, warnings, "Competitor") : [];
1993
+ const targetPage = targetPages[0]?.page ?? null;
1994
+ const pageRows = [
1995
+ ...targetPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title })),
1996
+ ...competitorPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }))
1997
+ ];
1998
+ const gaps = pageGapRows(targetPage, competitorPages);
1999
+ const questions = questionRows(paa);
2000
+ const aiRows = (serp.aiOverview?.citations ?? []).map((citation, index) => ({
2001
+ citation_position: index + 1,
2002
+ citation_text: citation.text ?? "",
2003
+ url: citation.href ?? "",
2004
+ domain: domainFromUrl2(citation.href ?? ""),
2005
+ is_target: targetDomain ? domainFromUrl2(citation.href ?? "") === targetDomain : false
2006
+ }));
2007
+ await ctx.artifacts.writeCsv("Organic results CSV", "organic-results.csv", ["position", "title", "url", "domain", "snippet", "is_target"], organic);
2008
+ await ctx.artifacts.writeCsv("Page comparison CSV", "page-comparison.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], pageRows);
2009
+ await ctx.artifacts.writeCsv("Content gaps CSV", "content-gaps.csv", ["source_position", "source_domain", "source_url", "heading_level", "competitor_heading", "target_coverage", "terms"], gaps);
2010
+ await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
2011
+ await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], aiRows);
2012
+ await ctx.artifacts.writeJson("SERP comparison evidence", "evidence.json", { input, serp, paa, organic, pageRows, gaps, questions, aiRows, warnings });
2013
+ await ctx.artifacts.writeText("Writer brief", "brief.md", [
2014
+ `# SERP Comparison Brief: ${input.keyword}`,
2015
+ "",
2016
+ `Target: ${targetDomain ?? input.url ?? "not specified"}`,
2017
+ `Location: ${input.location ?? "not specified"}`,
2018
+ "",
2019
+ "## Recommended Actions",
2020
+ "- Use `content-gaps.csv` to decide which missing sections deserve coverage.",
2021
+ "- Use `paa-questions.csv` for FAQ and answer-block candidates.",
2022
+ "- Use `ai-overview-citations.csv` to see whether the target is cited in AI Overview evidence.",
2023
+ "- Treat extracted page headings as evidence, not a complete semantic analysis."
2024
+ ].join("\n"));
2025
+ const summary = `${organic.length} organic results, ${pageRows.length} extracted pages, ${gaps.length} heading gaps, ${questions.length} PAA questions.`;
2026
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
2027
+ title: "SERP Comparison",
2028
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
2029
+ summary,
2030
+ warnings,
2031
+ tables: [
2032
+ { title: "Organic Results", columns: ["position", "title", "domain", "is_target"], rows: organic },
2033
+ { title: "Page Comparison", columns: ["position", "domain", "h1", "word_count", "heading_count", "schema_types"], rows: pageRows },
2034
+ { title: "Content Gaps", columns: ["source_position", "source_domain", "competitor_heading", "target_coverage", "terms"], rows: gaps.slice(0, 80) },
2035
+ { title: "PAA Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 80) }
2036
+ ]
2037
+ }));
2038
+ const status2 = warnings.length ? "partial" : "succeeded";
2039
+ const counts = { organic: organic.length, pages: pageRows.length, gaps: gaps.length, questions: questions.length, aiCitations: aiRows.length };
2040
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
2041
+ return { title: "SERP Comparison", summary, status: status2, counts, warnings, errors: [], reportPath };
2042
+ }
2043
+ };
2044
+ var paaExpansionBriefWorkflowDefinition = {
2045
+ id: "paa-expansion-brief",
2046
+ title: "PAA Expansion Brief",
2047
+ description: "Expand People Also Ask questions into an evidence-backed writer brief, section map, and source table.",
2048
+ inputSchema: PaaExpansionBriefInputSchema,
2049
+ async run(input, ctx) {
2050
+ await ctx.artifacts.writeManifest("running", {}, [], []);
2051
+ const warnings = [];
2052
+ const paa = await ctx.client.post("/harvest/sync", {
2053
+ query: input.keyword,
2054
+ location: input.location,
2055
+ maxQuestions: input.maxQuestions,
2056
+ depth: input.depth,
2057
+ format: "json"
2058
+ }, 3e5);
2059
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
2060
+ const questions = questionRows(paa);
2061
+ const sourceRows2 = sourceDomainRows(questions);
2062
+ const byIntent = /* @__PURE__ */ new Map();
2063
+ for (const row of questions) {
2064
+ const list = byIntent.get(String(row.intent)) ?? [];
2065
+ list.push(row);
2066
+ byIntent.set(String(row.intent), list);
2067
+ }
2068
+ const sectionRows = [...byIntent.entries()].map(([intent, rows]) => ({
2069
+ recommended_section: intent,
2070
+ question_count: rows.length,
2071
+ sample_questions: rows.slice(0, 5).map((row) => row.question).join(" | "),
2072
+ source_domains: [...new Set(rows.map((row) => row.source_domain).filter(Boolean))].slice(0, 5).join("; "),
2073
+ terms: textTerms(rows.map((row) => `${row.question} ${row.answer_excerpt}`).join(" "), 10)
2074
+ })).sort((a, b) => Number(b.question_count) - Number(a.question_count));
2075
+ await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
2076
+ await ctx.artifacts.writeCsv("Source domains CSV", "source-domains.csv", ["domain", "question_mentions", "source_url_count", "top_intents"], sourceRows2);
2077
+ await ctx.artifacts.writeCsv("Section map CSV", "section-map.csv", ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], sectionRows);
2078
+ await ctx.artifacts.writeJson("PAA brief evidence", "evidence.json", { input, paa, questions, sourceRows: sourceRows2, sectionRows });
2079
+ await ctx.artifacts.writeText("Writer brief", "writer-brief.md", [
2080
+ `# PAA Expansion Brief: ${input.keyword}`,
2081
+ "",
2082
+ `Location: ${input.location ?? "not specified"}`,
2083
+ "",
2084
+ "## Suggested Page Structure",
2085
+ ...sectionRows.map((row) => `- ${row.recommended_section}: answer ${row.question_count} related question(s). Sample: ${row.sample_questions}`),
2086
+ "",
2087
+ "## Writing Rules",
2088
+ "- Answer the highest-frequency question in the first 60 words of each section.",
2089
+ "- Use exact customer question language from `paa-questions.csv` for H2/H3 candidates.",
2090
+ "- Use `source-domains.csv` to identify which source types Google is already rewarding.",
2091
+ "- Do not invent citations; cite only rows that have a source URL."
2092
+ ].join("\n"));
2093
+ const summary = `${questions.length} PAA questions grouped into ${sectionRows.length} writing sections.`;
2094
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
2095
+ title: "PAA Expansion Brief",
2096
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
2097
+ summary,
2098
+ warnings,
2099
+ tables: [
2100
+ { title: "Section Map", columns: ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], rows: sectionRows },
2101
+ { title: "Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 120) },
2102
+ { title: "Source Domains", columns: ["domain", "question_mentions", "source_url_count", "top_intents"], rows: sourceRows2 }
2103
+ ]
2104
+ }));
2105
+ const counts = { questions: questions.length, sections: sectionRows.length, sourceDomains: sourceRows2.length };
2106
+ await ctx.artifacts.writeManifest("succeeded", counts, warnings, []);
2107
+ return { title: "PAA Expansion Brief", summary, status: "succeeded", counts, warnings, errors: [], reportPath };
2108
+ }
2109
+ };
2110
+ var aiOverviewLanguageWorkflowDefinition = {
2111
+ id: "ai-overview-language",
2112
+ title: "AI Overview Language Brief",
2113
+ description: "Turn AI Overview, citation, PAA, and ranking-page evidence into answer-block and citation-hook guidance.",
2114
+ inputSchema: AiOverviewLanguageInputSchema,
2115
+ async run(input, ctx) {
2116
+ await ctx.artifacts.writeManifest("running", {}, [], []);
2117
+ const warnings = [];
2118
+ const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
2119
+ const serp = await ctx.client.post("/harvest/sync", {
2120
+ query: input.keyword,
2121
+ location: input.location,
2122
+ serpOnly: true,
2123
+ maxQuestions: 1,
2124
+ format: "json"
2125
+ }, 18e4);
2126
+ await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
2127
+ let paa = null;
2128
+ try {
2129
+ paa = await ctx.client.post("/harvest/sync", {
2130
+ query: input.keyword,
2131
+ location: input.location,
2132
+ maxQuestions: input.maxQuestions,
2133
+ format: "json"
2134
+ }, 28e4);
2135
+ await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
2136
+ } catch (err) {
2137
+ warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
2138
+ }
2139
+ const citations = (serp.aiOverview?.citations ?? []).map((citation, index) => {
2140
+ const domain = domainFromUrl2(citation.href ?? "");
2141
+ return {
2142
+ citation_position: index + 1,
2143
+ citation_text: citation.text ?? "",
2144
+ url: citation.href ?? "",
2145
+ domain,
2146
+ is_target: targetDomain ? domain === targetDomain : false
2147
+ };
2148
+ });
2149
+ const aioSentences = splitSentences(serp.aiOverview?.text);
2150
+ const claimRows = aioSentences.map((sentence, index) => ({
2151
+ position: index + 1,
2152
+ claim_type: classifySentence(sentence),
2153
+ sentence,
2154
+ reusable_pattern: sentence.length > 140 ? `${sentence.slice(0, 140)}...` : sentence
2155
+ }));
2156
+ const questions = questionRows(paa);
2157
+ const citationSources = (serp.aiOverview?.citations ?? []).filter((citation) => citation.href).map((citation, index) => ({ position: index + 1, title: citation.text, url: citation.href, domain: domainFromUrl2(citation.href) })).slice(0, input.extractTop);
2158
+ const extractedCitations = input.extractTop > 0 ? await extractPages(ctx, citationSources, warnings, "Citation") : [];
2159
+ const extractedRows = extractedCitations.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }));
2160
+ const languageRows = [
2161
+ {
2162
+ block: "direct_answer",
2163
+ guidance: "Open with a 40-70 word answer that directly resolves the query before adding context.",
2164
+ evidence_basis: questions[0]?.question ?? input.keyword
2165
+ },
2166
+ {
2167
+ block: "criteria_or_steps",
2168
+ guidance: "List the criteria, steps, or decision factors Google is already compressing into AI Overview language.",
2169
+ evidence_basis: claimRows.filter((row) => ["criteria", "process"].includes(String(row.claim_type))).map((row) => row.sentence).slice(0, 3).join(" | ")
2170
+ },
2171
+ {
2172
+ block: "citation_hook",
2173
+ guidance: "Add source-worthy details competitors can cite: definitions, numbers, examples, process details, and named entity relationships.",
2174
+ evidence_basis: citations.map((row) => `${row.domain}: ${row.citation_text}`).slice(0, 5).join(" | ")
2175
+ },
2176
+ {
2177
+ block: "faq_followups",
2178
+ guidance: "Use PAA phrasing for follow-up sections so the page answers adjacent questions in Google language.",
2179
+ evidence_basis: questions.slice(0, 5).map((row) => row.question).join(" | ")
2180
+ }
2181
+ ];
2182
+ await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], citations);
2183
+ await ctx.artifacts.writeCsv("AI Overview claim patterns CSV", "claim-patterns.csv", ["position", "claim_type", "sentence", "reusable_pattern"], claimRows);
2184
+ await ctx.artifacts.writeCsv("Language guidance CSV", "language-guidance.csv", ["block", "guidance", "evidence_basis"], languageRows);
2185
+ await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
2186
+ await ctx.artifacts.writeCsv("Extracted citation pages CSV", "citation-pages.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], extractedRows);
2187
+ await ctx.artifacts.writeJson("AI Overview language evidence", "evidence.json", { input, serp, paa, citations, claimRows, languageRows, extractedRows, warnings });
2188
+ await ctx.artifacts.writeText("Answer block template", "answer-block-template.md", [
2189
+ `# AI Overview Language Brief: ${input.keyword}`,
2190
+ "",
2191
+ `AI Overview detected: ${serp.aiOverview?.detected ? "yes" : "no"}`,
2192
+ `Target cited: ${targetDomain ? citations.some((row) => row.is_target) ? "yes" : "no" : "target not specified"}`,
2193
+ "",
2194
+ "## Direct Answer Block",
2195
+ "Write one compact answer block that starts with the answer, not background. Keep it clear enough that Google could lift it as a standalone summary.",
2196
+ "",
2197
+ "## Suggested Follow-Up Blocks",
2198
+ ...languageRows.map((row) => `- ${row.block}: ${row.guidance}`),
2199
+ "",
2200
+ "## Evidence to Mirror",
2201
+ ...claimRows.slice(0, 8).map((row) => `- ${row.claim_type}: ${row.sentence}`),
2202
+ "",
2203
+ "## Citation Hooks",
2204
+ ...citations.slice(0, 8).map((row) => `- ${row.domain}: ${row.citation_text}`)
2205
+ ].join("\n"));
2206
+ const summary = `${citations.length} AI Overview citations, ${claimRows.length} claim patterns, ${questions.length} PAA questions, ${extractedRows.length} citation pages extracted.`;
2207
+ const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
2208
+ title: "AI Overview Language Brief",
2209
+ subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
2210
+ summary,
2211
+ warnings,
2212
+ tables: [
2213
+ { title: "Language Guidance", columns: ["block", "guidance", "evidence_basis"], rows: languageRows },
2214
+ { title: "AI Overview Citations", columns: ["citation_position", "citation_text", "domain", "is_target"], rows: citations },
2215
+ { title: "Claim Patterns", columns: ["position", "claim_type", "sentence"], rows: claimRows },
2216
+ { title: "PAA Follow-Ups", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 60) }
2217
+ ]
2218
+ }));
2219
+ const status2 = warnings.length || !serp.aiOverview?.detected ? "partial" : "succeeded";
2220
+ const counts = { citations: citations.length, claimPatterns: claimRows.length, questions: questions.length, extractedCitationPages: extractedRows.length };
2221
+ await ctx.artifacts.writeManifest(status2, counts, warnings, []);
2222
+ return { title: "AI Overview Language Brief", summary, status: status2, counts, warnings, errors: [], reportPath };
2223
+ }
2224
+ };
2225
+
2226
+ // src/workflows/registry.ts
2227
+ var DEFINITIONS = [
2228
+ directoryWorkflowDefinition,
2229
+ getLeadsWorkflowDefinition,
2230
+ agentPacketWorkflowDefinition,
2231
+ localCompetitiveAuditWorkflowDefinition,
2232
+ mapComparisonWorkflowDefinition,
2233
+ serpComparisonWorkflowDefinition,
2234
+ paaExpansionBriefWorkflowDefinition,
2235
+ aiOverviewLanguageWorkflowDefinition
2236
+ ];
2237
+ function listWorkflowDefinitions() {
2238
+ return DEFINITIONS.map(({ id, title, description }) => ({ id, title, description }));
2239
+ }
2240
+ function workflowDefinition(id) {
2241
+ const definition = DEFINITIONS.find((def) => def.id === id);
2242
+ if (!definition) throw new Error(`Unknown workflow "${id}". Available: ${DEFINITIONS.map((def) => def.id).join(", ")}`);
2243
+ return definition;
2244
+ }
2245
+ function resolveApiKey(options) {
2246
+ const apiKey = options.apiKey?.trim() || process.env.MCP_SCRAPER_API_KEY?.trim();
2247
+ if (!apiKey) throw new Error("MCP_SCRAPER_API_KEY is required for workflow runs. Pass --api-key or set the environment variable.");
2248
+ return apiKey;
2249
+ }
2250
+ function resolveApiUrl(options) {
2251
+ return options.apiUrl?.trim() || process.env.MCP_SCRAPER_API_URL?.trim() || "https://mcpscraper.dev";
2252
+ }
2253
+ async function runWorkflow(id, rawInput, options = {}) {
2254
+ const definition = workflowDefinition(id);
2255
+ const input = definition.inputSchema.parse(rawInput);
2256
+ const apiKey = resolveApiKey(options);
2257
+ const apiUrl = resolveApiUrl(options);
2258
+ const artifacts = await ArtifactWriter.create(definition.id, definition.title, input, options.outputDir, options.runId);
2259
+ const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl, options.headers);
2260
+ const ctx = { runId: artifacts.runId, startedAt: artifacts.startedAt, client, artifacts, signal: options.signal };
2261
+ try {
2262
+ if (definition.steps && definition.steps.length > 0) {
2263
+ let state = definition.createState ? definition.createState(input) : {};
2264
+ let summary = null;
2265
+ for (const step of definition.steps) {
2266
+ const outcome = await step.run({ input, state, ctx });
2267
+ state = outcome.state;
2268
+ if (outcome.summary) summary = outcome.summary;
2269
+ }
2270
+ if (!summary) throw new Error(`Workflow "${id}" produced no terminal step summary`);
2271
+ return summary;
2272
+ }
2273
+ if (definition.run) return await definition.run(input, ctx);
2274
+ throw new Error(`Workflow "${id}" has neither steps nor a run() implementation`);
2275
+ } catch (err) {
2276
+ const message = err instanceof Error ? err.message : String(err);
2277
+ await artifacts.writeText("Failure", "summary.md", `# ${definition.title}
2278
+
2279
+ ${message}
2280
+ `);
2281
+ await artifacts.writeManifest("failed", {}, [], [message]);
2282
+ throw err;
2283
+ }
2284
+ }
2285
+
2286
+ // src/cli/human-cli.ts
2287
+ function numberOpt(value) {
2288
+ if (value === void 0 || value === null || value === "") return void 0;
2289
+ const parsed = Number(value);
2290
+ return Number.isFinite(parsed) ? parsed : void 0;
2291
+ }
2292
+ function booleanOpt(value) {
2293
+ if (value === void 0) return void 0;
2294
+ if (typeof value === "boolean") return value;
2295
+ if (value === "true") return true;
2296
+ if (value === "false") return false;
2297
+ return void 0;
2298
+ }
2299
+ function compactInput(input) {
2300
+ return Object.fromEntries(Object.entries(input).filter(([, value]) => value !== void 0));
2301
+ }
2302
+ function workflowInput(id, opts) {
2303
+ if (id === "agent-packet" || id === "serp-comparison" || id === "ai-overview-language") {
2304
+ return compactInput({
2305
+ keyword: opts.keyword,
2306
+ domain: opts.domain,
2307
+ url: opts.url,
2308
+ location: opts.location,
2309
+ maxResults: numberOpt(opts.maxResults),
2310
+ maxQuestions: numberOpt(opts.maxQuestions),
2311
+ extractTop: numberOpt(opts.extractTop),
2312
+ includeSerp: opts.serp === false ? false : void 0,
2313
+ includePaa: opts.paa === false ? false : void 0,
2314
+ includeAiOverview: booleanOpt(opts.includeAiOverview),
2315
+ returnPartial: opts.returnPartial === false ? false : void 0
2316
+ });
2317
+ }
2318
+ if (id === "paa-expansion-brief") {
2319
+ return compactInput({
2320
+ keyword: opts.keyword,
2321
+ location: opts.location,
2322
+ maxQuestions: numberOpt(opts.maxQuestions),
2323
+ depth: numberOpt(opts.depth),
2324
+ returnPartial: opts.returnPartial === false ? false : void 0
2325
+ });
2326
+ }
2327
+ if (id === "directory" || id === "local-competitive-audit" || id === "map-comparison") {
2328
+ return compactInput({
2329
+ query: opts.query,
2330
+ location: opts.location,
2331
+ state: opts.state,
2332
+ minPopulation: numberOpt(opts.minPop ?? opts.minPopulation),
2333
+ maxCities: numberOpt(opts.maxCities),
2334
+ maxResultsPerCity: numberOpt(opts.perCity ?? opts.maxResultsPerCity ?? opts.maxResults),
2335
+ concurrency: numberOpt(opts.concurrency),
2336
+ proxyMode: opts.proxyMode,
2337
+ hydrateTop: id === "local-competitive-audit" || id === "map-comparison" ? numberOpt(opts.hydrateTop) : void 0,
2338
+ maxReviews: id === "local-competitive-audit" || id === "map-comparison" ? numberOpt(opts.reviews ?? opts.maxReviews) : void 0,
2339
+ returnPartial: opts.returnPartial === false ? false : void 0
2340
+ });
2341
+ }
2342
+ return compactInput(opts);
2343
+ }
2344
+ function writeOutput(data, json) {
2345
+ if (json) {
2346
+ process.stdout.write(`${JSON.stringify(data, null, 2)}
2347
+ `);
2348
+ } else if (typeof data === "string") {
2349
+ process.stdout.write(`${data}
2350
+ `);
2351
+ } else {
2352
+ process.stdout.write(`${JSON.stringify(data, null, 2)}
2353
+ `);
2354
+ }
2355
+ }
2356
+ function maskSecrets(value) {
2357
+ return value.replace(/sk_[A-Za-z0-9_-]+/g, "sk_***");
2358
+ }
2359
+ function runLocalCommand(command, args) {
2360
+ return new Promise((resolve) => {
2361
+ const child = (0, import_node_child_process2.spawn)(command, args, { stdio: ["ignore", "pipe", "pipe"] });
2362
+ const stdout = [];
2363
+ const stderr = [];
2364
+ let settled = false;
2365
+ const done = (result) => {
2366
+ if (settled) return;
2367
+ settled = true;
2368
+ resolve(result);
2369
+ };
2370
+ child.stdout.on("data", (chunk) => stdout.push(Buffer.from(chunk)));
2371
+ child.stderr.on("data", (chunk) => stderr.push(Buffer.from(chunk)));
2372
+ child.on("error", (err) => {
2373
+ done({ code: 127, stdout: "", stderr: err.message });
2374
+ });
2375
+ child.on("close", (code) => {
2376
+ done({
2377
+ code: typeof code === "number" ? code : 1,
2378
+ stdout: Buffer.concat(stdout).toString("utf8"),
2379
+ stderr: Buffer.concat(stderr).toString("utf8")
2380
+ });
2381
+ });
2382
+ });
2383
+ }
2384
+ function cadenceOpt(opts) {
2385
+ if (opts.daily) return "daily";
2386
+ if (opts.monthly) return "monthly";
2387
+ return "weekly";
2388
+ }
2389
+ function apiOptions(opts) {
2390
+ const apiKey = String(opts.apiKey ?? process.env.MCP_SCRAPER_API_KEY ?? "").trim();
2391
+ if (!apiKey) throw new Error("MCP_SCRAPER_API_KEY is required. Pass --api-key or set the environment variable.");
2392
+ return {
2393
+ apiUrl: String(opts.apiUrl ?? process.env.MCP_SCRAPER_API_URL ?? "https://mcpscraper.dev").replace(/\/$/, ""),
2394
+ apiKey
2395
+ };
2396
+ }
2397
+ function openExternalUrl(url) {
2398
+ const command = process.platform === "darwin" ? "open" : process.platform === "win32" ? "cmd" : "xdg-open";
2399
+ const args = process.platform === "win32" ? ["/c", "start", "", url] : [url];
2400
+ try {
2401
+ const child = (0, import_node_child_process2.spawn)(command, args, { detached: true, stdio: "ignore" });
2402
+ child.unref();
2403
+ return true;
2404
+ } catch {
2405
+ return false;
2406
+ }
2407
+ }
2408
+ async function apiRequest(path, method, opts, body) {
2409
+ const { apiUrl, apiKey } = apiOptions(opts);
2410
+ const res = await fetch(`${apiUrl}${path}`, {
2411
+ method,
2412
+ headers: {
2413
+ "Content-Type": "application/json",
2414
+ "x-api-key": apiKey
2415
+ },
2416
+ body: body == null ? void 0 : JSON.stringify(body)
2417
+ });
2418
+ const data = await res.json().catch(() => ({}));
2419
+ if (!res.ok) {
2420
+ const message = typeof data.error === "string" ? data.error : `API request failed with ${res.status}`;
2421
+ throw new Error(message);
2422
+ }
2423
+ return data;
2424
+ }
2425
+ function addWorkflowInputOptions(command) {
2426
+ return command.option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--url <url>", "Target page URL").option("--location <location>", "Target location").option("--max-results <n>", "Maximum SERP or Maps results").option("--max-questions <n>", "Maximum PAA questions").option("--extract-top <n>", "Top result pages or citation pages to extract").option("--depth <n>", "PAA expansion depth").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts");
2427
+ }
2428
+ function buildHumanCli() {
2429
+ const program = new import_commander.Command();
2430
+ program.name("mcp-scraper-cli").description("Human CLI for MCP Scraper setup, workflows, reports, and agent-ready SEO artifacts.").version(PACKAGE_VERSION);
2431
+ program.command("doctor").description("Check setup, API key, output directory, package version, and recommended MCP config.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Output directory to test").option("--json", "Print machine-readable JSON").action(async (opts) => {
2432
+ const output = await runDoctor({ apiKey: opts.apiKey, apiUrl: opts.apiUrl, outputDir: opts.outputDir });
2433
+ writeOutput(opts.json ? output : renderDoctor(output), opts.json);
2434
+ if (!output.ok) process.exitCode = 1;
2435
+ });
2436
+ const billing = program.command("billing").description("Inspect billing and start checkout flows.");
2437
+ const billingConcurrency = billing.command("concurrency").description("Manage MCP Scraper concurrency slots.");
2438
+ billingConcurrency.command("info").description("Show current concurrency limit and extra-slot price.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (opts) => {
2439
+ const result = await apiRequest("/billing/credits", "POST", opts, {});
2440
+ if (opts.json) writeOutput(result.concurrency, true);
2441
+ else {
2442
+ const upgrade = result.concurrency?.upgrade;
2443
+ writeOutput([
2444
+ `Current limit: ${result.concurrency?.current_limit ?? "unknown"} concurrent operations`,
2445
+ `Extra slots: ${result.concurrency?.current_extra_slots ?? "unknown"}`,
2446
+ `Extra concurrency slot: ${upgrade?.price_label ?? "$5/month"}`,
2447
+ `Upgrade command: ${upgrade?.terminal_command ?? "mcp-scraper-cli billing concurrency checkout"}`
2448
+ ].join("\n"), false);
2449
+ }
2450
+ });
2451
+ billingConcurrency.command("checkout").description("Create a hosted Stripe checkout for one extra concurrency slot.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").option("--no-open", "Print the checkout URL without opening a browser").action(async (opts) => {
2452
+ const result = await apiRequest("/billing/concurrency/terminal-checkout", "POST", opts, {});
2453
+ if (opts.json) {
2454
+ writeOutput(result, true);
2455
+ return;
2456
+ }
2457
+ const opened = opts.open !== false && openExternalUrl(result.checkout_url);
2458
+ writeOutput([
2459
+ `Extra concurrency slot: ${result.price?.price_label ?? "$5/month"}`,
2460
+ opened ? `Opened checkout: ${result.checkout_url}` : `Checkout URL: ${result.checkout_url}`,
2461
+ result.next_step ?? "Complete checkout, then retry the MCP request."
2462
+ ].join("\n"), false);
2463
+ });
2464
+ billing.command("subscribe <tier>").description("Subscribe to a plan (starter | growth | scale) via a hosted Stripe checkout link.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").option("--no-open", "Print the checkout URL without opening a browser").action(async (tier, opts) => {
2465
+ const result = await apiRequest("/billing/subscribe/terminal-checkout", "POST", opts, { tier: String(tier).toLowerCase() });
2466
+ if (opts.json) {
2467
+ writeOutput(result, true);
2468
+ return;
2469
+ }
2470
+ if (result.updated) {
2471
+ writeOutput(result.message ?? `Switched to ${result.tier}.`, false);
2472
+ return;
2473
+ }
2474
+ const opened = opts.open !== false && !!result.checkout_url && openExternalUrl(result.checkout_url);
2475
+ const interval = result.billing_interval === "year" ? "year" : "month";
2476
+ const amount = result.amount_usd ?? result.monthly_usd;
2477
+ const credits = result.credits_per_interval ?? result.credits_per_month;
2478
+ const perInterval = interval === "year" ? "/yr" : "/mo";
2479
+ const creditsPer = interval === "year" ? "credits/yr" : "credits/mo";
2480
+ const extras = [
2481
+ result.credits_never_expire ? "credits never expire" : null,
2482
+ result.includes_memory ? "includes Memory Pro" : null,
2483
+ result.intro
2484
+ ].filter(Boolean).join(" \xB7 ");
2485
+ writeOutput([
2486
+ `Plan: ${result.label ?? tier} \u2014 $${amount}${perInterval} \xB7 ${credits?.toLocaleString()} ${creditsPer} \xB7 ${result.concurrency} concurrency${extras ? ` \xB7 ${extras}` : ""}`,
2487
+ opened ? `Opened checkout: ${result.checkout_url}` : `Checkout URL: ${result.checkout_url}`,
2488
+ result.next_step ?? "Complete payment in the browser."
2489
+ ].join("\n"), false);
2490
+ });
2491
+ const agent = program.command("agent").description("Generate AI-agent install configs and workflow prompts.");
2492
+ agent.command("install <host>").description("Print or apply install/config instructions for codex, claude/claude-code, or claude-desktop.").option("--api-key <key>", "API key to place in generated config").option("--package <spec>", "npm package spec", "mcp-scraper@latest").option("--browser-profile <name>", "Default saved browser profile for browser_open sessions.").option("--save-browser-profile-changes", "Persist cookies and browser storage back to the named browser profile when sessions close").option("--apply", "Apply the config to this client when supported. For Claude Code, this upserts the user-scope mcp-scraper server.").option("--json", "Print machine-readable JSON").action(async (hostInput, opts) => {
2493
+ const host = normalizeAgentHost(hostInput);
2494
+ const apiKey = opts.apiKey ?? process.env.MCP_SCRAPER_API_KEY;
2495
+ if (opts.apply) {
2496
+ if (host !== "claude") throw new Error('--apply is currently supported for Claude Code only. Use host "claude" or "claude-code".');
2497
+ if (!apiKey?.trim()) {
2498
+ throw new Error("MCP_SCRAPER_API_KEY is required for --apply. Set it in the environment or pass --api-key.");
2499
+ }
2500
+ const configOptions = {
2501
+ apiKey,
2502
+ packageSpec: opts.package,
2503
+ browserProfileName: opts.browserProfile,
2504
+ browserProfileSaveChanges: opts.saveBrowserProfileChanges
2505
+ };
2506
+ const existing = await runLocalCommand("claude", claudeMcpGetArgs());
2507
+ const snapshot = existing.code === 0 ? parseClaudeMcpGet(existing.stdout) : null;
2508
+ const remove = await runLocalCommand("claude", claudeMcpRemoveArgs());
2509
+ const add = await runLocalCommand("claude", claudeMcpAddArgs(configOptions));
2510
+ if (add.code !== 0) {
2511
+ const reason = maskSecrets([add.stderr, add.stdout].filter(Boolean).join("\n").trim());
2512
+ let rollback;
2513
+ if (remove.code !== 0) {
2514
+ rollback = "No existing entry was removed, so your configuration is unchanged.";
2515
+ } else if (!snapshot) {
2516
+ rollback = "WARNING: the previous mcp-scraper entry was removed and could not be captured for rollback. Re-add it manually.";
2517
+ } else {
2518
+ const restore = await runLocalCommand("claude", claudeMcpRestoreArgs(snapshot));
2519
+ rollback = restore.code === 0 ? "Your previous mcp-scraper entry was restored; nothing was lost." : "WARNING: the previous mcp-scraper entry was removed and could NOT be restored. Re-add it with:\n claude " + claudeMcpRestoreArgs(snapshot).join(" ");
2520
+ }
2521
+ throw new Error([
2522
+ "Claude Code MCP registration failed.",
2523
+ reason || "No error output returned.",
2524
+ rollback,
2525
+ "Make sure Claude Code is installed and the `claude` command is on PATH."
2526
+ ].join("\n"));
2527
+ }
2528
+ const list = await runLocalCommand("claude", ["mcp", "list"]);
2529
+ const result = {
2530
+ host: "claude",
2531
+ applied: true,
2532
+ replacedExisting: remove.code === 0,
2533
+ command: "npx",
2534
+ args: combinedNpxArgs(configOptions),
2535
+ nextStep: "Fully exit Claude Code, start a new Claude terminal, then run: claude mcp list",
2536
+ list: maskSecrets([list.stdout, list.stderr].filter(Boolean).join("\n").trim())
2537
+ };
2538
+ if (opts.json) {
2539
+ writeOutput(result, true);
2540
+ return;
2541
+ }
2542
+ writeOutput([
2543
+ "Applied Claude Code MCP config: mcp-scraper",
2544
+ remove.code === 0 ? "Replaced existing mcp-scraper entry." : "No existing mcp-scraper entry found; added a new one.",
2545
+ "Command: npx " + combinedNpxArgs(configOptions).join(" "),
2546
+ opts.browserProfile ? `Browser profile: ${opts.browserProfile}` : "",
2547
+ "",
2548
+ "Next step: fully exit Claude Code, start a new Claude terminal, then run:",
2549
+ " claude mcp list",
2550
+ "",
2551
+ result.list ? `Current Claude MCP list:
2552
+ ${result.list}` : ""
2553
+ ].filter(Boolean).join("\n"), false);
2554
+ return;
2555
+ }
2556
+ const text = renderAgentInstall(host, {
2557
+ apiKey: opts.apiKey,
2558
+ packageSpec: opts.package,
2559
+ browserProfileName: opts.browserProfile,
2560
+ browserProfileSaveChanges: opts.saveBrowserProfileChanges
2561
+ });
2562
+ writeOutput(opts.json ? { host, text } : text, opts.json);
2563
+ });
2564
+ agent.command("prompt [name]").description("Print an agent prompt template.").option("--json", "Print machine-readable JSON").action((name, opts) => {
2565
+ if (!name) {
2566
+ writeOutput(opts.json ? { prompts: listPrompts() } : listPrompts().join("\n"), opts.json);
2567
+ return;
2568
+ }
2569
+ const text = renderPrompt(name);
2570
+ writeOutput(opts.json ? { name, text } : text, opts.json);
2571
+ });
2572
+ const workflow = program.command("workflow").description("Run named SEO workflows.");
2573
+ workflow.command("list").description("List available workflows.").option("--json", "Print machine-readable JSON").action((opts) => {
2574
+ const rows = listWorkflowDefinitions();
2575
+ if (opts.json) writeOutput({ workflows: rows }, true);
2576
+ else writeOutput(rows.map((row) => `${row.id} ${row.title}
2577
+ ${row.description}`).join("\n"), false);
2578
+ });
2579
+ workflow.command("run <id>").description("Run a workflow and save local artifacts.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").option("--keyword <keyword>", "Agent packet keyword").option("--domain <domain>", "Target domain").option("--url <url>", "Target page URL").option("--location <location>", "Target location").option("--max-results <n>", "Maximum SERP or Maps results").option("--max-questions <n>", "Maximum PAA questions").option("--extract-top <n>", "Top result pages or citation pages to extract").option("--depth <n>", "PAA expansion depth").option("--no-serp", "Skip SERP evidence for agent packet").option("--no-paa", "Skip PAA evidence for agent packet").option("--query <query>", "Business category or workflow query").option("--state <state>", "US state").option("--min-pop <n>", "Minimum city population").option("--max-cities <n>", "Maximum selected cities").option("--per-city <n>", "Maps results per city").option("--concurrency <n>", "City search concurrency").option("--proxy-mode <mode>", "Proxy mode: location, configured, or none").option("--hydrate-top <n>", "Profiles to hydrate per city for competitive audits").option("--reviews <n>", "Review cards to collect per hydrated profile").option("--no-return-partial", "Fail instead of writing partial artifacts").action(async (id, opts) => {
2580
+ const summary = await runWorkflow(id, workflowInput(id, opts), {
2581
+ apiKey: opts.apiKey,
2582
+ apiUrl: opts.apiUrl,
2583
+ outputDir: opts.outputDir
2584
+ });
2585
+ if (opts.json) writeOutput(summary, true);
2586
+ else {
2587
+ writeOutput([
2588
+ `${summary.title}: ${summary.status}`,
2589
+ summary.summary,
2590
+ summary.reportPath ? `Report: ${summary.reportPath}` : "",
2591
+ summary.warnings.length ? `Warnings:
2592
+ ${summary.warnings.map((w) => `- ${w}`).join("\n")}` : ""
2593
+ ].filter(Boolean).join("\n"), false);
2594
+ }
2595
+ });
2596
+ const report = program.command("report").description("List and open local workflow reports.");
2597
+ report.command("list").description("List recent workflow reports.").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").action(async (opts) => {
2598
+ const reports = await listWorkflowReports(opts.outputDir);
2599
+ writeOutput(opts.json ? { reports } : reports.map((r) => `${r.startedAt} ${r.workflow} ${r.status} ${r.reportPath ?? r.manifestPath}`).join("\n"), opts.json);
2600
+ });
2601
+ report.command("path [id]").description("Print a report path. Defaults to last.").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON").action(async (id = "last", opts) => {
2602
+ const found = await findWorkflowReport(id, opts.outputDir);
2603
+ if (!found?.reportPath) throw new Error(`No report found for "${id}"`);
2604
+ writeOutput(opts.json ? { path: found.reportPath, run: found } : found.reportPath, opts.json);
2605
+ });
2606
+ report.command("open [id]").description("Open a report. Defaults to last.").option("--output-dir <path>", "Workflow output directory").option("--json", "Print machine-readable JSON instead of opening").action(async (id = "last", opts) => {
2607
+ if (opts.json) {
2608
+ const found = await findWorkflowReport(id, opts.outputDir);
2609
+ if (!found?.reportPath) throw new Error(`No report found for "${id}"`);
2610
+ writeOutput({ path: found.reportPath, run: found }, true);
2611
+ return;
2612
+ }
2613
+ const path = await openWorkflowReport(id, opts.outputDir);
2614
+ writeOutput(`Opened: ${path}`, false);
2615
+ });
2616
+ const schedule = program.command("schedule").description("Create and manage hosted workflow schedules.");
2617
+ addWorkflowInputOptions(schedule.command("create <workflowId>").description("Create a recurring hosted workflow schedule.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--name <name>", "Schedule name").option("--daily", "Run daily").option("--weekly", "Run weekly").option("--monthly", "Run monthly").option("--timezone <tz>", "Schedule timezone", "UTC").option("--webhook <url>", "HTTPS webhook URL").option("--next-run-at <iso>", "First run time as an ISO timestamp").option("--json", "Print machine-readable JSON")).action(async (workflowId, opts) => {
2618
+ const result = await apiRequest("/workflows/schedules", "POST", opts, {
2619
+ workflowId,
2620
+ name: opts.name,
2621
+ input: workflowInput(workflowId, opts),
2622
+ cadence: cadenceOpt(opts),
2623
+ timezone: opts.timezone,
2624
+ webhookUrl: opts.webhook,
2625
+ nextRunAt: opts.nextRunAt
2626
+ });
2627
+ writeOutput(result, opts.json);
2628
+ });
2629
+ schedule.command("list").description("List hosted workflow schedules.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (opts) => {
2630
+ const result = await apiRequest("/workflows/schedules", "GET", opts);
2631
+ if (opts.json) writeOutput(result, true);
2632
+ else writeOutput(result.schedules.map((scheduleRow) => `${scheduleRow.id} ${scheduleRow.status} ${scheduleRow.workflow_id} ${scheduleRow.next_run_at ?? ""}`).join("\n"), false);
2633
+ });
2634
+ schedule.command("pause <id>").description("Pause a hosted workflow schedule.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (id, opts) => writeOutput(await apiRequest(`/workflows/schedules/${id}`, "PATCH", opts, { status: "paused" }), opts.json));
2635
+ schedule.command("resume <id>").description("Resume a hosted workflow schedule.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (id, opts) => writeOutput(await apiRequest(`/workflows/schedules/${id}`, "PATCH", opts, { status: "active" }), opts.json));
2636
+ schedule.command("delete <id>").description("Delete a hosted workflow schedule.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (id, opts) => writeOutput(await apiRequest(`/workflows/schedules/${id}`, "DELETE", opts), opts.json));
2637
+ schedule.command("run <id>").description("Run a hosted workflow schedule now.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (id, opts) => writeOutput(await apiRequest(`/workflows/schedules/${id}/run`, "POST", opts, {}), opts.json));
2638
+ const runs = program.command("runs").description("Inspect and download hosted workflow runs.");
2639
+ runs.command("list").description("List hosted workflow runs.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (opts) => {
2640
+ const result = await apiRequest("/workflows/runs", "GET", opts);
2641
+ if (opts.json) writeOutput(result, true);
2642
+ else writeOutput(result.runs.map((run) => `${run.id} ${run.status} ${run.workflow_id} ${run.queued_at}`).join("\n"), false);
2643
+ });
2644
+ runs.command("status <id>").description("Show a hosted workflow run.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--json", "Print machine-readable JSON").action(async (id, opts) => writeOutput(await apiRequest(`/workflows/runs/${id}`, "GET", opts), opts.json));
2645
+ runs.command("download <id>").description("Download hosted workflow run artifacts.").option("--api-key <key>", "MCP Scraper API key").option("--api-url <url>", "MCP Scraper API URL", "https://mcpscraper.dev").option("--output-dir <path>", "Download directory").option("--json", "Print machine-readable JSON").action(async (id, opts) => {
2646
+ const result = await apiRequest(`/workflows/runs/${id}`, "GET", opts);
2647
+ const { apiUrl, apiKey } = apiOptions(opts);
2648
+ const outDir = (0, import_node_path3.join)(opts.outputDir ?? workflowOutputBaseDir(), "workflow-downloads", id);
2649
+ await (0, import_promises4.mkdir)(outDir, { recursive: true });
2650
+ const downloaded = [];
2651
+ for (const artifact of result.run.artifacts ?? []) {
2652
+ const res = await fetch(`${apiUrl}/workflows/runs/${id}/artifacts/${artifact.id}`, { headers: { "x-api-key": apiKey } });
2653
+ if (!res.ok) throw new Error(`Failed to download ${artifact.label}: HTTP ${res.status}`);
2654
+ const file = (0, import_node_path3.join)(outDir, (0, import_node_path3.basename)(artifact.path));
2655
+ await (0, import_promises4.writeFile)(file, Buffer.from(await res.arrayBuffer()));
2656
+ downloaded.push(file);
2657
+ }
2658
+ writeOutput(opts.json ? { runId: id, files: downloaded } : downloaded.join("\n"), opts.json);
2659
+ });
2660
+ return program;
2661
+ }
2662
+ async function runHumanCli(argv = process.argv) {
2663
+ await buildHumanCli().parseAsync(argv);
2664
+ }
2665
+
2666
+ // bin/mcp-scraper-cli.ts
2667
+ runHumanCli().catch((err) => {
2668
+ console.error(err instanceof Error ? err.message : String(err));
2669
+ process.exit(1);
2670
+ });
2671
+ //# sourceMappingURL=mcp-scraper-cli.cjs.map