mcp-scraper 0.88.2 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/CHANGELOG.md +36 -15
  2. package/README.md +6 -5
  3. package/THIRD_PARTY_NOTICES.html +203 -0
  4. package/dist/analytics-repository-2BMT5JNE.js +1 -0
  5. package/dist/bin/api-server.js +2 -41
  6. package/dist/bin/mcp-scraper-cli.js +39 -756
  7. package/dist/bin/mcp-scraper-core.js +1 -60
  8. package/dist/bin/mcp-scraper-install.js +2 -25
  9. package/dist/bin/mcp-stdio-server.js +1 -19
  10. package/dist/bin/paa-harvest.js +1 -41
  11. package/dist/chunk-3GP5CYZX.js +1 -0
  12. package/dist/chunk-4AI7DOS7.js +59 -0
  13. package/dist/chunk-4FROKQJN.js +1 -0
  14. package/dist/chunk-7YGVI5J4.js +21710 -0
  15. package/dist/chunk-CCYWSJNG.js +16 -0
  16. package/dist/chunk-CFI6CXIV.js +182 -0
  17. package/dist/chunk-E2WRWV3A.js +1 -0
  18. package/dist/chunk-GMWKPIYX.js +1172 -0
  19. package/dist/chunk-HDPYG3XV.js +102 -0
  20. package/dist/chunk-HE45FFBU.js +1 -0
  21. package/dist/chunk-HUV2WTRW.js +1 -0
  22. package/dist/chunk-KJQXUZ4Y.js +4 -0
  23. package/dist/chunk-L4CGLFPU.js +4 -0
  24. package/dist/chunk-M22MM4N4.js +84 -0
  25. package/dist/chunk-MASR22K4.js +73 -0
  26. package/dist/chunk-MZN4U5BL.js +1 -0
  27. package/dist/chunk-PUHFVA7P.js +1280 -0
  28. package/dist/chunk-QPWPR5XG.js +10 -0
  29. package/dist/chunk-TMB56NCA.js +1 -0
  30. package/dist/chunk-TXENITMS.js +20 -0
  31. package/dist/chunk-W2BVJ7S2.js +13 -0
  32. package/dist/chunk-WO3N5FH2.js +5 -0
  33. package/dist/chunk-WSCGYRWA.js +2595 -0
  34. package/dist/chunk-X54CQLK2.js +1 -0
  35. package/dist/chunk-XLWNEVUZ.js +27 -0
  36. package/dist/chunk-XPZVJIZ2.js +100 -0
  37. package/dist/chunk-YQZGZBB4.js +1 -0
  38. package/dist/chunk-Z2QGQJS2.js +1 -0
  39. package/dist/db-F2MX63GI.js +1 -0
  40. package/dist/extract-bundle-SNUIHM3J.js +26 -0
  41. package/dist/gmail-service-BZ3H75XC.js +1 -0
  42. package/dist/index.cjs +21750 -6045
  43. package/dist/index.d.cts +14 -14
  44. package/dist/index.d.ts +14 -14
  45. package/dist/index.js +18 -315
  46. package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
  47. package/dist/location-data-repository-OTWHWMV6.js +1 -0
  48. package/dist/server-RFR2A5UJ.js +7303 -0
  49. package/dist/site-extract-repository-SE776XDC.js +1 -0
  50. package/dist/worker-XUDSM3AL.js +1 -0
  51. package/package.json +17 -124
  52. package/dist/analytics-repository-GGJJCVVP.js +0 -194
  53. package/dist/chunk-4QMUF6XM.js +0 -1013
  54. package/dist/chunk-6HAV7LCE.js +0 -265
  55. package/dist/chunk-ABF2CGOZ.js +0 -113
  56. package/dist/chunk-C5Z4OFKW.js +0 -404
  57. package/dist/chunk-DNM65UCK.js +0 -299
  58. package/dist/chunk-EQGTEHLZ.js +0 -592
  59. package/dist/chunk-F5GQJWZU.js +0 -732
  60. package/dist/chunk-GGZEC22A.js +0 -215
  61. package/dist/chunk-GXBZXWXB.js +0 -184
  62. package/dist/chunk-IHXAXYIS.js +0 -843
  63. package/dist/chunk-K3Z5AQYE.js +0 -683
  64. package/dist/chunk-K45K75OF.js +0 -6
  65. package/dist/chunk-LFW2FRPJ.js +0 -224
  66. package/dist/chunk-MZDNZQWT.js +0 -2078
  67. package/dist/chunk-NVUKO5NN.js +0 -256
  68. package/dist/chunk-OM7HVEJ3.js +0 -26
  69. package/dist/chunk-OPQIGAFB.js +0 -286
  70. package/dist/chunk-OZJMVCDK.js +0 -16
  71. package/dist/chunk-P7FWOMU7.js +0 -505
  72. package/dist/chunk-PGJQDMC2.js +0 -383
  73. package/dist/chunk-PKZS6SHW.js +0 -33139
  74. package/dist/chunk-RJ7JVYKU.js +0 -68
  75. package/dist/chunk-S24LFPL7.js +0 -5262
  76. package/dist/chunk-T3MZISOF.js +0 -240
  77. package/dist/chunk-UZPTGUDV.js +0 -1915
  78. package/dist/chunk-X623GTBV.js +0 -8290
  79. package/dist/chunk-YXNDOQXN.js +0 -4018
  80. package/dist/db-Z34LPZNR.js +0 -284
  81. package/dist/extract-bundle-565SBZCR.js +0 -1003
  82. package/dist/gmail-service-E6ALS7JG.js +0 -25
  83. package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
  84. package/dist/location-data-repository-WPRG62GE.js +0 -34
  85. package/dist/server-SQZ3A7SY.js +0 -86606
  86. package/dist/site-extract-repository-VYFZASPU.js +0 -69
  87. package/dist/worker-LDCAULWL.js +0 -146
@@ -1,1915 +0,0 @@
1
- import {
2
- rowsToCsv
3
- } from "./chunk-RJ7JVYKU.js";
4
- import {
5
- DEFAULT_MAPS_PROXY_MODE,
6
- postToMemoryLibrary
7
- } from "./chunk-GXBZXWXB.js";
8
- import {
9
- buildSerpEmailQuery,
10
- contactAttemptUrls,
11
- extractContactEvidence,
12
- extractSerpEmailEvidence,
13
- selectPrimaryEmail
14
- } from "./chunk-P7FWOMU7.js";
15
-
16
- // src/lib/slugify.ts
17
- function slugify(s) {
18
- return s.toLowerCase().replace(/\s+/g, "-").replace(/[^a-z0-9-]/g, "");
19
- }
20
-
21
- // src/workflows/artifact-writer.ts
22
- import { mkdir, readFile, stat, writeFile } from "fs/promises";
23
- import { existsSync } from "fs";
24
- import { homedir, platform } from "os";
25
- import { dirname, join } from "path";
26
- import { execFile } from "child_process";
27
- function workflowOutputBaseDir(outputDir) {
28
- return outputDir?.trim() || process.env.MCP_SCRAPER_OUTPUT_DIR?.trim() || join(homedir(), "Downloads", "mcp-scraper");
29
- }
30
- function timestamp() {
31
- return (/* @__PURE__ */ new Date()).toISOString().replace(/[:.]/g, "-");
32
- }
33
- function safeSlug(value) {
34
- return slugify(value).replace(/^-+|-+$/g, "").slice(0, 80) || "run";
35
- }
36
- var ArtifactWriter = class _ArtifactWriter {
37
- constructor(workflowId, title, runId, baseDir, runDir, startedAt, input) {
38
- this.workflowId = workflowId;
39
- this.title = title;
40
- this.runId = runId;
41
- this.baseDir = baseDir;
42
- this.runDir = runDir;
43
- this.startedAt = startedAt;
44
- this.input = input;
45
- }
46
- workflowId;
47
- title;
48
- runId;
49
- baseDir;
50
- runDir;
51
- startedAt;
52
- input;
53
- artifacts = [];
54
- static async create(workflowId, title, input, outputDir, forcedRunId) {
55
- const startedAt = (/* @__PURE__ */ new Date()).toISOString();
56
- const runId = forcedRunId?.trim() || `${timestamp()}-${safeSlug(workflowId)}-${Math.random().toString(36).slice(2, 8)}`;
57
- const nameSource = String(input.keyword ?? input.query ?? input.domain ?? input.state ?? workflowId);
58
- const baseDir = workflowOutputBaseDir(outputDir);
59
- const runDir = join(baseDir, "workflows", workflowId, `${timestamp()}-${safeSlug(nameSource)}-${safeSlug(runId).slice(0, 20)}`);
60
- await mkdir(runDir, { recursive: true });
61
- return new _ArtifactWriter(workflowId, title, runId, baseDir, runDir, startedAt, input);
62
- }
63
- async remember(kind, label, path, rows) {
64
- const size = await stat(path).then((s) => s.size).catch(() => void 0);
65
- this.artifacts.push({ kind, label, path, bytes: size, rows });
66
- await postToMemoryLibrary({ title: `${this.title} ${label}`, content: `artifact: ${path}`, source: `mcp-scraper-workflow:${this.workflowId}:${this.runId}` });
67
- return path;
68
- }
69
- async writeJson(label, relativePath, data) {
70
- const path = join(this.runDir, relativePath);
71
- await mkdir(dirname(path), { recursive: true });
72
- await writeFile(path, JSON.stringify(data, null, 2), "utf8");
73
- return this.remember("json", label, path);
74
- }
75
- async writeText(label, relativePath, text, kind = "markdown") {
76
- const path = join(this.runDir, relativePath);
77
- await mkdir(dirname(path), { recursive: true });
78
- await writeFile(path, text, "utf8");
79
- return this.remember(kind, label, path);
80
- }
81
- async writeCsv(label, relativePath, headers, rows) {
82
- const path = join(this.runDir, relativePath);
83
- await mkdir(dirname(path), { recursive: true });
84
- await writeFile(path, rowsToCsv(headers, rows), "utf8");
85
- return this.remember("csv", label, path, rows.length);
86
- }
87
- async writeHtml(label, relativePath, html) {
88
- return this.writeText(label, relativePath, html, "html_report");
89
- }
90
- async writeManifest(status, counts, warnings, errors) {
91
- const manifest = {
92
- workflow: this.workflowId,
93
- title: this.title,
94
- runId: this.runId,
95
- status,
96
- startedAt: this.startedAt,
97
- completedAt: status === "running" ? null : (/* @__PURE__ */ new Date()).toISOString(),
98
- input: this.input,
99
- artifacts: this.artifacts,
100
- warnings,
101
- errors,
102
- counts
103
- };
104
- const path = join(this.runDir, "manifest.json");
105
- await writeFile(path, JSON.stringify(manifest, null, 2), "utf8");
106
- await updateWorkflowIndex(manifest, path, this.baseDir);
107
- return path;
108
- }
109
- };
110
- function indexPath(baseDir = workflowOutputBaseDir()) {
111
- return join(baseDir, "workflows", "index.json");
112
- }
113
- async function readIndex(baseDir) {
114
- const path = indexPath(baseDir);
115
- if (!existsSync(path)) return { runs: [] };
116
- try {
117
- return JSON.parse(await readFile(path, "utf8"));
118
- } catch {
119
- return { runs: [] };
120
- }
121
- }
122
- async function updateWorkflowIndex(manifest, manifestPath, baseDir) {
123
- if (manifest.status === "running") return;
124
- const path = indexPath(baseDir);
125
- const index = await readIndex(baseDir);
126
- const report = manifest.artifacts.find((a) => a.kind === "html_report")?.path ?? null;
127
- const entry = {
128
- workflow: manifest.workflow,
129
- runId: manifest.runId,
130
- status: manifest.status,
131
- startedAt: manifest.startedAt,
132
- reportPath: report,
133
- manifestPath,
134
- summary: `${manifest.title} (${manifest.status})`
135
- };
136
- index.runs = [entry, ...index.runs.filter((r) => r.runId !== manifest.runId)].slice(0, 200);
137
- await mkdir(dirname(path), { recursive: true });
138
- await writeFile(path, JSON.stringify(index, null, 2), "utf8");
139
- }
140
- async function listWorkflowReports(outputDir) {
141
- return (await readIndex(workflowOutputBaseDir(outputDir))).runs;
142
- }
143
- async function findWorkflowReport(id, outputDir) {
144
- const runs = await listWorkflowReports(outputDir);
145
- if (id === "last") return runs[0] ?? null;
146
- return runs.find((run) => run.runId === id) ?? null;
147
- }
148
- async function openWorkflowReport(id, outputDir) {
149
- const run = await findWorkflowReport(id, outputDir);
150
- if (!run?.reportPath) throw new Error(`No report found for "${id}"`);
151
- if (platform() === "darwin") {
152
- await new Promise((resolve, reject) => {
153
- execFile("open", [run.reportPath], (err) => err ? reject(err) : resolve());
154
- });
155
- }
156
- return run.reportPath;
157
- }
158
-
159
- // src/workflows/registry.ts
160
- import { readFile as readFile2 } from "fs/promises";
161
- import { z as z6 } from "zod";
162
-
163
- // src/workflows/http-client.ts
164
- var WorkflowHttpClient = class {
165
- constructor(apiUrl, apiKey, fetchImpl = fetch, extraHeaders = {}) {
166
- this.apiUrl = apiUrl;
167
- this.apiKey = apiKey;
168
- this.fetchImpl = fetchImpl;
169
- this.extraHeaders = extraHeaders;
170
- }
171
- apiUrl;
172
- apiKey;
173
- fetchImpl;
174
- extraHeaders;
175
- async post(path, body, timeoutMs = 18e4) {
176
- const res = await this.fetchImpl(`${this.apiUrl.replace(/\/$/, "")}${path}`, {
177
- method: "POST",
178
- headers: {
179
- ...this.extraHeaders,
180
- "Content-Type": "application/json",
181
- "x-api-key": this.apiKey
182
- },
183
- body: JSON.stringify(body),
184
- signal: AbortSignal.timeout(timeoutMs)
185
- });
186
- const data = await res.json().catch(() => ({}));
187
- if (!res.ok) {
188
- const message = typeof data.error === "string" ? data.error : `Workflow API ${path} failed with ${res.status}`;
189
- throw new Error(message);
190
- }
191
- return data;
192
- }
193
- };
194
-
195
- // src/workflows/workflows/agent-packet.ts
196
- import { z } from "zod";
197
-
198
- // src/workflows/report-renderer.ts
199
- function escapeHtml(value) {
200
- return String(value ?? "").replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/"/g, "&quot;").replace(/'/g, "&#39;");
201
- }
202
- function renderTable(table) {
203
- const rows = table.rows.map((row) => `<tr>${table.columns.map((col) => `<td>${escapeHtml(row[col])}</td>`).join("")}</tr>`).join("\n");
204
- return [
205
- `<section>`,
206
- `<h2>${escapeHtml(table.title)}</h2>`,
207
- `<div class="table-wrap"><table>`,
208
- `<thead><tr>${table.columns.map((col) => `<th>${escapeHtml(col)}</th>`).join("")}</tr></thead>`,
209
- `<tbody>${rows || `<tr><td colspan="${table.columns.length}">No rows</td></tr>`}</tbody>`,
210
- `</table></div>`,
211
- `</section>`
212
- ].join("\n");
213
- }
214
- function renderWorkflowReport(input) {
215
- const warnings = input.warnings?.length ? `<section class="warnings"><h2>Warnings</h2><ul>${input.warnings.map((w) => `<li>${escapeHtml(w)}</li>`).join("")}</ul></section>` : "";
216
- const sections = (input.sections ?? []).map((section) => `<section><h2>${escapeHtml(section.title)}</h2><p>${escapeHtml(section.body)}</p></section>`).join("\n");
217
- const tables = (input.tables ?? []).map(renderTable).join("\n");
218
- return [
219
- "<!doctype html>",
220
- '<html lang="en">',
221
- "<head>",
222
- '<meta charset="utf-8">',
223
- '<meta name="viewport" content="width=device-width, initial-scale=1">',
224
- `<title>${escapeHtml(input.title)}</title>`,
225
- "<style>",
226
- ':root{font-family:Inter,ui-sans-serif,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;color:#172026;background:#f6f7f8}',
227
- "body{margin:0;padding:32px}main{max-width:1180px;margin:0 auto;background:#fff;border:1px solid #d9dee3;border-radius:8px;padding:28px}",
228
- "h1{font-size:30px;line-height:1.15;margin:0 0 8px}h2{font-size:18px;margin:28px 0 12px}p{line-height:1.55;color:#3d4952}.sub{color:#66737f;margin:0 0 20px}",
229
- ".summary{background:#eef6f8;border-left:4px solid #12829a;padding:14px 16px;margin:20px 0}.warnings{background:#fff7e8;border-left:4px solid #c87b00;padding:12px 16px}",
230
- ".table-wrap{overflow:auto;border:1px solid #dfe5ea;border-radius:8px}table{border-collapse:collapse;width:100%;font-size:13px}th,td{border-bottom:1px solid #edf0f2;padding:9px 10px;text-align:left;vertical-align:top}th{background:#f3f5f7;color:#31404c;position:sticky;top:0}tr:last-child td{border-bottom:0}",
231
- "</style>",
232
- "</head>",
233
- "<body><main>",
234
- `<h1>${escapeHtml(input.title)}</h1>`,
235
- input.subtitle ? `<p class="sub">${escapeHtml(input.subtitle)}</p>` : "",
236
- `<div class="summary">${escapeHtml(input.summary)}</div>`,
237
- warnings,
238
- sections,
239
- tables,
240
- "</main></body></html>"
241
- ].join("\n");
242
- }
243
-
244
- // src/workflows/workflows/agent-packet.ts
245
- var AgentPacketInputSchema = z.object({
246
- keyword: z.string().min(1),
247
- domain: z.string().optional(),
248
- location: z.string().optional(),
249
- maxQuestions: z.number().int().min(1).max(200).default(40),
250
- includeSerp: z.boolean().default(true),
251
- includePaa: z.boolean().default(true),
252
- includeAiOverview: z.boolean().default(true),
253
- returnPartial: z.boolean().default(true)
254
- });
255
- function normalizeDomain(value) {
256
- if (!value) return null;
257
- try {
258
- const url = new URL(value.includes("://") ? value : `https://${value}`);
259
- return url.hostname.replace(/^www\./, "").toLowerCase();
260
- } catch {
261
- return value.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0]?.toLowerCase() || null;
262
- }
263
- }
264
- function domainFromUrl(url) {
265
- return normalizeDomain(url) ?? "";
266
- }
267
- function sourceRows(input, serp, paa) {
268
- const target = normalizeDomain(input.domain);
269
- const rows = [];
270
- for (const result of serp?.organicResults ?? []) {
271
- const domain = normalizeDomain(result.domain) ?? domainFromUrl(result.url);
272
- rows.push({
273
- surface: "organic",
274
- position: result.position,
275
- title: result.title,
276
- url: result.url,
277
- domain,
278
- question: "",
279
- answer_excerpt: result.snippet ?? "",
280
- is_target: target ? domain === target : false,
281
- is_competitor: target ? domain !== target : true
282
- });
283
- }
284
- for (const row of paa?.flat ?? []) {
285
- const url = row.source_cite ?? "";
286
- const domain = normalizeDomain(row.source_site ?? "") ?? (url ? domainFromUrl(url) : "");
287
- rows.push({
288
- surface: "paa",
289
- position: "",
290
- title: row.source_title ?? row.source_site ?? "",
291
- url,
292
- domain,
293
- question: row.question,
294
- answer_excerpt: (row.answer ?? "").slice(0, 240),
295
- is_target: target ? domain === target : false,
296
- is_competitor: target ? Boolean(domain && domain !== target) : Boolean(domain)
297
- });
298
- }
299
- for (const [i, cite] of (serp?.aiOverview?.citations ?? []).entries()) {
300
- const url = cite.href ?? "";
301
- const domain = url ? domainFromUrl(url) : "";
302
- rows.push({
303
- surface: "ai_overview",
304
- position: i + 1,
305
- title: cite.text ?? "",
306
- url,
307
- domain,
308
- question: "",
309
- answer_excerpt: "",
310
- is_target: target ? domain === target : false,
311
- is_competitor: target ? Boolean(domain && domain !== target) : Boolean(domain)
312
- });
313
- }
314
- return rows;
315
- }
316
- function competitorRows(rows, targetDomain) {
317
- const byDomain = /* @__PURE__ */ new Map();
318
- for (const row of rows) {
319
- const domain = String(row.domain ?? "");
320
- if (!domain || domain === targetDomain) continue;
321
- const entry = byDomain.get(domain) ?? { organic: [], paa: 0, ai: 0, urls: /* @__PURE__ */ new Set() };
322
- if (row.surface === "organic" && row.position) entry.organic.push(Number(row.position));
323
- if (row.surface === "paa") entry.paa += 1;
324
- if (row.surface === "ai_overview") entry.ai += 1;
325
- if (row.url) entry.urls.add(String(row.url));
326
- byDomain.set(domain, entry);
327
- }
328
- return [...byDomain.entries()].map(([domain, entry]) => ({
329
- domain,
330
- organic_best_position: entry.organic.length ? Math.min(...entry.organic) : "",
331
- organic_count: entry.organic.length,
332
- paa_mentions: entry.paa,
333
- ai_overview_citations: entry.ai,
334
- source_url_count: entry.urls.size
335
- })).sort((a, b) => Number(a.organic_best_position || 999) - Number(b.organic_best_position || 999));
336
- }
337
- var agentPacketWorkflowDefinition = {
338
- id: "agent-packet",
339
- title: "Agent-Ready SEO Packet",
340
- description: "Create an evidence folder for AI agents from live SERP/PAA/AI search surfaces.",
341
- inputSchema: AgentPacketInputSchema,
342
- createState: () => ({ serp: null, paa: null, warnings: [] }),
343
- steps: [
344
- {
345
- id: "harvest-serp",
346
- title: "Harvest organic SERP + AI Overview",
347
- async run({ input, state, ctx }) {
348
- if (!input.includeSerp) {
349
- return { state, output: { skipped: true, reason: "includeSerp is false" } };
350
- }
351
- try {
352
- const serp = await ctx.client.post("/harvest/sync", {
353
- query: input.keyword,
354
- location: input.location,
355
- serpOnly: true,
356
- includeAllSerpFeatures: true,
357
- maxQuestions: 1,
358
- format: "json"
359
- }, 62e4);
360
- await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
361
- return {
362
- state: { ...state, serp },
363
- output: {
364
- organicResults: serp.organicResults?.length ?? 0,
365
- localPack: serp.localPack?.length ?? 0,
366
- aiOverviewDetected: Boolean(serp.aiOverview?.detected),
367
- aiOverviewCitations: serp.aiOverview?.citations?.length ?? 0
368
- }
369
- };
370
- } catch (err) {
371
- const message = `SERP evidence unavailable: ${err instanceof Error ? err.message : String(err)}`;
372
- return { state: { ...state, warnings: [...state.warnings, message] }, output: { error: message }, warnings: [message] };
373
- }
374
- }
375
- },
376
- {
377
- id: "harvest-paa",
378
- title: "Harvest People Also Ask",
379
- async run({ input, state, ctx }) {
380
- if (!input.includePaa) {
381
- return { state, output: { skipped: true, reason: "includePaa is false" } };
382
- }
383
- try {
384
- const paa = await ctx.client.post("/harvest/sync", {
385
- query: input.keyword,
386
- location: input.location,
387
- maxQuestions: input.maxQuestions,
388
- format: "json"
389
- }, 735e3);
390
- await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
391
- return {
392
- state: { ...state, paa },
393
- output: { paaQuestions: paa.flat?.length ?? 0 }
394
- };
395
- } catch (err) {
396
- const message = `PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`;
397
- return { state: { ...state, warnings: [...state.warnings, message] }, output: { error: message }, warnings: [message] };
398
- }
399
- }
400
- },
401
- {
402
- id: "assemble",
403
- title: "Assemble evidence packet",
404
- async run({ input, state, ctx }) {
405
- const { serp, paa } = state;
406
- const warnings = state.warnings;
407
- if (!serp && !paa && !input.returnPartial) throw new Error("No SEO evidence was collected");
408
- const rows = sourceRows(input, serp, paa);
409
- const target = normalizeDomain(input.domain);
410
- const competitors = competitorRows(rows, target);
411
- const evidence = {
412
- input,
413
- serp: serp ?? { status: "skipped" },
414
- paa: paa ?? { status: "skipped" },
415
- target: {
416
- domain: target,
417
- organicPositions: rows.filter((r) => r.surface === "organic" && r.is_target).map((r) => Number(r.position)),
418
- citedInAiOverview: rows.some((r) => r.surface === "ai_overview" && r.is_target),
419
- citedInPaa: rows.some((r) => r.surface === "paa" && r.is_target)
420
- },
421
- competitors,
422
- warnings
423
- };
424
- await ctx.artifacts.writeJson("Evidence JSON", "evidence.json", evidence);
425
- await ctx.artifacts.writeCsv("Sources CSV", "sources.csv", ["surface", "position", "title", "url", "domain", "question", "answer_excerpt", "is_target", "is_competitor"], rows);
426
- await ctx.artifacts.writeCsv("Competitors CSV", "competitors.csv", ["domain", "organic_best_position", "organic_count", "paa_mentions", "ai_overview_citations", "source_url_count"], competitors);
427
- const brief = [
428
- `# SEO Evidence Brief: ${input.keyword}`,
429
- "",
430
- `Location: ${input.location ?? "not specified"}`,
431
- `Target domain: ${target ?? "not specified"}`,
432
- "",
433
- "## Evidence Summary",
434
- `- Organic rows: ${rows.filter((r) => r.surface === "organic").length}`,
435
- `- PAA rows: ${rows.filter((r) => r.surface === "paa").length}`,
436
- `- AI Overview citations: ${rows.filter((r) => r.surface === "ai_overview").length}`,
437
- `- Competitor domains: ${competitors.length}`,
438
- "",
439
- "## Recommended Use",
440
- "Use the CSV files as source of truth. Tie content recommendations to evidence rows and mark unsupported ideas as assumptions."
441
- ].join("\n");
442
- await ctx.artifacts.writeText("Brief", "brief.md", brief);
443
- await ctx.artifacts.writeText("Agent tasks", "tasks.md", [
444
- "# Agent Tasks",
445
- "",
446
- "- [ ] Read `evidence.json` before writing recommendations.",
447
- "- [ ] Compare the target domain against `competitors.csv`.",
448
- "- [ ] Use `sources.csv` for citations and source-grounded page sections.",
449
- "- [ ] Mark unsupported recommendations as assumptions."
450
- ].join("\n"));
451
- await ctx.artifacts.writeText("Agent instructions", "agent-instructions.md", [
452
- "# Agent Instructions",
453
- "",
454
- "You are working from an MCP Scraper SEO evidence packet. Use `evidence.json` and CSV files as source of truth. Do not invent citations. If a recommendation is not supported by evidence, mark it as an assumption."
455
- ].join("\n"));
456
- const summary = `${rows.length} evidence rows and ${competitors.length} competitor domains collected.`;
457
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
458
- title: "Agent-Ready SEO Packet",
459
- subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
460
- summary,
461
- warnings,
462
- tables: [
463
- { title: "Competitor Domains", columns: ["domain", "organic_best_position", "organic_count", "paa_mentions", "ai_overview_citations", "source_url_count"], rows: competitors.slice(0, 50) },
464
- { title: "Evidence Sources", columns: ["surface", "position", "title", "domain", "question", "is_target"], rows: rows.slice(0, 100) }
465
- ]
466
- }));
467
- const status = warnings.length ? "partial" : "succeeded";
468
- const counts = { sources: rows.length, competitors: competitors.length };
469
- await ctx.artifacts.writeManifest(status, counts, warnings, []);
470
- return {
471
- state,
472
- output: { sources: rows.length, competitors: competitors.length, status },
473
- summary: { title: "Agent-Ready SEO Packet", summary, status, counts, warnings, errors: [], reportPath }
474
- };
475
- }
476
- }
477
- ]
478
- };
479
-
480
- // src/workflows/workflows/directory.ts
481
- import { z as z2 } from "zod";
482
- var DirectoryWorkflowCliInputSchema = z2.object({
483
- query: z2.string().min(1),
484
- state: z2.string().min(2).default("TN"),
485
- minPopulation: z2.number().int().min(0).default(1e5),
486
- maxCities: z2.number().int().min(1).max(100).default(25),
487
- maxResultsPerCity: z2.number().int().min(1).max(50).default(20),
488
- concurrency: z2.number().int().min(1).max(5).default(5),
489
- proxyMode: z2.enum(["configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE),
490
- saveCsv: z2.boolean().default(true)
491
- });
492
- function directoryRows(result) {
493
- const rows = [];
494
- for (const city of result.cities) {
495
- if (!city.results.length) {
496
- rows.push({
497
- source_query: result.query,
498
- source_location: city.location,
499
- city: city.city,
500
- state: city.state,
501
- population: city.population,
502
- result_position: null,
503
- business_name: null,
504
- review_stars: null,
505
- review_count: null,
506
- category: null,
507
- address: null,
508
- phone: null,
509
- website_url: null,
510
- place_url: null,
511
- cid: null,
512
- cid_decimal: null,
513
- result_status: city.status,
514
- error: city.error
515
- });
516
- continue;
517
- }
518
- for (const business of city.results) {
519
- rows.push({
520
- source_query: result.query,
521
- source_location: city.location,
522
- city: city.city,
523
- state: city.state,
524
- population: city.population,
525
- result_position: business.position,
526
- business_name: business.name,
527
- review_stars: business.rating,
528
- review_count: business.reviewCount,
529
- category: business.category,
530
- address: business.address,
531
- phone: business.phone,
532
- website_url: business.websiteUrl,
533
- place_url: business.placeUrl,
534
- cid: business.cid,
535
- cid_decimal: business.cidDecimal,
536
- result_status: city.status,
537
- error: city.error
538
- });
539
- }
540
- }
541
- return rows;
542
- }
543
- var DIRECTORY_CSV_HEADERS = [
544
- "source_query",
545
- "source_location",
546
- "city",
547
- "state",
548
- "population",
549
- "result_position",
550
- "business_name",
551
- "review_stars",
552
- "review_count",
553
- "category",
554
- "address",
555
- "phone",
556
- "website_url",
557
- "place_url",
558
- "cid",
559
- "cid_decimal",
560
- "result_status",
561
- "error"
562
- ];
563
- var directoryWorkflowDefinition = {
564
- id: "directory",
565
- title: "Directory Workflow",
566
- description: "Select city markets and export Google Maps business candidates.",
567
- inputSchema: DirectoryWorkflowCliInputSchema,
568
- async run(input, ctx) {
569
- await ctx.artifacts.writeManifest("running", {}, [], []);
570
- const result = await ctx.client.post("/directory/run", { ...input, background: false }, 9e5);
571
- await ctx.artifacts.writeJson("Directory evidence", "evidence.json", result);
572
- const rows = directoryRows(result);
573
- await ctx.artifacts.writeCsv("Directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, rows);
574
- const cityRows = result.cities.map((city) => ({
575
- city: city.city,
576
- state: city.state,
577
- population: city.population,
578
- status: city.status,
579
- result_count: city.resultCount,
580
- error: city.error ?? ""
581
- }));
582
- const topRows = rows.filter((row) => row.business_name).slice(0, 100).map((row) => ({
583
- city: row.city,
584
- position: row.result_position,
585
- business: row.business_name,
586
- rating: row.review_stars,
587
- reviews: row.review_count,
588
- category: row.category,
589
- website: row.website_url
590
- }));
591
- const summary = `${result.selectedCityCount} cities processed with ${result.totalResultCount} Maps results.`;
592
- await ctx.artifacts.writeText("Summary", "summary.md", `# Directory Workflow
593
-
594
- ${summary}
595
- `);
596
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
597
- title: "Directory Workflow",
598
- subtitle: `${input.query} \xB7 ${input.state}`,
599
- summary,
600
- warnings: result.warnings,
601
- tables: [
602
- { title: "Cities", columns: ["city", "state", "population", "status", "result_count", "error"], rows: cityRows },
603
- { title: "Top Results", columns: ["city", "position", "business", "rating", "reviews", "category", "website"], rows: topRows }
604
- ]
605
- }));
606
- const status = result.cities.some((city) => city.status === "failed") ? "partial" : "succeeded";
607
- const counts = { cities: result.selectedCityCount, results: result.totalResultCount, rows: rows.length };
608
- await ctx.artifacts.writeManifest(status, counts, result.warnings, []);
609
- return { title: "Directory Workflow", summary, status, counts, warnings: result.warnings, errors: [], reportPath };
610
- }
611
- };
612
-
613
- // src/workflows/workflows/get-leads.ts
614
- import { z as z3 } from "zod";
615
- var GetLeadsInputSchema = z3.object({
616
- query: z3.string().min(1).describe('Business category, niche, or keyword to search on Google Maps, e.g. "roofers", "med spas", "dentists". Do not include the city here.'),
617
- location: z3.string().min(1).describe('City / market to search, e.g. "Houston, TX" or "Austin, Texas".'),
618
- maxResults: z3.number().int().min(1).max(50).default(25).describe("How many Maps businesses to collect for the market. Maximum 50."),
619
- enrichWebsites: z3.boolean().default(true).describe("Visit each business website (home + contact pages) to harvest email and social links. Uses the proxy/browser-backed extractor so blocked sites still resolve."),
620
- emailSearchFallback: z3.enum(["off", "serp_snippets"]).default("off").describe("After bounded website extraction finds no email, optionally run one billed SERP lookup and retain exact-domain public email occurrences from result titles/snippets."),
621
- hydrateReviewCounts: z3.boolean().default(true).describe("Deep-dive each profile to confirm the review count and booking URL that the Maps search list omits."),
622
- concurrency: z3.number().int().min(1).max(4).default(3).describe("How many businesses to enrich in parallel. Keep low to respect per-account concurrency limits."),
623
- proxyMode: z3.enum(["configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE).describe("Proxy behavior for the Maps search. Leave unset for clean egress; country/region localization comes from gl/hl plus the city or region in the query.")
624
- });
625
- var LEADS_CSV_HEADERS = [
626
- "position",
627
- "business_name",
628
- "review_stars",
629
- "review_count",
630
- "category",
631
- "address",
632
- "phone",
633
- "website",
634
- "email",
635
- "email_2",
636
- "email_domain_match",
637
- "email_status",
638
- "socials",
639
- "booking_url",
640
- "place_url",
641
- "cid",
642
- "source_location"
643
- ];
644
- async function mapLimit(items, limit, fn) {
645
- const out = new Array(items.length);
646
- let next = 0;
647
- async function worker() {
648
- while (next < items.length) {
649
- const index = next;
650
- next += 1;
651
- out[index] = await fn(items[index], index);
652
- }
653
- }
654
- await Promise.all(Array.from({ length: Math.min(limit, items.length) }, () => worker()));
655
- return out;
656
- }
657
- function normalizedHostname(url) {
658
- try {
659
- return new URL(url).hostname.toLowerCase().replace(/^www\./, "").replace(/\.$/, "");
660
- } catch {
661
- return "";
662
- }
663
- }
664
- function exactEmailDomainMatch(email, websiteUrl) {
665
- const websiteDomain = normalizedHostname(websiteUrl);
666
- const emailDomain = email.slice(email.lastIndexOf("@") + 1).toLowerCase().replace(/^www\./, "").replace(/\.$/, "");
667
- return Boolean(websiteDomain && emailDomain === websiteDomain);
668
- }
669
- function mergeUniqueEmails(target, additions) {
670
- const seen = new Set(target.map((item) => item.value.toLowerCase()));
671
- for (const item of additions) {
672
- if (target.length >= 20) break;
673
- const key = item.value.toLowerCase();
674
- if (seen.has(key)) continue;
675
- seen.add(key);
676
- target.push(item);
677
- }
678
- }
679
- function mergeUniqueUrls(target, additions, max) {
680
- const seen = new Set(target.map((item) => item.url));
681
- for (const item of additions) {
682
- if (target.length >= max) break;
683
- if (seen.has(item.url)) continue;
684
- seen.add(item.url);
685
- target.push(item);
686
- }
687
- }
688
- function orderedEmailEvidence(evidence, businessName) {
689
- const remaining = [...evidence];
690
- const ordered = [];
691
- const entity = { entityType: "business", name: businessName, organization: businessName };
692
- while (remaining.length) {
693
- const next = selectPrimaryEmail(remaining, entity);
694
- if (!next) {
695
- ordered.push(...remaining);
696
- break;
697
- }
698
- ordered.push(next);
699
- const index = remaining.findIndex((item) => item.value.toLowerCase() === next.value.toLowerCase());
700
- remaining.splice(index, 1);
701
- }
702
- return ordered;
703
- }
704
- var getLeadsWorkflowDefinition = {
705
- id: "get-leads",
706
- title: "Get Leads",
707
- description: "Build an outreach-ready local lead list: Google Maps search for a niche in a market, confirm review counts + booking URLs, then visit each business website (home + contact) to harvest email and social links. An opt-in SERP fallback can recover exact-domain public email occurrences after bounded pages are empty. Saves a leads CSV.",
708
- inputSchema: GetLeadsInputSchema,
709
- async run(input, ctx) {
710
- await ctx.artifacts.writeManifest("running", {}, [], []);
711
- const warnings = [];
712
- const errors = [];
713
- const search = await ctx.client.post("/maps/search", {
714
- query: input.query,
715
- location: input.location,
716
- maxResults: input.maxResults,
717
- proxyMode: input.proxyMode
718
- }, 18e4);
719
- await ctx.artifacts.writeJson("Maps search raw", "raw/maps-search.json", search);
720
- const seen = /* @__PURE__ */ new Set();
721
- const deduped = search.results.filter((result) => {
722
- const key = result.cid ?? result.placeUrl ?? result.name.toLowerCase();
723
- if (seen.has(key)) return false;
724
- seen.add(key);
725
- return true;
726
- });
727
- const enriched = await mapLimit(deduped, input.concurrency, async (result) => {
728
- let reviewCount = result.reviewCount ?? "";
729
- let bookingUrl = "";
730
- let website = result.websiteUrl ?? "";
731
- if (input.hydrateReviewCounts) {
732
- try {
733
- const place = await ctx.client.post("/maps/place", {
734
- businessName: result.name,
735
- location: input.location,
736
- includeReviews: false
737
- }, 18e4);
738
- reviewCount = place.reviewCount ?? reviewCount;
739
- bookingUrl = place.bookingUrl ?? "";
740
- website = website || (place.website ?? "");
741
- } catch (err) {
742
- warnings.push(`Review hydration failed for ${result.name}: ${err instanceof Error ? err.message : String(err)}`);
743
- }
744
- }
745
- let email = "";
746
- let email2 = "";
747
- let socials = [];
748
- let emailStatus = website ? "no email found" : "no website";
749
- let domainMatch = "";
750
- const emailEvidence = [];
751
- const socialProfileEvidence = [];
752
- const discoveredPageUrls = [];
753
- const attemptedUrls = [];
754
- if (input.enrichWebsites && website) {
755
- let attempts = contactAttemptUrls(website, discoveredPageUrls, 3);
756
- let reached = false;
757
- const attempted = /* @__PURE__ */ new Set();
758
- while (attempted.size < 3) {
759
- const attemptUrl = attempts.find((url) => !attempted.has(url));
760
- if (!attemptUrl) break;
761
- attempted.add(attemptUrl);
762
- attemptedUrls.push(attemptUrl);
763
- try {
764
- const page = await ctx.client.post("/extract-url", { url: attemptUrl }, 12e4);
765
- reached = true;
766
- const parsed = page.contactEvidence ?? extractContactEvidence(page.bodyHtml ?? page.bodyMarkdown ?? "", attemptUrl);
767
- mergeUniqueEmails(emailEvidence, parsed.emails);
768
- mergeUniqueUrls(socialProfileEvidence, parsed.socialProfiles, 20);
769
- mergeUniqueUrls(discoveredPageUrls, parsed.candidatePageUrls, 20);
770
- attempts = contactAttemptUrls(website, discoveredPageUrls, 3);
771
- if (emailEvidence.length) break;
772
- } catch {
773
- attempts = contactAttemptUrls(website, discoveredPageUrls, 3);
774
- continue;
775
- }
776
- }
777
- if (!emailEvidence.length && input.emailSearchFallback === "serp_snippets") {
778
- const query = buildSerpEmailQuery(website);
779
- if (query) {
780
- try {
781
- const serp = await ctx.client.post("/harvest/sync", {
782
- query,
783
- serpOnly: true,
784
- pages: 1
785
- }, 18e4);
786
- mergeUniqueEmails(emailEvidence, extractSerpEmailEvidence({
787
- websiteUrl: website,
788
- results: serp.result?.organicResults ?? []
789
- }));
790
- } catch (err) {
791
- warnings.push(`Email snippet search failed for ${result.name}: ${err instanceof Error ? err.message : String(err)}`);
792
- }
793
- }
794
- }
795
- socials = socialProfileEvidence.map((item) => item.url).slice(0, 5);
796
- const orderedEmails = orderedEmailEvidence(emailEvidence, result.name);
797
- const domainEmails = orderedEmails.filter((item) => exactEmailDomainMatch(item.value, website));
798
- const selectedEmails = domainEmails.length ? domainEmails.slice(0, 2) : orderedEmails.slice(0, 1);
799
- email = selectedEmails[0]?.value ?? "";
800
- email2 = selectedEmails[1]?.value ?? "";
801
- if (email) {
802
- domainMatch = exactEmailDomainMatch(email, website) ? "yes" : "NO-verify";
803
- emailStatus = selectedEmails.some((item) => item.sourceType === "serp_snippet") ? "serp snippet \u2014 public occurrence, verify" : domainMatch === "yes" ? "ok" : "ok (off-domain \u2014 verify)";
804
- } else if (!reached) {
805
- emailStatus = "site unreachable";
806
- } else if (socials.length) {
807
- emailStatus = "form/social only";
808
- }
809
- }
810
- return {
811
- row: {
812
- position: result.position,
813
- business_name: result.name,
814
- review_stars: result.rating ?? "",
815
- review_count: reviewCount,
816
- category: result.category ?? "",
817
- address: result.address ?? "",
818
- phone: result.phone ?? "",
819
- website,
820
- email,
821
- email_2: email2,
822
- email_domain_match: domainMatch,
823
- email_status: emailStatus,
824
- socials: socials.join(" ; "),
825
- booking_url: bookingUrl,
826
- place_url: result.placeUrl,
827
- cid: result.cid ?? "",
828
- source_location: input.location
829
- },
830
- emailEvidence: orderedEmailEvidence(emailEvidence, result.name),
831
- socialProfileEvidence,
832
- attemptedUrls
833
- };
834
- });
835
- const rows = enriched.map((item) => item.row);
836
- await ctx.artifacts.writeCsv("Leads CSV", "exports/leads.csv", LEADS_CSV_HEADERS, rows);
837
- await ctx.artifacts.writeJson("Leads evidence", "evidence.json", {
838
- input,
839
- search: { query: search.searchQuery, resultCount: search.resultCount },
840
- rows: enriched.map((item) => ({
841
- ...item.row,
842
- emailEvidence: item.emailEvidence,
843
- socialProfileEvidence: item.socialProfileEvidence,
844
- attemptedUrls: item.attemptedUrls
845
- }))
846
- });
847
- const withEmail = rows.filter((r) => r.email).length;
848
- const domainMatched = rows.filter((r) => r.email_domain_match === "yes").length;
849
- const withReviewCount = rows.filter((r) => r.review_count).length;
850
- const withSocials = rows.filter((r) => r.socials).length;
851
- const withWebsite = rows.filter((r) => r.website).length;
852
- const summary = `${rows.length} ${input.query} in ${input.location}: ${withEmail} emails (${domainMatched} domain-matched), ${withReviewCount} review counts, ${withSocials} with socials.`;
853
- await ctx.artifacts.writeText("Summary", "summary.md", `# Get Leads
854
-
855
- ${summary}
856
- `);
857
- await ctx.artifacts.writeText("Agent tasks", "tasks.md", [
858
- "# Get Leads \u2014 next steps",
859
- "",
860
- "- [ ] Verify deliverability of the harvested emails before outreach (MX/SMTP check).",
861
- '- [ ] For "form/social only" rows, use the website contact form or the captured social profile.',
862
- '- [ ] Treat "off-domain \u2014 verify" emails with care; they may belong to a web developer or parent brand.',
863
- "- [ ] Respect CAN-SPAM / GDPR \u2014 scraped contact data is not consent to email."
864
- ].join("\n"));
865
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
866
- title: "Get Leads",
867
- subtitle: `${input.query} \xB7 ${input.location}`,
868
- summary,
869
- warnings,
870
- tables: [
871
- {
872
- title: "Leads",
873
- columns: ["position", "business_name", "review_stars", "review_count", "phone", "website", "email", "email_2", "email_status", "socials", "booking_url"],
874
- rows
875
- }
876
- ]
877
- }));
878
- const status = errors.length ? "failed" : warnings.length ? "partial" : "succeeded";
879
- const counts = {
880
- businesses: rows.length,
881
- withWebsite,
882
- emails: withEmail,
883
- domainMatchedEmails: domainMatched,
884
- reviewCounts: withReviewCount,
885
- withSocials
886
- };
887
- await ctx.artifacts.writeManifest(status, counts, warnings, errors);
888
- return { title: "Get Leads", summary, status, counts, warnings, errors, reportPath };
889
- }
890
- };
891
-
892
- // src/workflows/workflows/local-competitive-audit.ts
893
- import { z as z4 } from "zod";
894
- var LocalCompetitiveAuditInputSchema = z4.object({
895
- query: z4.string().min(1),
896
- state: z4.string().min(2).default("TN"),
897
- minPopulation: z4.number().int().min(0).default(1e5),
898
- maxCities: z4.number().int().min(1).max(100).default(25),
899
- maxResultsPerCity: z4.number().int().min(1).max(50).default(20),
900
- hydrateTop: z4.number().int().min(0).max(10).default(5),
901
- maxReviews: z4.number().int().min(0).max(500).default(50),
902
- concurrency: z4.number().int().min(1).max(5).default(5),
903
- proxyMode: z4.enum(["configured", "none"]).default(DEFAULT_MAPS_PROXY_MODE),
904
- returnPartial: z4.boolean().default(true)
905
- });
906
- async function mapLimit2(items, limit, fn) {
907
- const out = new Array(items.length);
908
- let next = 0;
909
- async function worker() {
910
- while (next < items.length) {
911
- const index = next;
912
- next += 1;
913
- out[index] = await fn(items[index], index);
914
- }
915
- }
916
- await Promise.all(Array.from({ length: Math.min(limit, items.length) }, () => worker()));
917
- return out;
918
- }
919
- function numberFrom(value) {
920
- if (value === null || value === void 0 || value === "") return null;
921
- const parsed = Number(String(value).replace(/[^\d.]/g, ""));
922
- return Number.isFinite(parsed) ? parsed : null;
923
- }
924
- function median(values) {
925
- const nums = values.filter((v) => v !== null).sort((a, b) => a - b);
926
- if (!nums.length) return null;
927
- return nums[Math.floor(nums.length / 2)] ?? null;
928
- }
929
- function termsFrom(texts) {
930
- const stop = /* @__PURE__ */ new Set(["the", "and", "for", "with", "that", "this", "was", "were", "are", "you", "our", "they", "had", "have", "not", "but", "from", "very", "great", "good"]);
931
- const counts = /* @__PURE__ */ new Map();
932
- for (const text of texts) {
933
- for (const token of text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/)) {
934
- if (token.length < 4 || stop.has(token)) continue;
935
- counts.set(token, (counts.get(token) ?? 0) + 1);
936
- }
937
- }
938
- return [...counts.entries()].sort((a, b) => b[1] - a[1]).slice(0, 8).map(([term, count]) => `${term} (${count})`).join("; ");
939
- }
940
- var localCompetitiveAuditWorkflowDefinition = {
941
- id: "local-competitive-audit",
942
- title: "Local Competitive Audit",
943
- description: "Audit local Maps competitors, categories, review counts, and review themes across city markets.",
944
- inputSchema: LocalCompetitiveAuditInputSchema,
945
- async run(input, ctx) {
946
- await ctx.artifacts.writeManifest("running", {}, [], []);
947
- const warnings = [];
948
- const errors = [];
949
- const directory = await ctx.client.post("/directory/run", {
950
- query: input.query,
951
- state: input.state,
952
- minPopulation: input.minPopulation,
953
- maxCities: input.maxCities,
954
- maxResultsPerCity: input.maxResultsPerCity,
955
- concurrency: input.concurrency,
956
- proxyMode: input.proxyMode,
957
- saveCsv: true,
958
- background: false
959
- }, 9e5);
960
- await ctx.artifacts.writeJson("Directory raw JSON", "raw/directory-workflow.json", directory);
961
- const baseRows = directoryRows(directory);
962
- await ctx.artifacts.writeCsv("Base directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, baseRows);
963
- const selected = directory.cities.flatMap(
964
- (city) => city.results.slice(0, input.hydrateTop).map((result) => ({ city, result }))
965
- );
966
- const seen = /* @__PURE__ */ new Set();
967
- const deduped = selected.filter(({ city, result }) => {
968
- const key = result.cid ?? result.placeUrl ?? `${city.location}:${result.name.toLowerCase()}`;
969
- if (seen.has(key)) return false;
970
- seen.add(key);
971
- return true;
972
- });
973
- const hydrated = await mapLimit2(deduped, 3, async ({ city, result }, index) => {
974
- try {
975
- const detail = await ctx.client.post("/maps/place", {
976
- businessName: result.name,
977
- location: city.location,
978
- includeReviews: input.maxReviews > 0,
979
- maxReviews: Math.max(1, input.maxReviews)
980
- }, 18e4);
981
- await ctx.artifacts.writeJson(`${result.name} profile`, `raw/maps-place-intel/${index + 1}-${result.name.toLowerCase().replace(/[^a-z0-9]+/g, "-")}.json`, detail);
982
- return { city, result, detail, error: null };
983
- } catch (err) {
984
- const message = err instanceof Error ? err.message : String(err);
985
- warnings.push(`Profile hydration failed for ${result.name} (${city.location}): ${message}`);
986
- return { city, result, detail: null, error: message };
987
- }
988
- });
989
- const competitorRows2 = hydrated.map(({ city, result, detail, error }) => {
990
- const reviews = detail?.reviews ?? [];
991
- const ownerResponses = reviews.filter((r) => r.ownerResponse).length;
992
- return {
993
- city: city.city,
994
- state: city.state,
995
- source_location: city.location,
996
- result_position: result.position,
997
- business_name: result.name,
998
- category: detail?.category ?? result.category,
999
- review_stars: detail?.rating ?? result.rating,
1000
- review_count: detail?.reviewCount ?? result.reviewCount,
1001
- phone: result.phone,
1002
- website_url: detail?.website ?? result.websiteUrl,
1003
- place_url: result.placeUrl,
1004
- cid: result.cid,
1005
- cid_decimal: result.cidDecimal,
1006
- hydrated: detail ? "true" : "false",
1007
- reviews_status: detail?.reviewsStatus ?? "",
1008
- review_topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1009
- owner_response_rate: reviews.length ? (ownerResponses / reviews.length).toFixed(2) : "",
1010
- error: error ?? ""
1011
- };
1012
- });
1013
- const reviewInsightRows = hydrated.map(({ city, result, detail }) => {
1014
- const reviews = detail?.reviews ?? [];
1015
- const positive = reviews.filter((r) => Number(r.stars) >= 5 && r.text).map((r) => r.text);
1016
- const negative = reviews.filter((r) => Number(r.stars) > 0 && Number(r.stars) <= 3 && r.text).map((r) => r.text);
1017
- return {
1018
- city: city.city,
1019
- business_name: result.name,
1020
- cid: result.cid,
1021
- review_count: detail?.reviewCount ?? result.reviewCount,
1022
- average_rating: detail?.rating ?? result.rating,
1023
- topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1024
- praise_terms: termsFrom(positive),
1025
- complaint_terms: termsFrom(negative),
1026
- positive_sample: positive[0]?.slice(0, 280) ?? "",
1027
- negative_sample: negative[0]?.slice(0, 280) ?? ""
1028
- };
1029
- });
1030
- const citySummaryRows = directory.cities.map((city) => {
1031
- const resultReviewCounts = city.results.map((r) => numberFrom(r.reviewCount));
1032
- const resultRatings = city.results.map((r) => numberFrom(r.rating));
1033
- const topThreeReviews = city.results.slice(0, 3).map((r) => numberFrom(r.reviewCount)).filter((v) => v !== null);
1034
- const categories = /* @__PURE__ */ new Map();
1035
- for (const result of city.results) {
1036
- if (!result.category) continue;
1037
- categories.set(result.category, (categories.get(result.category) ?? 0) + 1);
1038
- }
1039
- const medReviews = median(resultReviewCounts);
1040
- const medRating = median(resultRatings);
1041
- const topThreeAvg = topThreeReviews.length ? Math.round(topThreeReviews.reduce((a, b) => a + b, 0) / topThreeReviews.length) : null;
1042
- const difficulty = Math.min(100, Math.round((topThreeAvg ?? 0) / 20 + (medRating ?? 0) * 10 + city.results.filter((r) => r.websiteUrl).length));
1043
- const opportunity = Math.max(0, 100 - difficulty);
1044
- return {
1045
- city: city.city,
1046
- state: city.state,
1047
- population: city.population,
1048
- result_count: city.resultCount,
1049
- median_review_count: medReviews ?? "",
1050
- median_rating: medRating ?? "",
1051
- top_three_average_review_count: topThreeAvg ?? "",
1052
- top_categories: [...categories.entries()].sort((a, b) => b[1] - a[1]).slice(0, 5).map(([cat, count]) => `${cat} (${count})`).join("; "),
1053
- opportunity_score: opportunity,
1054
- difficulty_score: difficulty
1055
- };
1056
- });
1057
- await ctx.artifacts.writeCsv("Competitors CSV", "competitors.csv", ["city", "state", "source_location", "result_position", "business_name", "category", "review_stars", "review_count", "phone", "website_url", "place_url", "cid", "cid_decimal", "hydrated", "reviews_status", "review_topics", "owner_response_rate", "error"], competitorRows2);
1058
- await ctx.artifacts.writeCsv("Review insights CSV", "review-insights.csv", ["city", "business_name", "cid", "review_count", "average_rating", "topics", "praise_terms", "complaint_terms", "positive_sample", "negative_sample"], reviewInsightRows);
1059
- await ctx.artifacts.writeCsv("City summary CSV", "city-summary.csv", ["city", "state", "population", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "opportunity_score", "difficulty_score"], citySummaryRows);
1060
- await ctx.artifacts.writeJson("Audit evidence", "evidence.json", { input, directory, hydrated, citySummaryRows, competitorRows: competitorRows2, reviewInsightRows });
1061
- await ctx.artifacts.writeText("Agent tasks", "tasks.md", [
1062
- "# Local Competitive Audit Tasks",
1063
- "",
1064
- "- [ ] Review city opportunity and difficulty scores as heuristics, not absolute truth.",
1065
- "- [ ] Use review topics and samples as customer-language evidence.",
1066
- "- [ ] Compare target GBP category, review count, and website quality against top competitors.",
1067
- "- [ ] Identify cities with low review-count leaders and weak website coverage."
1068
- ].join("\n"));
1069
- const summary = `${directory.selectedCityCount} cities, ${directory.totalResultCount} Maps results, ${hydrated.filter((h) => h.detail).length} hydrated profiles.`;
1070
- await ctx.artifacts.writeText("Summary", "summary.md", `# Local Competitive Audit
1071
-
1072
- ${summary}
1073
- `);
1074
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1075
- title: "Local Competitive Audit",
1076
- subtitle: `${input.query} \xB7 ${input.state}`,
1077
- summary,
1078
- warnings,
1079
- tables: [
1080
- { title: "City Opportunity", columns: ["city", "state", "population", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "opportunity_score", "difficulty_score"], rows: citySummaryRows },
1081
- { title: "Hydrated Competitors", columns: ["city", "result_position", "business_name", "category", "review_stars", "review_count", "hydrated", "review_topics"], rows: competitorRows2.slice(0, 100) },
1082
- { title: "Review Insights", columns: ["city", "business_name", "topics", "praise_terms", "complaint_terms"], rows: reviewInsightRows.slice(0, 100) }
1083
- ]
1084
- }));
1085
- const status = warnings.length || directory.cities.some((city) => city.status === "failed") ? "partial" : "succeeded";
1086
- const counts = { cities: directory.selectedCityCount, mapsResults: directory.totalResultCount, hydratedProfiles: hydrated.filter((h) => h.detail).length };
1087
- await ctx.artifacts.writeManifest(status, counts, warnings, errors);
1088
- return { title: "Local Competitive Audit", summary, status, counts, warnings, errors, reportPath };
1089
- }
1090
- };
1091
-
1092
- // src/workflows/workflows/comparison-briefs.ts
1093
- import { z as z5 } from "zod";
1094
-
1095
- // src/workflows/workflows/seo-workflow-utils.ts
1096
- var STOP_WORDS = /* @__PURE__ */ new Set([
1097
- "about",
1098
- "after",
1099
- "also",
1100
- "because",
1101
- "been",
1102
- "best",
1103
- "both",
1104
- "from",
1105
- "have",
1106
- "into",
1107
- "more",
1108
- "most",
1109
- "near",
1110
- "only",
1111
- "over",
1112
- "than",
1113
- "that",
1114
- "their",
1115
- "them",
1116
- "then",
1117
- "there",
1118
- "these",
1119
- "they",
1120
- "this",
1121
- "what",
1122
- "when",
1123
- "where",
1124
- "which",
1125
- "while",
1126
- "with",
1127
- "your",
1128
- "will",
1129
- "would",
1130
- "should",
1131
- "could",
1132
- "does",
1133
- "were",
1134
- "cost",
1135
- "costs"
1136
- ]);
1137
- function normalizeDomain2(value) {
1138
- if (!value) return null;
1139
- try {
1140
- const url = new URL(value.includes("://") ? value : `https://${value}`);
1141
- return url.hostname.replace(/^www\./, "").toLowerCase();
1142
- } catch {
1143
- return value.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0]?.toLowerCase() || null;
1144
- }
1145
- }
1146
- function domainFromUrl2(url) {
1147
- return normalizeDomain2(url) ?? "";
1148
- }
1149
- function numberFrom2(value) {
1150
- if (value === null || value === void 0 || value === "") return null;
1151
- const parsed = Number(String(value).replace(/[^\d.]/g, ""));
1152
- return Number.isFinite(parsed) ? parsed : null;
1153
- }
1154
- function median2(values) {
1155
- const nums = values.filter((v) => v !== null).sort((a, b) => a - b);
1156
- if (!nums.length) return null;
1157
- return nums[Math.floor(nums.length / 2)] ?? null;
1158
- }
1159
- function textTerms(text, limit = 12) {
1160
- const counts = /* @__PURE__ */ new Map();
1161
- for (const token of text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/)) {
1162
- if (token.length < 4 || STOP_WORDS.has(token)) continue;
1163
- counts.set(token, (counts.get(token) ?? 0) + 1);
1164
- }
1165
- return [...counts.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, limit).map(([term, count]) => `${term} (${count})`).join("; ");
1166
- }
1167
- function classifyQuestion(question) {
1168
- const q = question.toLowerCase();
1169
- if (/\b(cost|price|pricing|charge|expensive|cheap)\b/.test(q)) return "cost";
1170
- if (/\b(best|top|recommended|reviews?|compare|versus|vs)\b/.test(q)) return "comparison";
1171
- if (/\b(how|steps?|process|way to)\b/.test(q)) return "process";
1172
- if (/\b(why|worth|important|benefit)\b/.test(q)) return "why";
1173
- if (/\b(near me|city|local|nearby)\b/.test(q)) return "local";
1174
- if (/\b(can|does|do|is|are|should|will)\b/.test(q)) return "decision";
1175
- return "definition";
1176
- }
1177
- function questionRows(paa) {
1178
- return (paa?.flat ?? []).filter((row) => row.question).map((row, index) => {
1179
- const url = row.source_cite ?? "";
1180
- const domain = normalizeDomain2(row.source_site ?? "") ?? domainFromUrl2(url);
1181
- return {
1182
- position: index + 1,
1183
- intent: classifyQuestion(row.question ?? ""),
1184
- question: row.question ?? "",
1185
- answer_excerpt: (row.answer ?? "").slice(0, 320),
1186
- source_title: row.source_title ?? "",
1187
- source_domain: domain,
1188
- source_url: url
1189
- };
1190
- });
1191
- }
1192
- function sourceDomainRows(rows) {
1193
- const byDomain = /* @__PURE__ */ new Map();
1194
- for (const row of rows) {
1195
- const domain = String(row.source_domain ?? row.domain ?? "");
1196
- if (!domain) continue;
1197
- const entry = byDomain.get(domain) ?? { questions: 0, urls: /* @__PURE__ */ new Set(), intents: /* @__PURE__ */ new Map() };
1198
- entry.questions += row.question ? 1 : 0;
1199
- if (row.source_url || row.url) entry.urls.add(String(row.source_url ?? row.url));
1200
- if (row.intent) entry.intents.set(String(row.intent), (entry.intents.get(String(row.intent)) ?? 0) + 1);
1201
- byDomain.set(domain, entry);
1202
- }
1203
- return [...byDomain.entries()].map(([domain, entry]) => ({
1204
- domain,
1205
- question_mentions: entry.questions,
1206
- source_url_count: entry.urls.size,
1207
- top_intents: [...entry.intents.entries()].sort((a, b) => b[1] - a[1]).map(([intent, count]) => `${intent} (${count})`).join("; ")
1208
- })).sort((a, b) => Number(b.question_mentions) - Number(a.question_mentions) || String(a.domain).localeCompare(String(b.domain)));
1209
- }
1210
- function splitSentences(text) {
1211
- return (text ?? "").replace(/\s+/g, " ").split(/(?<=[.!?])\s+/).map((sentence) => sentence.trim()).filter((sentence) => sentence.length > 20).slice(0, 20);
1212
- }
1213
- function classifySentence(sentence) {
1214
- const s = sentence.toLowerCase();
1215
- if (/\bis\b|\bare\b|\bmeans\b|\brefers to\b/.test(s)) return "definition";
1216
- if (/\binclude\b|\bconsider\b|\bfactors?\b|\bcriteria\b/.test(s)) return "criteria";
1217
- if (/\bfirst\b|\bthen\b|\bsteps?\b|\bprocess\b/.test(s)) return "process";
1218
- if (/\bvs\b|\bthan\b|\bcompare\b|\bdifference\b/.test(s)) return "comparison";
1219
- if (/\bcost\b|\bprice\b|\baverage\b|\brange\b/.test(s)) return "cost";
1220
- return "claim";
1221
- }
1222
- function pageSummaryRow(page, source) {
1223
- return {
1224
- position: source.position ?? "",
1225
- domain: source.domain,
1226
- url: source.url,
1227
- serp_title: source.title ?? "",
1228
- page_title: page.title ?? "",
1229
- h1: page.h1 ?? "",
1230
- meta_description: page.metaDescription ?? "",
1231
- word_count: page.wordCount ?? "",
1232
- heading_count: page.headings?.length ?? 0,
1233
- schema_types: (page.schemaTypes ?? []).join("; ")
1234
- };
1235
- }
1236
- async function mapLimit3(items, limit, fn) {
1237
- const out = new Array(items.length);
1238
- let next = 0;
1239
- async function worker() {
1240
- while (next < items.length) {
1241
- const index = next;
1242
- next += 1;
1243
- out[index] = await fn(items[index], index);
1244
- }
1245
- }
1246
- await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, () => worker()));
1247
- return out;
1248
- }
1249
-
1250
- // src/workflows/workflows/comparison-briefs.ts
1251
- var ProxyModeSchema = z5.enum(["configured", "none"]);
1252
- var MapComparisonInputSchema = z5.object({
1253
- query: z5.string().min(1),
1254
- location: z5.string().optional(),
1255
- state: z5.string().optional(),
1256
- minPopulation: z5.number().int().min(0).default(1e5),
1257
- maxCities: z5.number().int().min(1).max(100).default(5),
1258
- maxResultsPerCity: z5.number().int().min(1).max(50).default(20),
1259
- hydrateTop: z5.number().int().min(0).max(10).default(5),
1260
- maxReviews: z5.number().int().min(0).max(500).default(25),
1261
- concurrency: z5.number().int().min(1).max(5).default(5),
1262
- proxyMode: ProxyModeSchema.default(DEFAULT_MAPS_PROXY_MODE),
1263
- returnPartial: z5.boolean().default(true)
1264
- }).refine((input) => input.location || input.state, {
1265
- message: "Either location or state is required for map-comparison"
1266
- });
1267
- var SerpComparisonInputSchema = z5.object({
1268
- keyword: z5.string().min(1),
1269
- domain: z5.string().optional(),
1270
- url: z5.string().url().optional(),
1271
- location: z5.string().optional(),
1272
- maxResults: z5.number().int().min(1).max(20).default(10),
1273
- maxQuestions: z5.number().int().min(1).max(200).default(40),
1274
- extractTop: z5.number().int().min(0).max(10).default(5),
1275
- includePaa: z5.boolean().default(true),
1276
- includeAiOverview: z5.boolean().default(true),
1277
- returnPartial: z5.boolean().default(true)
1278
- });
1279
- var PaaExpansionBriefInputSchema = z5.object({
1280
- keyword: z5.string().min(1),
1281
- location: z5.string().optional(),
1282
- maxQuestions: z5.number().int().min(1).max(300).default(80),
1283
- depth: z5.number().int().min(1).max(6).default(3),
1284
- returnPartial: z5.boolean().default(true)
1285
- });
1286
- var AiOverviewLanguageInputSchema = z5.object({
1287
- keyword: z5.string().min(1),
1288
- domain: z5.string().optional(),
1289
- url: z5.string().url().optional(),
1290
- location: z5.string().optional(),
1291
- maxQuestions: z5.number().int().min(1).max(200).default(40),
1292
- extractTop: z5.number().int().min(0).max(8).default(3),
1293
- returnPartial: z5.boolean().default(true)
1294
- });
1295
- function businessRowsFromMaps(location, query, results) {
1296
- return results.map((result) => ({
1297
- source_query: query,
1298
- source_location: location,
1299
- city: location.split(",")[0]?.trim() ?? location,
1300
- state: location.split(",")[1]?.trim() ?? "",
1301
- population: "",
1302
- result_position: result.position,
1303
- business_name: result.name,
1304
- review_stars: result.rating ?? "",
1305
- review_count: result.reviewCount ?? "",
1306
- category: result.category ?? "",
1307
- address: result.address ?? "",
1308
- phone: result.phone ?? "",
1309
- website_url: result.websiteUrl ?? "",
1310
- place_url: result.placeUrl ?? "",
1311
- cid: result.cid ?? "",
1312
- cid_decimal: result.cidDecimal ?? "",
1313
- result_status: "ok",
1314
- error: ""
1315
- }));
1316
- }
1317
- function marketRows(rows) {
1318
- const byLocation = /* @__PURE__ */ new Map();
1319
- for (const row of rows) {
1320
- const key = String(row.source_location ?? row.location ?? "");
1321
- if (!key) continue;
1322
- const list = byLocation.get(key) ?? [];
1323
- list.push(row);
1324
- byLocation.set(key, list);
1325
- }
1326
- return [...byLocation.entries()].map(([location, list]) => {
1327
- const counts = list.map((row) => numberFrom2(row.review_count));
1328
- const ratings = list.map((row) => numberFrom2(row.review_stars));
1329
- const topThree = list.slice(0, 3).map((row) => numberFrom2(row.review_count)).filter((v) => v !== null);
1330
- const categories = /* @__PURE__ */ new Map();
1331
- for (const row of list) {
1332
- const category = String(row.category ?? "");
1333
- if (category) categories.set(category, (categories.get(category) ?? 0) + 1);
1334
- }
1335
- return {
1336
- source_location: location,
1337
- result_count: list.filter((row) => row.business_name).length,
1338
- median_review_count: median2(counts) ?? "",
1339
- median_rating: median2(ratings) ?? "",
1340
- top_three_average_review_count: topThree.length ? Math.round(topThree.reduce((a, b) => a + b, 0) / topThree.length) : "",
1341
- top_categories: [...categories.entries()].sort((a, b) => b[1] - a[1]).slice(0, 5).map(([cat, count]) => `${cat} (${count})`).join("; "),
1342
- websites_present: list.filter((row) => row.website_url).length
1343
- };
1344
- });
1345
- }
1346
- function comparisonRows(rows) {
1347
- const benchmarkByLocation = /* @__PURE__ */ new Map();
1348
- for (const market of marketRows(rows)) {
1349
- benchmarkByLocation.set(String(market.source_location), numberFrom2(market.top_three_average_review_count) ?? 0);
1350
- }
1351
- return rows.filter((row) => row.business_name).map((row) => {
1352
- const reviews = numberFrom2(row.review_count) ?? 0;
1353
- const benchmark = benchmarkByLocation.get(String(row.source_location)) ?? 0;
1354
- const websiteMissing = !row.website_url;
1355
- const rank = numberFrom2(row.result_position) ?? 999;
1356
- return {
1357
- source_location: row.source_location,
1358
- result_position: row.result_position,
1359
- business_name: row.business_name,
1360
- category: row.category,
1361
- review_stars: row.review_stars,
1362
- review_count: row.review_count,
1363
- review_gap_to_top3_average: benchmark ? Math.max(0, benchmark - reviews) : "",
1364
- website_url: row.website_url,
1365
- place_url: row.place_url,
1366
- comparison_note: rank <= 3 ? "visible leader" : websiteMissing ? "ranking without website" : reviews < benchmark ? "review-light competitor" : "visible competitor"
1367
- };
1368
- });
1369
- }
1370
- function organicRows(serp, targetDomain) {
1371
- return (serp?.organicResults ?? []).map((result) => {
1372
- const url = result.url ?? "";
1373
- const domain = normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(url);
1374
- return {
1375
- position: result.position ?? "",
1376
- title: result.title ?? "",
1377
- url,
1378
- domain,
1379
- snippet: result.snippet ?? "",
1380
- is_target: targetDomain ? domain === targetDomain : false
1381
- };
1382
- });
1383
- }
1384
- function pageGapRows(targetPage, competitorPages) {
1385
- const targetHeadingText = new Set((targetPage?.headings ?? []).map((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim()));
1386
- const targetTerms = new Set((targetPage?.headings ?? []).flatMap((h) => h.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/).filter(Boolean)));
1387
- const rows = [];
1388
- for (const { source, page } of competitorPages) {
1389
- for (const heading of page.headings ?? []) {
1390
- if (heading.level > 3 || !heading.text) continue;
1391
- const normalized = heading.text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").trim();
1392
- const terms = normalized.split(/\s+/).filter((term) => term.length > 3);
1393
- const overlap = terms.filter((term) => targetTerms.has(term)).length;
1394
- const covered = targetHeadingText.has(normalized) || overlap >= Math.max(2, Math.ceil(terms.length / 2));
1395
- if (covered && targetPage) continue;
1396
- rows.push({
1397
- source_position: source.position ?? "",
1398
- source_domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url),
1399
- source_url: source.url ?? "",
1400
- heading_level: heading.level,
1401
- competitor_heading: heading.text,
1402
- target_coverage: targetPage ? "not found in target headings" : "no target page extracted",
1403
- terms: textTerms(heading.text, 6)
1404
- });
1405
- }
1406
- }
1407
- return rows.slice(0, 150);
1408
- }
1409
- async function extractPages(ctx, sources, warnings, labelPrefix) {
1410
- return mapLimit3(sources.filter((source) => source.url), 2, async (source, index) => {
1411
- try {
1412
- const page = await ctx.client.post("/extract-url", { url: source.url }, 18e4);
1413
- await ctx.artifacts.writeJson(`${labelPrefix} ${index + 1}`, `raw/extract-url/${labelPrefix.toLowerCase()}-${index + 1}.json`, page);
1414
- return { source, page };
1415
- } catch (err) {
1416
- warnings.push(`Page extraction failed for ${source.url}: ${err instanceof Error ? err.message : String(err)}`);
1417
- return null;
1418
- }
1419
- }).then((items) => items.filter((item) => item !== null));
1420
- }
1421
- var mapComparisonWorkflowDefinition = {
1422
- id: "map-comparison",
1423
- title: "Maps Comparison",
1424
- description: "Compare Google Maps competitors by rank, reviews, stars, categories, websites, and profile/review signals.",
1425
- inputSchema: MapComparisonInputSchema,
1426
- async run(input, ctx) {
1427
- await ctx.artifacts.writeManifest("running", {}, [], []);
1428
- const warnings = [];
1429
- let rows;
1430
- let directory = null;
1431
- let mapsSearch = null;
1432
- if (input.location) {
1433
- mapsSearch = await ctx.client.post("/maps/search", {
1434
- query: input.query,
1435
- location: input.location,
1436
- maxResults: input.maxResultsPerCity,
1437
- proxyMode: input.proxyMode
1438
- }, 24e4);
1439
- rows = businessRowsFromMaps(input.location, input.query, mapsSearch.results);
1440
- await ctx.artifacts.writeJson("Maps search raw JSON", "raw/maps-search.json", mapsSearch);
1441
- } else {
1442
- directory = await ctx.client.post("/directory/run", {
1443
- query: input.query,
1444
- state: input.state,
1445
- minPopulation: input.minPopulation,
1446
- maxCities: input.maxCities,
1447
- maxResultsPerCity: input.maxResultsPerCity,
1448
- concurrency: input.concurrency,
1449
- proxyMode: input.proxyMode,
1450
- saveCsv: true,
1451
- background: false
1452
- }, 9e5);
1453
- rows = directoryRows(directory);
1454
- await ctx.artifacts.writeJson("Directory raw JSON", "raw/directory-workflow.json", directory);
1455
- await ctx.artifacts.writeCsv("Directory CSV", "exports/directory.csv", DIRECTORY_CSV_HEADERS, rows);
1456
- warnings.push(...directory.warnings);
1457
- }
1458
- const compareRows = comparisonRows(rows);
1459
- const selected = compareRows.slice(0, input.hydrateTop * Math.max(1, input.location ? 1 : input.maxCities));
1460
- const hydrated = await mapLimit3(selected, 3, async (row, index) => {
1461
- try {
1462
- const detail = await ctx.client.post("/maps/place", {
1463
- businessName: row.business_name,
1464
- location: row.source_location,
1465
- includeReviews: input.maxReviews > 0,
1466
- maxReviews: Math.max(1, input.maxReviews)
1467
- }, 18e4);
1468
- await ctx.artifacts.writeJson(`${row.business_name} profile`, `raw/maps-place-intel/${index + 1}-${String(row.business_name).toLowerCase().replace(/[^a-z0-9]+/g, "-")}.json`, detail);
1469
- return { row, detail, error: "" };
1470
- } catch (err) {
1471
- const message = err instanceof Error ? err.message : String(err);
1472
- warnings.push(`Profile hydration failed for ${row.business_name}: ${message}`);
1473
- return { row, detail: null, error: message };
1474
- }
1475
- });
1476
- const profileRows = hydrated.map(({ row, detail, error }) => ({
1477
- source_location: row.source_location,
1478
- result_position: row.result_position,
1479
- business_name: row.business_name,
1480
- category: detail?.category ?? row.category,
1481
- review_stars: detail?.rating ?? row.review_stars,
1482
- review_count: detail?.reviewCount ?? row.review_count,
1483
- website_url: detail?.website ?? row.website_url,
1484
- review_topics: (detail?.reviewTopics ?? []).map((t) => `${t.label} (${t.count})`).join("; "),
1485
- about_attributes: (detail?.aboutAttributes ?? []).map((a) => `${a.section}: ${a.attribute}`).join("; "),
1486
- reviews_status: detail?.reviewsStatus ?? "",
1487
- error
1488
- }));
1489
- const markets = marketRows(rows);
1490
- await ctx.artifacts.writeCsv("Maps results CSV", "maps-results.csv", ["source_query", "source_location", "city", "state", "population", "result_position", "business_name", "review_stars", "review_count", "category", "address", "phone", "website_url", "place_url", "cid", "cid_decimal", "result_status", "error"], rows);
1491
- await ctx.artifacts.writeCsv("Comparison CSV", "map-comparison.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "website_url", "place_url", "comparison_note"], compareRows);
1492
- await ctx.artifacts.writeCsv("Profile insights CSV", "profile-insights.csv", ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "website_url", "review_topics", "about_attributes", "reviews_status", "error"], profileRows);
1493
- await ctx.artifacts.writeJson("Maps comparison evidence", "evidence.json", { input, directory, mapsSearch, rows, compareRows, profileRows, markets, warnings });
1494
- await ctx.artifacts.writeText("Brief", "brief.md", [
1495
- `# Maps Comparison: ${input.query}`,
1496
- "",
1497
- `Markets: ${markets.map((row) => row.source_location).join(", ")}`,
1498
- "",
1499
- "## How to Use",
1500
- "- Compare rank position against review count and category patterns.",
1501
- "- Treat review gaps and missing websites as opportunity signals, not guarantees.",
1502
- "- Use profile topics and attributes as evidence for local content and GBP improvements."
1503
- ].join("\n"));
1504
- const summary = `${compareRows.length} Maps competitors compared across ${markets.length} market(s); ${profileRows.filter((row) => !row.error).length} profiles hydrated.`;
1505
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1506
- title: "Maps Comparison",
1507
- subtitle: `${input.query}${input.location ? ` \xB7 ${input.location}` : input.state ? ` \xB7 ${input.state}` : ""}`,
1508
- summary,
1509
- warnings,
1510
- tables: [
1511
- { title: "Market Benchmarks", columns: ["source_location", "result_count", "median_review_count", "median_rating", "top_three_average_review_count", "top_categories", "websites_present"], rows: markets },
1512
- { title: "Competitor Comparison", columns: ["source_location", "result_position", "business_name", "category", "review_stars", "review_count", "review_gap_to_top3_average", "comparison_note"], rows: compareRows.slice(0, 120) },
1513
- { title: "Profile Insights", columns: ["source_location", "business_name", "review_topics", "about_attributes", "error"], rows: profileRows }
1514
- ]
1515
- }));
1516
- const status = warnings.length ? "partial" : "succeeded";
1517
- const counts = { markets: markets.length, competitors: compareRows.length, hydratedProfiles: profileRows.filter((row) => !row.error).length };
1518
- await ctx.artifacts.writeManifest(status, counts, warnings, []);
1519
- return { title: "Maps Comparison", summary, status, counts, warnings, errors: [], reportPath };
1520
- }
1521
- };
1522
- var serpComparisonWorkflowDefinition = {
1523
- id: "serp-comparison",
1524
- title: "SERP Comparison",
1525
- description: "Compare ranking pages, SERP features, PAA evidence, AI Overview citations, and page-level content gaps.",
1526
- inputSchema: SerpComparisonInputSchema,
1527
- async run(input, ctx) {
1528
- await ctx.artifacts.writeManifest("running", {}, [], []);
1529
- const warnings = [];
1530
- const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
1531
- const serp = await ctx.client.post("/harvest/sync", {
1532
- query: input.keyword,
1533
- location: input.location,
1534
- serpOnly: true,
1535
- includeAllSerpFeatures: true,
1536
- maxQuestions: 1,
1537
- format: "json"
1538
- }, 62e4);
1539
- await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
1540
- let paa = null;
1541
- if (input.includePaa) {
1542
- try {
1543
- paa = await ctx.client.post("/harvest/sync", {
1544
- query: input.keyword,
1545
- location: input.location,
1546
- maxQuestions: input.maxQuestions,
1547
- includeAllSerpFeatures: true,
1548
- format: "json"
1549
- }, 735e3);
1550
- await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1551
- } catch (err) {
1552
- warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
1553
- }
1554
- }
1555
- const organic = organicRows(serp, targetDomain).slice(0, input.maxResults);
1556
- const organicSources = (serp.organicResults ?? []).slice(0, input.maxResults);
1557
- const targetSource = input.url ? { position: 0, title: "Target page", url: input.url, domain: domainFromUrl2(input.url) } : organicSources.find((result) => targetDomain && (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) === targetDomain);
1558
- const competitorSources = organicSources.filter((result) => result.url && result.url !== targetSource?.url).filter((result) => !targetDomain || (normalizeDomain2(result.domain ?? "") ?? domainFromUrl2(result.url)) !== targetDomain).slice(0, input.extractTop);
1559
- const targetPages = targetSource ? await extractPages(ctx, [targetSource], warnings, "Target") : [];
1560
- const competitorPages = input.extractTop > 0 ? await extractPages(ctx, competitorSources, warnings, "Competitor") : [];
1561
- const targetPage = targetPages[0]?.page ?? null;
1562
- const pageRows = [
1563
- ...targetPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title })),
1564
- ...competitorPages.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }))
1565
- ];
1566
- const gaps = pageGapRows(targetPage, competitorPages);
1567
- const questions = questionRows(paa);
1568
- const aiRows = (serp.aiOverview?.citations ?? []).map((citation, index) => ({
1569
- citation_position: index + 1,
1570
- citation_text: citation.text ?? "",
1571
- url: citation.href ?? "",
1572
- domain: domainFromUrl2(citation.href ?? ""),
1573
- is_target: targetDomain ? domainFromUrl2(citation.href ?? "") === targetDomain : false
1574
- }));
1575
- await ctx.artifacts.writeCsv("Organic results CSV", "organic-results.csv", ["position", "title", "url", "domain", "snippet", "is_target"], organic);
1576
- await ctx.artifacts.writeCsv("Page comparison CSV", "page-comparison.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], pageRows);
1577
- await ctx.artifacts.writeCsv("Content gaps CSV", "content-gaps.csv", ["source_position", "source_domain", "source_url", "heading_level", "competitor_heading", "target_coverage", "terms"], gaps);
1578
- await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
1579
- await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], aiRows);
1580
- await ctx.artifacts.writeJson("SERP comparison evidence", "evidence.json", { input, serp, paa, organic, pageRows, gaps, questions, aiRows, warnings });
1581
- await ctx.artifacts.writeText("Writer brief", "brief.md", [
1582
- `# SERP Comparison Brief: ${input.keyword}`,
1583
- "",
1584
- `Target: ${targetDomain ?? input.url ?? "not specified"}`,
1585
- `Location: ${input.location ?? "not specified"}`,
1586
- "",
1587
- "## Recommended Actions",
1588
- "- Use `content-gaps.csv` to decide which missing sections deserve coverage.",
1589
- "- Use `paa-questions.csv` for FAQ and answer-block candidates.",
1590
- "- Use `ai-overview-citations.csv` to see whether the target is cited in AI Overview evidence.",
1591
- "- Treat extracted page headings as evidence, not a complete semantic analysis."
1592
- ].join("\n"));
1593
- const summary = `${organic.length} organic results, ${pageRows.length} extracted pages, ${gaps.length} heading gaps, ${questions.length} PAA questions.`;
1594
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1595
- title: "SERP Comparison",
1596
- subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
1597
- summary,
1598
- warnings,
1599
- tables: [
1600
- { title: "Organic Results", columns: ["position", "title", "domain", "is_target"], rows: organic },
1601
- { title: "Page Comparison", columns: ["position", "domain", "h1", "word_count", "heading_count", "schema_types"], rows: pageRows },
1602
- { title: "Content Gaps", columns: ["source_position", "source_domain", "competitor_heading", "target_coverage", "terms"], rows: gaps.slice(0, 80) },
1603
- { title: "PAA Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 80) }
1604
- ]
1605
- }));
1606
- const status = warnings.length ? "partial" : "succeeded";
1607
- const counts = { organic: organic.length, pages: pageRows.length, gaps: gaps.length, questions: questions.length, aiCitations: aiRows.length };
1608
- await ctx.artifacts.writeManifest(status, counts, warnings, []);
1609
- return { title: "SERP Comparison", summary, status, counts, warnings, errors: [], reportPath };
1610
- }
1611
- };
1612
- var paaExpansionBriefWorkflowDefinition = {
1613
- id: "paa-expansion-brief",
1614
- title: "PAA Expansion Brief",
1615
- description: "Expand People Also Ask questions into an evidence-backed writer brief, section map, and source table.",
1616
- inputSchema: PaaExpansionBriefInputSchema,
1617
- async run(input, ctx) {
1618
- await ctx.artifacts.writeManifest("running", {}, [], []);
1619
- const warnings = [];
1620
- const paa = await ctx.client.post("/harvest/sync", {
1621
- query: input.keyword,
1622
- location: input.location,
1623
- maxQuestions: input.maxQuestions,
1624
- depth: input.depth,
1625
- format: "json"
1626
- }, 735e3);
1627
- await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1628
- const questions = questionRows(paa);
1629
- const sourceRows2 = sourceDomainRows(questions);
1630
- const byIntent = /* @__PURE__ */ new Map();
1631
- for (const row of questions) {
1632
- const list = byIntent.get(String(row.intent)) ?? [];
1633
- list.push(row);
1634
- byIntent.set(String(row.intent), list);
1635
- }
1636
- const sectionRows = [...byIntent.entries()].map(([intent, rows]) => ({
1637
- recommended_section: intent,
1638
- question_count: rows.length,
1639
- sample_questions: rows.slice(0, 5).map((row) => row.question).join(" | "),
1640
- source_domains: [...new Set(rows.map((row) => row.source_domain).filter(Boolean))].slice(0, 5).join("; "),
1641
- terms: textTerms(rows.map((row) => `${row.question} ${row.answer_excerpt}`).join(" "), 10)
1642
- })).sort((a, b) => Number(b.question_count) - Number(a.question_count));
1643
- await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
1644
- await ctx.artifacts.writeCsv("Source domains CSV", "source-domains.csv", ["domain", "question_mentions", "source_url_count", "top_intents"], sourceRows2);
1645
- await ctx.artifacts.writeCsv("Section map CSV", "section-map.csv", ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], sectionRows);
1646
- await ctx.artifacts.writeJson("PAA brief evidence", "evidence.json", { input, paa, questions, sourceRows: sourceRows2, sectionRows });
1647
- await ctx.artifacts.writeText("Writer brief", "writer-brief.md", [
1648
- `# PAA Expansion Brief: ${input.keyword}`,
1649
- "",
1650
- `Location: ${input.location ?? "not specified"}`,
1651
- "",
1652
- "## Suggested Page Structure",
1653
- ...sectionRows.map((row) => `- ${row.recommended_section}: answer ${row.question_count} related question(s). Sample: ${row.sample_questions}`),
1654
- "",
1655
- "## Writing Rules",
1656
- "- Answer the highest-frequency question in the first 60 words of each section.",
1657
- "- Use exact customer question language from `paa-questions.csv` for H2/H3 candidates.",
1658
- "- Use `source-domains.csv` to identify which source types Google is already rewarding.",
1659
- "- Do not invent citations; cite only rows that have a source URL."
1660
- ].join("\n"));
1661
- const summary = `${questions.length} PAA questions grouped into ${sectionRows.length} writing sections.`;
1662
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1663
- title: "PAA Expansion Brief",
1664
- subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
1665
- summary,
1666
- warnings,
1667
- tables: [
1668
- { title: "Section Map", columns: ["recommended_section", "question_count", "sample_questions", "source_domains", "terms"], rows: sectionRows },
1669
- { title: "Questions", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 120) },
1670
- { title: "Source Domains", columns: ["domain", "question_mentions", "source_url_count", "top_intents"], rows: sourceRows2 }
1671
- ]
1672
- }));
1673
- const counts = { questions: questions.length, sections: sectionRows.length, sourceDomains: sourceRows2.length };
1674
- await ctx.artifacts.writeManifest("succeeded", counts, warnings, []);
1675
- return { title: "PAA Expansion Brief", summary, status: "succeeded", counts, warnings, errors: [], reportPath };
1676
- }
1677
- };
1678
- var aiOverviewLanguageWorkflowDefinition = {
1679
- id: "ai-overview-language",
1680
- title: "AI Overview Language Brief",
1681
- description: "Turn AI Overview, citation, PAA, and ranking-page evidence into answer-block and citation-hook guidance.",
1682
- inputSchema: AiOverviewLanguageInputSchema,
1683
- async run(input, ctx) {
1684
- await ctx.artifacts.writeManifest("running", {}, [], []);
1685
- const warnings = [];
1686
- const targetDomain = normalizeDomain2(input.domain ?? input.url ?? null);
1687
- const serp = await ctx.client.post("/harvest/sync", {
1688
- query: input.keyword,
1689
- location: input.location,
1690
- serpOnly: true,
1691
- includeAiOverview: true,
1692
- maxQuestions: 1,
1693
- format: "json"
1694
- }, 62e4);
1695
- await ctx.artifacts.writeJson("SERP raw JSON", "raw/serp.json", serp);
1696
- let paa = null;
1697
- try {
1698
- paa = await ctx.client.post("/harvest/sync", {
1699
- query: input.keyword,
1700
- location: input.location,
1701
- maxQuestions: input.maxQuestions,
1702
- includeAiOverview: true,
1703
- format: "json"
1704
- }, 735e3);
1705
- await ctx.artifacts.writeJson("PAA raw JSON", "raw/paa.json", paa);
1706
- } catch (err) {
1707
- warnings.push(`PAA evidence unavailable: ${err instanceof Error ? err.message : String(err)}`);
1708
- }
1709
- const citations = (serp.aiOverview?.citations ?? []).map((citation, index) => {
1710
- const domain = domainFromUrl2(citation.href ?? "");
1711
- return {
1712
- citation_position: index + 1,
1713
- citation_text: citation.text ?? "",
1714
- url: citation.href ?? "",
1715
- domain,
1716
- is_target: targetDomain ? domain === targetDomain : false
1717
- };
1718
- });
1719
- const aioSentences = splitSentences(serp.aiOverview?.text);
1720
- const claimRows = aioSentences.map((sentence, index) => ({
1721
- position: index + 1,
1722
- claim_type: classifySentence(sentence),
1723
- sentence,
1724
- reusable_pattern: sentence.length > 140 ? `${sentence.slice(0, 140)}...` : sentence
1725
- }));
1726
- const questions = questionRows(paa);
1727
- const citationSources = (serp.aiOverview?.citations ?? []).filter((citation) => citation.href).map((citation, index) => ({ position: index + 1, title: citation.text, url: citation.href, domain: domainFromUrl2(citation.href) })).slice(0, input.extractTop);
1728
- const extractedCitations = input.extractTop > 0 ? await extractPages(ctx, citationSources, warnings, "Citation") : [];
1729
- const extractedRows = extractedCitations.map(({ source, page }) => pageSummaryRow(page, { url: source.url ?? "", domain: normalizeDomain2(source.domain ?? "") ?? domainFromUrl2(source.url), position: source.position, title: source.title }));
1730
- const languageRows = [
1731
- {
1732
- block: "direct_answer",
1733
- guidance: "Open with a 40-70 word answer that directly resolves the query before adding context.",
1734
- evidence_basis: questions[0]?.question ?? input.keyword
1735
- },
1736
- {
1737
- block: "criteria_or_steps",
1738
- guidance: "List the criteria, steps, or decision factors Google is already compressing into AI Overview language.",
1739
- evidence_basis: claimRows.filter((row) => ["criteria", "process"].includes(String(row.claim_type))).map((row) => row.sentence).slice(0, 3).join(" | ")
1740
- },
1741
- {
1742
- block: "citation_hook",
1743
- guidance: "Add source-worthy details competitors can cite: definitions, numbers, examples, process details, and named entity relationships.",
1744
- evidence_basis: citations.map((row) => `${row.domain}: ${row.citation_text}`).slice(0, 5).join(" | ")
1745
- },
1746
- {
1747
- block: "faq_followups",
1748
- guidance: "Use PAA phrasing for follow-up sections so the page answers adjacent questions in Google language.",
1749
- evidence_basis: questions.slice(0, 5).map((row) => row.question).join(" | ")
1750
- }
1751
- ];
1752
- await ctx.artifacts.writeCsv("AI Overview citations CSV", "ai-overview-citations.csv", ["citation_position", "citation_text", "url", "domain", "is_target"], citations);
1753
- await ctx.artifacts.writeCsv("AI Overview claim patterns CSV", "claim-patterns.csv", ["position", "claim_type", "sentence", "reusable_pattern"], claimRows);
1754
- await ctx.artifacts.writeCsv("Language guidance CSV", "language-guidance.csv", ["block", "guidance", "evidence_basis"], languageRows);
1755
- await ctx.artifacts.writeCsv("PAA questions CSV", "paa-questions.csv", ["position", "intent", "question", "answer_excerpt", "source_title", "source_domain", "source_url"], questions);
1756
- await ctx.artifacts.writeCsv("Extracted citation pages CSV", "citation-pages.csv", ["position", "domain", "url", "serp_title", "page_title", "h1", "meta_description", "word_count", "heading_count", "schema_types"], extractedRows);
1757
- await ctx.artifacts.writeJson("AI Overview language evidence", "evidence.json", { input, serp, paa, citations, claimRows, languageRows, extractedRows, warnings });
1758
- await ctx.artifacts.writeText("Answer block template", "answer-block-template.md", [
1759
- `# AI Overview Language Brief: ${input.keyword}`,
1760
- "",
1761
- `AI Overview detected: ${serp.aiOverview?.detected ? "yes" : "no"}`,
1762
- `Target cited: ${targetDomain ? citations.some((row) => row.is_target) ? "yes" : "no" : "target not specified"}`,
1763
- "",
1764
- "## Direct Answer Block",
1765
- "Write one compact answer block that starts with the answer, not background. Keep it clear enough that Google could lift it as a standalone summary.",
1766
- "",
1767
- "## Suggested Follow-Up Blocks",
1768
- ...languageRows.map((row) => `- ${row.block}: ${row.guidance}`),
1769
- "",
1770
- "## Evidence to Mirror",
1771
- ...claimRows.slice(0, 8).map((row) => `- ${row.claim_type}: ${row.sentence}`),
1772
- "",
1773
- "## Citation Hooks",
1774
- ...citations.slice(0, 8).map((row) => `- ${row.domain}: ${row.citation_text}`)
1775
- ].join("\n"));
1776
- const summary = `${citations.length} AI Overview citations, ${claimRows.length} claim patterns, ${questions.length} PAA questions, ${extractedRows.length} citation pages extracted.`;
1777
- const reportPath = await ctx.artifacts.writeHtml("HTML report", "report.html", renderWorkflowReport({
1778
- title: "AI Overview Language Brief",
1779
- subtitle: `${input.keyword}${input.location ? ` \xB7 ${input.location}` : ""}`,
1780
- summary,
1781
- warnings,
1782
- tables: [
1783
- { title: "Language Guidance", columns: ["block", "guidance", "evidence_basis"], rows: languageRows },
1784
- { title: "AI Overview Citations", columns: ["citation_position", "citation_text", "domain", "is_target"], rows: citations },
1785
- { title: "Claim Patterns", columns: ["position", "claim_type", "sentence"], rows: claimRows },
1786
- { title: "PAA Follow-Ups", columns: ["position", "intent", "question", "source_domain"], rows: questions.slice(0, 60) }
1787
- ]
1788
- }));
1789
- const status = warnings.length || !serp.aiOverview?.detected ? "partial" : "succeeded";
1790
- const counts = { citations: citations.length, claimPatterns: claimRows.length, questions: questions.length, extractedCitationPages: extractedRows.length };
1791
- await ctx.artifacts.writeManifest(status, counts, warnings, []);
1792
- return { title: "AI Overview Language Brief", summary, status, counts, warnings, errors: [], reportPath };
1793
- }
1794
- };
1795
-
1796
- // src/workflows/registry.ts
1797
- var DEFINITIONS = [
1798
- directoryWorkflowDefinition,
1799
- getLeadsWorkflowDefinition,
1800
- agentPacketWorkflowDefinition,
1801
- localCompetitiveAuditWorkflowDefinition,
1802
- mapComparisonWorkflowDefinition,
1803
- serpComparisonWorkflowDefinition,
1804
- paaExpansionBriefWorkflowDefinition,
1805
- aiOverviewLanguageWorkflowDefinition
1806
- ];
1807
- function listWorkflowDefinitions() {
1808
- return DEFINITIONS.map(({ id, title, description }) => ({ id, title, description }));
1809
- }
1810
- function workflowDefinition(id) {
1811
- const definition = DEFINITIONS.find((def) => def.id === id);
1812
- if (!definition) throw new Error(`Unknown workflow "${id}". Available: ${DEFINITIONS.map((def) => def.id).join(", ")}`);
1813
- return definition;
1814
- }
1815
- function workflowStepCount(id) {
1816
- const definition = workflowDefinition(id);
1817
- return definition.steps?.length ?? 1;
1818
- }
1819
- function workflowSupportsSteps(id) {
1820
- const definition = workflowDefinition(id);
1821
- return Boolean(definition.steps && definition.steps.length > 0);
1822
- }
1823
- function resolveApiKey(options) {
1824
- const apiKey = options.apiKey?.trim() || process.env.MCP_SCRAPER_API_KEY?.trim();
1825
- if (!apiKey) throw new Error("MCP_SCRAPER_API_KEY is required for workflow runs. Pass --api-key or set the environment variable.");
1826
- return apiKey;
1827
- }
1828
- function resolveApiUrl(options) {
1829
- return options.apiUrl?.trim() || process.env.MCP_SCRAPER_API_URL?.trim() || "https://mcpscraper.dev";
1830
- }
1831
- async function runWorkflow(id, rawInput, options = {}) {
1832
- const definition = workflowDefinition(id);
1833
- const input = definition.inputSchema.parse(rawInput);
1834
- const apiKey = resolveApiKey(options);
1835
- const apiUrl = resolveApiUrl(options);
1836
- const artifacts = await ArtifactWriter.create(definition.id, definition.title, input, options.outputDir, options.runId);
1837
- const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl, options.headers);
1838
- const ctx = { runId: artifacts.runId, startedAt: artifacts.startedAt, client, artifacts, signal: options.signal };
1839
- try {
1840
- if (definition.steps && definition.steps.length > 0) {
1841
- let state = definition.createState ? definition.createState(input) : {};
1842
- let summary = null;
1843
- for (const step of definition.steps) {
1844
- const outcome = await step.run({ input, state, ctx });
1845
- state = outcome.state;
1846
- if (outcome.summary) summary = outcome.summary;
1847
- }
1848
- if (!summary) throw new Error(`Workflow "${id}" produced no terminal step summary`);
1849
- return summary;
1850
- }
1851
- if (definition.run) return await definition.run(input, ctx);
1852
- throw new Error(`Workflow "${id}" has neither steps nor a run() implementation`);
1853
- } catch (err) {
1854
- const message = err instanceof Error ? err.message : String(err);
1855
- await artifacts.writeText("Failure", "summary.md", `# ${definition.title}
1856
-
1857
- ${message}
1858
- `);
1859
- await artifacts.writeManifest("failed", {}, [], [message]);
1860
- throw err;
1861
- }
1862
- }
1863
- async function runWorkflowStep(id, rawInput, opts) {
1864
- const definition = workflowDefinition(id);
1865
- const steps = definition.steps;
1866
- if (!steps || steps.length === 0) throw new Error(`Workflow "${id}" does not support stepwise execution`);
1867
- if (opts.stepIndex < 0 || opts.stepIndex >= steps.length) {
1868
- throw new Error(`Step index ${opts.stepIndex} is out of range for "${id}" (${steps.length} steps)`);
1869
- }
1870
- const options = opts.options ?? {};
1871
- const input = definition.inputSchema.parse(rawInput);
1872
- const apiKey = resolveApiKey(options);
1873
- const apiUrl = resolveApiUrl(options);
1874
- const artifacts = await ArtifactWriter.create(definition.id, definition.title, input, options.outputDir, options.runId);
1875
- const client = new WorkflowHttpClient(apiUrl, apiKey, options.fetchImpl, options.headers);
1876
- const ctx = { runId: artifacts.runId, startedAt: artifacts.startedAt, client, artifacts, signal: options.signal };
1877
- const state = opts.state ?? (definition.createState ? definition.createState(input) : {});
1878
- const step = steps[opts.stepIndex];
1879
- const outcome = await step.run({ input, state, ctx });
1880
- const writtenArtifacts = await Promise.all(
1881
- artifacts.artifacts.map(async (artifact) => ({
1882
- ...artifact,
1883
- content: await readFile2(artifact.path, "utf8").catch(() => "")
1884
- }))
1885
- );
1886
- return {
1887
- runId: artifacts.runId,
1888
- workflowId: definition.id,
1889
- title: definition.title,
1890
- stepIndex: opts.stepIndex,
1891
- stepId: step.id,
1892
- stepTitle: step.title,
1893
- totalSteps: steps.length,
1894
- isLast: opts.stepIndex === steps.length - 1,
1895
- output: outcome.output,
1896
- state: outcome.state,
1897
- warnings: outcome.warnings ?? [],
1898
- summary: outcome.summary ?? null,
1899
- artifacts: writtenArtifacts
1900
- };
1901
- }
1902
-
1903
- export {
1904
- slugify,
1905
- workflowOutputBaseDir,
1906
- listWorkflowReports,
1907
- findWorkflowReport,
1908
- openWorkflowReport,
1909
- listWorkflowDefinitions,
1910
- workflowDefinition,
1911
- workflowStepCount,
1912
- workflowSupportsSteps,
1913
- runWorkflow,
1914
- runWorkflowStep
1915
- };