@houtini/seo-audit-console 0.2.3 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/README.md +11 -4
  2. package/dist/audit/keywordList.d.ts +40 -0
  3. package/dist/audit/keywordList.d.ts.map +1 -0
  4. package/dist/audit/keywordList.js +95 -0
  5. package/dist/audit/keywordList.js.map +1 -0
  6. package/dist/audit/recon.d.ts +56 -0
  7. package/dist/audit/recon.d.ts.map +1 -0
  8. package/dist/audit/recon.js +134 -0
  9. package/dist/audit/recon.js.map +1 -0
  10. package/dist/core/AuditDatabase.d.ts.map +1 -1
  11. package/dist/core/AuditDatabase.js +39 -0
  12. package/dist/core/AuditDatabase.js.map +1 -1
  13. package/dist/core/FirecrawlClient.d.ts +35 -0
  14. package/dist/core/FirecrawlClient.d.ts.map +1 -0
  15. package/dist/core/FirecrawlClient.js +69 -0
  16. package/dist/core/FirecrawlClient.js.map +1 -0
  17. package/dist/core/SupadataClient.d.ts +26 -0
  18. package/dist/core/SupadataClient.d.ts.map +1 -0
  19. package/dist/core/SupadataClient.js +57 -0
  20. package/dist/core/SupadataClient.js.map +1 -0
  21. package/dist/core/dashboardData.d.ts +13 -0
  22. package/dist/core/dashboardData.d.ts.map +1 -1
  23. package/dist/core/dashboardData.js +46 -3
  24. package/dist/core/dashboardData.js.map +1 -1
  25. package/dist/core/marketSizing.d.ts +45 -0
  26. package/dist/core/marketSizing.d.ts.map +1 -0
  27. package/dist/core/marketSizing.js +104 -0
  28. package/dist/core/marketSizing.js.map +1 -0
  29. package/dist/core/reconFetch.d.ts +29 -0
  30. package/dist/core/reconFetch.d.ts.map +1 -0
  31. package/dist/core/reconFetch.js +49 -0
  32. package/dist/core/reconFetch.js.map +1 -0
  33. package/dist/core/reconResearch.d.ts +42 -0
  34. package/dist/core/reconResearch.d.ts.map +1 -0
  35. package/dist/core/reconResearch.js +101 -0
  36. package/dist/core/reconResearch.js.map +1 -0
  37. package/dist/core/serpFootprint.d.ts +42 -0
  38. package/dist/core/serpFootprint.d.ts.map +1 -0
  39. package/dist/core/serpFootprint.js +99 -0
  40. package/dist/core/serpFootprint.js.map +1 -0
  41. package/dist/core/serpRecon.d.ts +46 -0
  42. package/dist/core/serpRecon.d.ts.map +1 -0
  43. package/dist/core/serpRecon.js +71 -0
  44. package/dist/core/serpRecon.js.map +1 -0
  45. package/dist/core/webServer.d.ts +27 -0
  46. package/dist/core/webServer.d.ts.map +1 -0
  47. package/dist/core/webServer.js +156 -0
  48. package/dist/core/webServer.js.map +1 -0
  49. package/dist/server.d.ts.map +1 -1
  50. package/dist/server.js +519 -15
  51. package/dist/server.js.map +1 -1
  52. package/dist/src/ui/dashboard.html +269 -251
  53. package/dist/src/ui/sync-progress.html +2 -2
  54. package/package.json +1 -1
  55. package/server.json +18 -4
package/dist/server.js CHANGED
@@ -8,6 +8,23 @@ import path from 'node:path';
8
8
  import { fileURLToPath } from 'node:url';
9
9
  import { z } from 'zod';
10
10
  import { getDashboardData } from './core/dashboardData.js';
11
+ import { startDashboardServer, stopDashboardServer, dashboardServerUrl, listLocalProperties } from './core/webServer.js';
12
+ import { computeSerpFootprint, persistSerpFootprint } from './core/serpFootprint.js';
13
+ import { computeMarketSizing, persistMarketSizing } from './core/marketSizing.js';
14
+ import { FirecrawlClient } from './core/FirecrawlClient.js';
15
+ import { SupadataClient } from './core/SupadataClient.js';
16
+ import { fetchCompetitorContent } from './core/reconResearch.js';
17
+ import { fetchOwnPage } from './core/reconFetch.js';
18
+ import { parseSerpForRecon, reconVerdict } from './core/serpRecon.js';
19
+ import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos } from './audit/recon.js';
20
+ /** A clickable browser-dashboard link appended to tool outputs — the user should always
21
+ * know the full interactive report is one click away (or one serve_dashboard call away). */
22
+ function browserLink(siteUrl) {
23
+ const base = dashboardServerUrl();
24
+ if (base)
25
+ return `\n\nBrowser dashboard: ${base}/dashboard${siteUrl ? `?siteUrl=${encodeURIComponent(siteUrl)}` : ''}`;
26
+ return `\n\nTip: run serve_dashboard to open the full interactive dashboard in your browser.`;
27
+ }
11
28
  import { runAudit, runSingleCheck, listChecks } from './audit/engine.js';
12
29
  import { buildAuditMarkdown } from './audit/report.js';
13
30
  import { diffLatest, buildDriftMarkdown } from './audit/drift.js';
@@ -34,6 +51,8 @@ import { Backlinks } from './core/Backlinks.js';
34
51
  import { WikidataClient } from './core/WikidataClient.js';
35
52
  import { Entities } from './core/Entities.js';
36
53
  import { JobManager } from './core/JobManager.js';
54
+ import { gscFreshness } from './core/gscFreshness.js';
55
+ import { clusterKeywordList } from './audit/keywordList.js';
37
56
  const SERVER_NAME = 'seo-audit-console';
38
57
  const SERVER_VERSION = JSON.parse(readFileSync(new URL('../package.json', import.meta.url), 'utf8')).version;
39
58
  // Where per-property crawl/audit DBs live. Resolution order (computed once per run):
@@ -91,6 +110,18 @@ ARCHETYPE CHAINS (compose along these lines):
91
110
  2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
92
111
  3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
93
112
 
113
+ COST DISCIPLINE (behave like a strategist who knows the margins):
114
+ - Free and instant, use liberally: everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
115
+ - Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
116
+ - Paid per-keyword (SERP): related_terms and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
117
+ - Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
118
+
119
+ AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
120
+ 1. Baseline: refresh_property → run_audit → serve_dashboard (share the URL).
121
+ 2. Market: serp_features (feature exposure) + domain_visibility + competitors_domain → topic_gaps vs the named rivals.
122
+ 3. Content plan: suggest_pages (demand you already have) + topic_gaps (demand rivals own) → draft_content briefs.
123
+ 4. Fix cycle: fix_finding per top finding → re-crawl → detect_changes to prove the fix landed.
124
+
94
125
  Before planning ANY complex multi-source question, call composition_cookbook — it returns the full data-surface map and worked recipes using these exact tool and table names.`;
95
126
  const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
96
127
 
@@ -134,6 +165,15 @@ Raw access: query_audit runs any single check with full evidence; every table ab
134
165
  - **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
135
166
  - **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
136
167
 
168
+ ## Agency engagement recipes (the deliverable arc)
169
+
170
+ - **Week-one baseline:** refresh_property → run_audit → serve_dashboard. Share the dashboard URL; the ranked findings ARE the technical workstream.
171
+ - **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
172
+ - **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
173
+ - **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
174
+ - **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
175
+ - **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
176
+
137
177
  Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
138
178
  // The check catalogue rendered as markdown — single source of truth is listChecks();
139
179
  // shared by the seo-audit://checks-reference resource (and buildable for any category subset).
@@ -213,6 +253,12 @@ export function createServer() {
213
253
  : null;
214
254
  const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
215
255
  const backlinks = dfs ? new Backlinks(dfs, dataDir()) : null;
256
+ // Firecrawl (competitor-page scraping for content recon) — optional; degrades gracefully.
257
+ const firecrawlKey = process.env.FIRECRAWL_API_KEY;
258
+ const firecrawl = firecrawlKey ? new FirecrawlClient(firecrawlKey, path.join(dataDir(), 'firecrawl-cache.db')) : null;
259
+ // Supadata (transcribes the ranking videos for content recon) — optional; degrades gracefully.
260
+ const supadataKey = process.env.SUPADATA_API_KEY;
261
+ const supadata = supadataKey ? new SupadataClient(supadataKey, path.join(dataDir(), 'supadata-cache.db')) : null;
216
262
  const entities = new Entities(new WikidataClient(path.join(dataDir(), 'wikidata-cache.db')), dataDir());
217
263
  const refresh = new Refresh(sync, crawler, inspector, rankTracker);
218
264
  const requireGsc = (v) => {
@@ -336,7 +382,7 @@ export function createServer() {
336
382
  }, async ({ siteUrl, scope, categories, includeJudgement }) => {
337
383
  const result = runAudit(dataDir(), siteUrl, { scope, categories, includeJudgement });
338
384
  return {
339
- content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) }],
385
+ content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) + browserLink(siteUrl) }],
340
386
  structuredContent: result,
341
387
  };
342
388
  });
@@ -740,7 +786,7 @@ export function createServer() {
740
786
  // ── DataForSEO (cached 20 days, single-worker) ──────────────────────────
741
787
  server.registerTool('keyword_volume', {
742
788
  title: 'Keyword search volume (DataForSEO)',
743
- description: 'True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
789
+ description: '[Paid: Keywords API, cheap, cached 20d | Use for: demand sizing] True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
744
790
  inputSchema: {
745
791
  keywords: z.array(z.string()).min(1).max(700),
746
792
  location: z.union([z.string(), z.number()]).optional(),
@@ -759,7 +805,7 @@ export function createServer() {
759
805
  });
760
806
  server.registerTool('related_terms', {
761
807
  title: 'Related terms (People Also Ask + related searches)',
762
- description: 'People Also Ask questions and related searches for a keyword (DataForSEO SERP). Powers click-through "related terms" on the keyword charts. SERP call — cached 20 days.',
808
+ description: '[Paid: SERP call PER KEYWORD - only for clicked/explicit keywords, never lists | Use for: the SERP context of a single keyword] People Also Ask questions and related searches for a keyword (DataForSEO SERP). Powers click-through "related terms" on the keyword charts. SERP call — cached 20 days.',
763
809
  inputSchema: { keyword: z.string(), location: z.union([z.string(), z.number()]).optional(), languageCode: z.string().optional() },
764
810
  }, async ({ keyword, location, languageCode }) => {
765
811
  const client = requireDfs(dfs);
@@ -788,9 +834,312 @@ export function createServer() {
788
834
  db.close();
789
835
  }
790
836
  });
837
+ // content_opportunities — the content marketer's report: everything the stored data
838
+ // says about what to WRITE, REFRESH and REWRITE, in one free composition.
839
+ server.registerTool('content_opportunities', {
840
+ title: 'Content opportunity report (write / refresh / rewrite)',
841
+ description: 'The content marketer\'s report, composed entirely from stored data - NO paid calls. Four sections: WRITE NEXT (new pages proposed from queries you already earn impressions for but have no winning page - suggest_pages), REFRESH NOW (pages that lost 20%+ of their clicks vs the prior period - content decay), REWRITE SNIPPETS (page-1 rankings earning far below expected CTR - title/meta rewrites, the fastest wins), and STRENGTHEN (keyword clusters where you rank 4-20 - one push from the money positions). Every line traces to real Search Console data. Chain into draft_content for a brief, or keyword_volume to size a cluster against the market.',
842
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(3).max(30).optional().describe('Rows per section (default 10)') },
843
+ }, async ({ siteUrl, limit }) => {
844
+ const n = limit ?? 10;
845
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
846
+ let proposals;
847
+ let weak;
848
+ try {
849
+ proposals = suggestPages(db.db, { maxProposals: n }).proposals;
850
+ const fresh = gscFreshness(db.db);
851
+ weak = clusterKeywordList(db.db, undefined, { maxDate: fresh.effectiveMax }).clusters
852
+ .filter(c => c.verdict === 'weak' && c.bestPosition != null && c.bestPosition >= 4).slice(0, n);
853
+ }
854
+ finally {
855
+ db.close();
856
+ }
857
+ const dash = getDashboardData(dataDir(), siteUrl);
858
+ const decay = (dash.contentDecay ?? []).slice(0, n);
859
+ const snippets = (dash.quickWins ?? []).filter(q => q.type === 'snippet').sort((a, b) => b.potential - a.potential).slice(0, n);
860
+ const sect = [];
861
+ sect.push(`## 1. Write next - demand you already have, no winning page\n` + (proposals.length
862
+ ? `| Proposed page (head term) | Impressions/mo | Best pos today | Queries |\n|---|---|---|---|\n` +
863
+ proposals.map(p => `| ${p.headTerm.replace(/\|/g, '\\|')} | ${fmtNum(p.totalImpressions)} | ${p.bestPosition} | ${p.queries.length} |`).join('\n')
864
+ : '_No unserved-demand gaps found._'));
865
+ sect.push(`## 2. Refresh now - pages losing clicks\n` + (decay.length
866
+ ? `| Page | Clicks were | Now | Drop | Clicks lost |\n|---|---|---|---|---|\n` +
867
+ decay.map(d => `| ${d.urlKey.replace(/\|/g, '\\|')} | ${fmtNum(d.prevClicks)} | ${fmtNum(d.clicks)} | ${d.dropPct}% | ${fmtNum(d.lost)} |`).join('\n')
868
+ : '_No significant decay - nothing lost 20%+ of its clicks._'));
869
+ sect.push(`## 3. Rewrite snippets - ranking well, under-clicked\n` + (snippets.length
870
+ ? `| Query | Position | CTR | Expected | Clicks recoverable/mo |\n|---|---|---|---|---|\n` +
871
+ snippets.map(q => `| ${q.query.replace(/\|/g, '\\|')} | ${q.position} | ${(q.ctr * 100).toFixed(1)}% | ${(q.expectedCtr * 100).toFixed(1)}% | ${fmtNum(Math.round(q.potential))} |`).join('\n')
872
+ : '_No page-1 CTR gaps found._'));
873
+ sect.push(`## 4. Strengthen - clusters one push from the money positions\n` + (weak.length
874
+ ? `| Cluster | Best pos | Impressions 90d | Keywords | Ranking URL |\n|---|---|---|---|---|\n` +
875
+ weak.map(c => `| ${c.head.replace(/\|/g, '\\|')} | ${c.bestPosition} | ${fmtNum(c.impressions)} | ${c.keywords.length} | ${c.url ?? '-'} |`).join('\n')
876
+ : '_No weak clusters - your rankings are polarised own/absent._'));
877
+ const md = `# Content opportunity report - ${siteUrl}\n\nEverything below is from your own Search Console + crawl data (no paid calls, no invented keywords).\n\n` +
878
+ sect.join('\n\n') +
879
+ `\n\nNext moves: draft_content for a brief on any row · keyword_volume to size a cluster against the market · topic_gaps to add what competitors own.` +
880
+ browserLink(siteUrl);
881
+ return {
882
+ content: [{ type: 'text', text: md }],
883
+ structuredContent: {
884
+ siteUrl,
885
+ writeNext: proposals, refreshNow: decay, rewriteSnippets: snippets,
886
+ strengthen: weak.map(c => ({ head: c.head, bestPosition: c.bestPosition, impressions: c.impressions, url: c.url, keywords: c.keywords.length })),
887
+ },
888
+ };
889
+ });
890
+ // ── Content recon (recon_targets) — the data-intensive "why are we losing, what to do" mission ──
891
+ server.registerTool('recon_targets', {
892
+ title: 'Content recon: why a page is losing, and what to do about it',
893
+ description: '[Paid: DataForSEO SERP per page (~$0.004 each), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, or competitive-gap. It writes a per-page classification plus deterministic to-dos (schema gaps, freshness, cannibalisation, video format) into a trackable ledger, and returns the competitor set (organic-above + AI-Overview references + ranking videos) for the research session to diff. Set scrapeCompetitors:true to also pull the top competitors as content, routed by source: YouTube/video → Supadata transcript (SUPADATA_API_KEY), Reddit → its .json, other pages → Firecrawl (FIRECRAWL_API_KEY) with a free HTTP fallback. Cloudflare-challenge sites (e.g. PCMag) still can\'t be fetched from a server and come back as a per-URL error — the SERP still tells you they rank; if one matters, ask the user to paste its copy or supply a text file and diff that in. Transcribe the videos (usually what wins these SERPs) and use the reachable pages. Then write findings back with save_recon_todo; track with recon_todos.',
894
+ inputSchema: {
895
+ siteUrl: z.string(),
896
+ limit: z.number().int().min(1).max(15).optional().describe('Pages per batch (default 5)'),
897
+ minImpressions: z.number().int().min(1).optional(),
898
+ location: z.union([z.string(), z.number()]).optional(),
899
+ urls: z.array(z.string()).optional().describe('Analyse these exact pages instead of auto-selecting'),
900
+ scrapeCompetitors: z.boolean().optional().describe('Also fetch the top competitors (video→transcript, pages→markdown/HTML)'),
901
+ crawlAs: z.enum(['browser', 'googlebot']).optional().describe('UA for the free HTTP fetch: browser (default, mimics a visit from Google — gets Reddit + mid-tier) or googlebot'),
902
+ },
903
+ }, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, crawlAs }) => {
904
+ const client = requireDfs(dfs);
905
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
906
+ try {
907
+ const fresh = gscFreshness(db.db);
908
+ if (!fresh.effectiveMax)
909
+ return { content: [{ type: 'text', text: `No synced GSC data for ${siteUrl} — run refresh_property first.` }], structuredContent: { error: 'empty', siteUrl } };
910
+ const today = new Date().toISOString().slice(0, 10);
911
+ const ownDomain = dfsHost(siteUrl);
912
+ const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
913
+ let targets;
914
+ if (urls?.length) {
915
+ targets = [];
916
+ for (const u of urls) {
917
+ const key = urlKey(u, { hostForm });
918
+ const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
919
+ FROM search_analytics WHERE page_key=? AND query IS NOT NULL AND date > date(?, '-28 days') AND date <= ? GROUP BY query ORDER BY imp DESC LIMIT 1`)
920
+ .get(key, fresh.effectiveMax, fresh.effectiveMax);
921
+ if (row?.query)
922
+ targets.push({ urlKey: key, query: row.query, impressions: row.imp, clicks: row.clk, position: Math.round(row.pos * 10) / 10, priorPosition: null, slipped: false, competingUrls: 0 });
923
+ }
924
+ }
925
+ else {
926
+ targets = selectReconTargets(db.db, { ...(limit != null ? { limit } : {}), ...(minImpressions != null ? { minImpressions } : {}), maxDate: fresh.effectiveMax });
927
+ }
928
+ if (!targets.length)
929
+ return { content: [{ type: 'text', text: `No recon targets found (declining/striking-distance pages, position 3-15, ${minImpressions ?? 300}+ impressions). Try a lower minImpressions or pass explicit urls.` }], structuredContent: { siteUrl, targets: [] } };
930
+ const results = [];
931
+ let cost = 0;
932
+ for (const t of targets) {
933
+ let own = null;
934
+ try {
935
+ own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
936
+ }
937
+ catch { /* page unreachable — classify on SERP alone */ }
938
+ const serpResp = await client.serpOrganic(t.query, location, 'en', 20);
939
+ cost += serpResp.cost;
940
+ const serp = parseSerpForRecon(serpResp, ownDomain);
941
+ const verdict = reconVerdict(serp);
942
+ const baseline = { organicRank: serp.ourOrganicRank, aioCitesUs: serp.aioCitesUs, gscPosition: t.position, gscImpressions: t.impressions, at: today };
943
+ persistReconPage(db.db, {
944
+ urlKey: t.urlKey, query: t.query, verdict: verdict.verdict, verdictNote: verdict.note,
945
+ organicRank: serp.ourOrganicRank, gscPosition: t.position, gscImpressions: t.impressions,
946
+ aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, videoPresent: serp.videoPresent,
947
+ hasProductSchema: own?.hasProductOrReview ?? false, schemaTypes: own?.jsonLdTypes ?? [],
948
+ competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
949
+ });
950
+ const todos = own ? deterministicTodos(t, own, serp, verdict, today) : [];
951
+ const inserted = todos.length ? insertTodos(db.db, t.urlKey, t.query, todos, baseline, 'auto') : 0;
952
+ let competitorContent = undefined;
953
+ if (scrapeCompetitors && (firecrawl || supadata)) {
954
+ // Build a routed candidate set: organic-above + a couple of AI-Overview references +
955
+ // one ranking video, deduped, our own domain removed. Each URL routes by domain
956
+ // (YouTube→supadata, Reddit→.json, else→firecrawl). Bounded to keep the payload sane.
957
+ const seen = new Set();
958
+ const candidates = [];
959
+ const add = (u) => { if (u && !seen.has(u) && dfsHost(u) !== ownDomain) {
960
+ seen.add(u);
961
+ candidates.push(u);
962
+ } };
963
+ serp.organicAbove.forEach(o => add(o.url));
964
+ serp.aioReferences.slice(0, 4).forEach(r => add(r.url));
965
+ serp.videoItems.slice(0, 1).forEach(v => add(v.url));
966
+ competitorContent = [];
967
+ for (const url of candidates.slice(0, 3)) {
968
+ competitorContent.push(await fetchCompetitorContent(url, { firecrawl, supadata }, { maxChars: 4000, ua: crawlAs }));
969
+ }
970
+ }
971
+ results.push({
972
+ urlKey: t.urlKey, query: t.query, impressions: t.impressions, gscPosition: t.position, priorPosition: t.priorPosition, slipped: t.slipped,
973
+ organicRank: serp.ourOrganicRank, aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, videoPresent: serp.videoPresent,
974
+ verdict: verdict.verdict, verdictNote: verdict.note,
975
+ ownHeadings: own?.headings.map(h => h.heading) ?? null, schemaTypes: own?.jsonLdTypes ?? null, hasProductSchema: own?.hasProductOrReview ?? null, dateModified: own?.dateModified ?? null,
976
+ todos, todosInserted: inserted,
977
+ competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
978
+ ...(competitorContent ? { competitorContent } : {}),
979
+ });
980
+ }
981
+ const md = `# Content recon — ${siteUrl}\n\n${results.length} page(s), ${results.reduce((s, r) => s + r.todosInserted, 0)} to-dos saved. DataForSEO SERP cost $${cost.toFixed(4)}.\n\n` +
982
+ results.map(r => {
983
+ const flags = [r.aioPresent ? (r.aioCitesUs ? 'AIO: cited' : 'AIO: NOT cited') : 'no AIO', r.videoPresent ? 'video pack' : null].filter(Boolean).join(' · ');
984
+ return `## ${r.urlKey}\n"${r.query}" — organic #${r.organicRank ?? '?'} (GSC avg ${r.gscPosition}${r.slipped ? `, slipped from ${r.priorPosition}` : ''}), ${r.impressions} impr. ${flags}\n**${r.verdict}** — ${r.verdictNote}\n` +
985
+ (r.todos.length ? '\nTo do:\n' + r.todos.map((t) => `- [${t.type}] ${t.action}`).join('\n') : '');
986
+ }).join('\n\n') +
987
+ (() => {
988
+ const errs = results.flatMap(r => (r.competitorContent ?? []).filter((c) => c.error));
989
+ if (!errs.length)
990
+ return '';
991
+ const needKey = errs.filter(c => /not set/i.test(c.error));
992
+ const blocked = errs.filter(c => !/not set/i.test(c.error));
993
+ let note = '';
994
+ if (blocked.length)
995
+ note += `\n\n${blocked.length} competitor(s) couldn't be fetched (bot-protection like Cloudflare, a block, or a fetch error): ${blocked.slice(0, 5).map(c => c.url).join(', ')}. If one matters, paste its copy here or drop a text file and I'll diff it in.`;
996
+ if (needKey.length)
997
+ note += `\n\n${needKey.length} competitor(s) skipped for a missing API key: ${[...new Set(needKey.map(c => c.error))].join('; ')}.`;
998
+ return note;
999
+ })() +
1000
+ `\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
1001
+ return { content: [{ type: 'text', text: md }], structuredContent: { siteUrl, cost, targets: results } };
1002
+ }
1003
+ finally {
1004
+ db.close();
1005
+ }
1006
+ });
1007
+ // save_recon_todo — the research session writes its content-gap / originality findings back
1008
+ // into the ledger against a page (source: 'research').
1009
+ server.registerTool('save_recon_todo', {
1010
+ title: 'Save content-recon to-dos (research writeback)',
1011
+ description: 'Write content-recon findings back into the trackable ledger for a page — the gaps and originality the research session found by diffing competitors (firecrawl) and videos (supadata) against our content. Each to-do is an action with a type (content-gap / originality / schema / freshness / format), rationale and evidence. Snapshots the page baseline so the fix\'s effect on rank/AIO-citation is measurable later. Run recon_targets first (it classifies the page and seeds the deterministic to-dos); this adds the judgement ones. Track everything with recon_todos.',
1012
+ inputSchema: {
1013
+ siteUrl: z.string(),
1014
+ urlKey: z.string().describe('The page (any URL form — normalised to its key)'),
1015
+ todos: z.array(z.object({
1016
+ action: z.string(),
1017
+ type: z.string().optional(),
1018
+ rationale: z.string().optional(),
1019
+ evidence: z.record(z.any()).optional(),
1020
+ priority: z.number().optional(),
1021
+ })).min(1),
1022
+ },
1023
+ }, async ({ siteUrl, urlKey: rawUrl, todos }) => {
1024
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1025
+ try {
1026
+ const key = urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' });
1027
+ const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position FROM recon_page WHERE url_key=?`).get(key);
1028
+ const baseline = page ? { organicRank: page.organic_rank, aioCitesUs: !!page.aio_cites_us, gscPosition: page.gsc_position, at: new Date().toISOString().slice(0, 10) } : {};
1029
+ const drafts = todos.map(t => ({ action: t.action, type: t.type ?? 'content-gap', rationale: t.rationale ?? '', evidence: t.evidence ?? {}, priority: t.priority ?? 0 }));
1030
+ const n = insertTodos(db.db, key, page?.query ?? '', drafts, baseline, 'research');
1031
+ return { content: [{ type: 'text', text: `Saved ${n} recon to-do(s) for ${key}${n < drafts.length ? ` (${drafts.length - n} already open)` : ''}. Track with recon_todos.` }], structuredContent: { urlKey: key, inserted: n } };
1032
+ }
1033
+ finally {
1034
+ db.close();
1035
+ }
1036
+ });
1037
+ // recon_todos — list, track and annotate the ledger. No id → list (optionally filtered);
1038
+ // id → update status and/or append a dated annotation, and optionally re-measure the outcome.
1039
+ server.registerTool('recon_todos', {
1040
+ title: 'List, track and annotate content-recon to-dos',
1041
+ description: 'The content-recon to-do board. With no id: list to-dos (optionally filter by page or status), grouped by page with each page\'s verdict — the pick-a-page-to-work-on surface, and the hand-off to content-machine. With id: update one to-do — set status (open → researching → drafted → shipped → dismissed) and/or append a dated annotation note (your own observations, the history). On status:shipped with remeasure:true it re-fetches the SERP and records the outcome, so you can see whether the fix moved you from AIO-uncited to cited, or up the organic ranks.',
1042
+ inputSchema: {
1043
+ siteUrl: z.string(),
1044
+ urlKey: z.string().optional().describe('Filter the list to one page'),
1045
+ status: z.enum(['open', 'researching', 'drafted', 'shipped', 'dismissed']).optional().describe('Filter the list, or the new status when id is set'),
1046
+ id: z.number().int().optional().describe('Update this to-do'),
1047
+ note: z.string().optional().describe('Append a dated annotation to this to-do'),
1048
+ remeasure: z.boolean().optional().describe('On status:shipped, re-fetch the SERP and record the outcome (paid)'),
1049
+ location: z.union([z.string(), z.number()]).optional(),
1050
+ },
1051
+ }, async ({ siteUrl, urlKey: rawUrl, status, id, note, remeasure, location }) => {
1052
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1053
+ try {
1054
+ if (id != null) {
1055
+ const row = db.db.prepare(`SELECT * FROM recon_todo WHERE id=?`).get(id);
1056
+ if (!row)
1057
+ throw new Error(`No recon to-do #${id}.`);
1058
+ let notes = row.notes ?? '';
1059
+ if (note)
1060
+ notes = (notes ? notes + '\n' : '') + `[${new Date().toISOString().slice(0, 10)}] ${note}`;
1061
+ let outcome = row.outcome;
1062
+ let outcomeNote = '';
1063
+ if (status === 'shipped' && remeasure && dfs) {
1064
+ const serpResp = await dfs.serpOrganic(row.query, location, 'en', 20);
1065
+ const serp = parseSerpForRecon(serpResp, dfsHost(siteUrl));
1066
+ outcome = JSON.stringify({ organicRank: serp.ourOrganicRank, aioCitesUs: serp.aioCitesUs, at: new Date().toISOString().slice(0, 10) });
1067
+ const base = row.baseline ? JSON.parse(row.baseline) : {};
1068
+ outcomeNote = ` Outcome: organic ${base.organicRank ?? '?'}→${serp.ourOrganicRank ?? '?'}, AIO cited ${base.aioCitesUs ? 'yes' : 'no'}→${serp.aioCitesUs ? 'yes' : 'no'}.`;
1069
+ }
1070
+ db.db.prepare(`UPDATE recon_todo SET status=COALESCE(?,status), notes=?, outcome=COALESCE(?,outcome), updated_at=datetime('now') WHERE id=?`)
1071
+ .run(status ?? null, notes, outcome ?? null, id);
1072
+ return { content: [{ type: 'text', text: `Updated to-do #${id}${status ? ` → ${status}` : ''}${note ? ' (note added)' : ''}.${outcomeNote}` }], structuredContent: { id, status: status ?? row.status, outcome: outcome ? JSON.parse(outcome) : null } };
1073
+ }
1074
+ const key = rawUrl ? urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' }) : null;
1075
+ const where = [];
1076
+ const args = [];
1077
+ if (key) {
1078
+ where.push('url_key=?');
1079
+ args.push(key);
1080
+ }
1081
+ if (status) {
1082
+ where.push('status=?');
1083
+ args.push(status);
1084
+ }
1085
+ const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
1086
+ FROM recon_todo ${where.length ? 'WHERE ' + where.join(' AND ') : ''} ORDER BY url_key, priority DESC`).all(...args);
1087
+ if (!rows.length)
1088
+ return { content: [{ type: 'text', text: `No recon to-dos${key ? ` for ${key}` : ''}${status ? ` with status ${status}` : ''}. Run recon_targets to generate some.` }], structuredContent: { todos: [] } };
1089
+ const byPage = new Map();
1090
+ for (const r of rows)
1091
+ (byPage.get(r.url_key) ?? byPage.set(r.url_key, []).get(r.url_key)).push(r);
1092
+ const verdictStmt = db.db.prepare(`SELECT verdict, verdict_note FROM recon_page WHERE url_key=?`);
1093
+ const md = `# Content-recon to-dos — ${rows.length} item(s)${status ? `, status ${status}` : ''}\n\n` +
1094
+ [...byPage.entries()].map(([url, items]) => {
1095
+ const v = verdictStmt.get(url);
1096
+ return `## ${url}${v ? `\n_${v.verdict}_ — ${v.verdict_note}` : ''}\n\n| # | Status | Type | Action |\n|---|---|---|---|\n` +
1097
+ items.map(i => `| ${i.id} | ${i.status} | ${i.type ?? ''} | ${String(i.action).replace(/\|/g, '\\|')} |`).join('\n') +
1098
+ (items.some(i => i.notes) ? '\n\nNotes:\n' + items.filter(i => i.notes).map(i => `- #${i.id}: ${String(i.notes).replace(/\n/g, ' / ')}`).join('\n') : '');
1099
+ }).join('\n\n') +
1100
+ browserLink(siteUrl);
1101
+ return { content: [{ type: 'text', text: md }], structuredContent: { todos: rows } };
1102
+ }
1103
+ finally {
1104
+ db.close();
1105
+ }
1106
+ });
1107
+ // keyword_list — demand-first clustering ("list mode"): a keyword list becomes topics
1108
+ // with own/weak/absent verdicts, clustered by the URL Google already answers them with.
1109
+ server.registerTool('keyword_list', {
1110
+ title: 'Cluster a keyword list into topics with own/weak/absent verdicts',
1111
+ description: 'Demand-first keyword clustering from stored data - NO paid calls. Give it a keyword list (or omit keywords to use your top 500 GSC queries) and it clusters them by the page Google ALREADY answers them with (two keywords that rank via the same URL belong together - the strongest clustering signal, free from your own GSC data), then groups non-ranking keywords lexically. Each cluster gets a deterministic verdict: OWN (best position <=3), WEAK (4-20), ABSENT (no ranking page), plus the ranking URL, summed 90-day impressions and clicks. The keyword-research workhorse: paste a client keyword list, get the topic map and where you stand. Chain with keyword_volume for market volumes on the interesting clusters, or draft_content for the absent ones.',
1112
+ inputSchema: {
1113
+ siteUrl: z.string(),
1114
+ keywords: z.array(z.string()).max(2000).optional().describe('The keyword list. Omit to derive from your top 500 GSC queries by impressions'),
1115
+ limit: z.number().int().min(1).max(200).optional().describe('Clusters to show in the table (default 40)'),
1116
+ },
1117
+ }, async ({ siteUrl, keywords, limit }) => {
1118
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1119
+ try {
1120
+ const fresh = gscFreshness(db.db);
1121
+ const r = clusterKeywordList(db.db, keywords, { maxDate: fresh.effectiveMax });
1122
+ const show = r.clusters.slice(0, limit ?? 40);
1123
+ const counts = { own: 0, weak: 0, absent: 0 };
1124
+ for (const c of r.clusters)
1125
+ counts[c.verdict]++;
1126
+ const rowsMd = show.map(c => `| ${c.head.replace(/\|/g, '\\|')} | ${c.verdict.toUpperCase()} | ${c.bestPosition ?? '-'} | ${fmtNum(c.impressions)} | ${c.keywords.length} | ${c.url ? c.url.replace(/\|/g, '\\|') : '-'} |`);
1127
+ const md = `**Keyword list clustered for ${siteUrl}** - ${r.totalKeywords} keywords (${r.derived ? 'derived from your GSC demand' : 'your list'}, ${r.inGsc} with GSC data) → ${r.clusters.length} clusters: ${counts.own} own · ${counts.weak} weak · ${counts.absent} absent\n\n` +
1128
+ `| Cluster (head term) | Verdict | Best pos | Impressions 90d | Keywords | Ranking URL |\n|---|---|---|---|---|---|\n` +
1129
+ rowsMd.join('\n') +
1130
+ `\n\nVerdicts are deterministic from your GSC positions. ABSENT clusters are content opportunities - chain into keyword_volume (market volume) or draft_content (a brief).`;
1131
+ return {
1132
+ content: [{ type: 'text', text: md }],
1133
+ structuredContent: { ...r, clusters: r.clusters.slice(0, 200) },
1134
+ };
1135
+ }
1136
+ finally {
1137
+ db.close();
1138
+ }
1139
+ });
791
1140
  server.registerTool('search_intent', {
792
1141
  title: 'Search intent classification (DataForSEO Labs)',
793
- description: 'Classify the search intent (informational / navigational / commercial / transactional) of keywords — primary label + probability + secondary intents. Use it to spot intent mismatch: e.g. a transactional product page ranking for an informational query (a common cause of high impressions / low CTR). Pass siteUrl to persist the intents so run_audit can surface intent-vs-pagetype-mismatch. Labs call, cached 20 days, up to 1000 keywords. Language-only (no location).',
1142
+ description: '[Paid: Labs, cheap, cached 20d, up to 1000 kw/call | Use for: intent mapping before content planning] Classify the search intent (informational / navigational / commercial / transactional) of keywords — primary label + probability + secondary intents. Use it to spot intent mismatch: e.g. a transactional product page ranking for an informational query (a common cause of high impressions / low CTR). Pass siteUrl to persist the intents so run_audit can surface intent-vs-pagetype-mismatch. Labs call, cached 20 days, up to 1000 keywords. Language-only (no location).',
794
1143
  inputSchema: { keywords: z.array(z.string()).min(1).max(1000), languageCode: z.string().optional(), siteUrl: z.string().optional() },
795
1144
  }, async ({ keywords, languageCode, siteUrl }) => {
796
1145
  const client = requireDfs(dfs);
@@ -824,7 +1173,7 @@ export function createServer() {
824
1173
  });
825
1174
  server.registerTool('page_lighthouse', {
826
1175
  title: 'Page Lighthouse — lab Core Web Vitals (DataForSEO On-Page)',
827
- description: 'Run a live Lighthouse audit for ONE url: lab Core Web Vitals (LCP, CLS, TBT, FCP, Speed Index), category scores (performance/SEO/best-practices/accessibility), and the top time-saving opportunities. Complements GSC/CrUX field data (aggregate + delayed) with on-demand lab data. Pass siteUrl to persist the CWV so run_audit can surface high-yield-cwv-fail. Paid On-Page call (~2000 credits), slow (~20–120s), cached 20 days.',
1176
+ description: '[Paid: On-Page, ONE URL per call | Use for: CWV evidence on a page that earns clicks] Run a live Lighthouse audit for ONE url: lab Core Web Vitals (LCP, CLS, TBT, FCP, Speed Index), category scores (performance/SEO/best-practices/accessibility), and the top time-saving opportunities. Complements GSC/CrUX field data (aggregate + delayed) with on-demand lab data. Pass siteUrl to persist the CWV so run_audit can surface high-yield-cwv-fail. Paid On-Page call (~2000 credits), slow (~20–120s), cached 20 days.',
828
1177
  inputSchema: { url: z.string().url(), forMobile: z.boolean().optional(), siteUrl: z.string().optional() },
829
1178
  }, async ({ url, forMobile, siteUrl }) => {
830
1179
  const client = requireDfs(dfs);
@@ -878,7 +1227,7 @@ export function createServer() {
878
1227
  });
879
1228
  server.registerTool('competitors_domain', {
880
1229
  title: 'Competitor domains (DataForSEO Labs)',
881
- description: 'Discover the domains competing with a target for the same organic keywords (ranked by keyword overlap), with intersection counts and organic traffic estimates. The seed list for content-gap analysis (feed these into page_intersection). Labs call, cached 20 days. Pass location + language for the right market.',
1230
+ description: '[Paid: Labs, cheap, cached 20d | Use for: naming the real competitor set - the opening move of competitive work] Discover the domains competing with a target for the same organic keywords (ranked by keyword overlap), with intersection counts and organic traffic estimates. The seed list for content-gap analysis (feed these into page_intersection). Labs call, cached 20 days. Pass location + language for the right market.',
882
1231
  inputSchema: { target: z.string(), location: z.union([z.string(), z.number()]).optional(), languageCode: z.string().optional(), limit: z.number().int().min(1).max(100).optional() },
883
1232
  }, async ({ target, location, languageCode, limit }) => {
884
1233
  const client = requireDfs(dfs);
@@ -900,7 +1249,7 @@ export function createServer() {
900
1249
  });
901
1250
  server.registerTool('page_intersection', {
902
1251
  title: 'Content gap — page intersection (DataForSEO Labs)',
903
- description: 'Find keywords that competitor pages rank for but your page does NOT (the content gap). Pass competitor URLs as `competitorUrls` (max 20; wildcards like https://site.com/blog/* allowed) and your own URL(s) in `excludePages` (max 10) to subtract. Returns gap keywords sorted by search volume, with each competitor’s rank. Labs call, cached 20 days. Pass location + language.',
1252
+ description: '[Paid: Labs, cheap, cached 20d | Use for: page-level content gap vs named rival URLs] Find keywords that competitor pages rank for but your page does NOT (the content gap). Pass competitor URLs as `competitorUrls` (max 20; wildcards like https://site.com/blog/* allowed) and your own URL(s) in `excludePages` (max 10) to subtract. Returns gap keywords sorted by search volume, with each competitor’s rank. Labs call, cached 20 days. Pass location + language.',
904
1253
  inputSchema: {
905
1254
  competitorUrls: z.array(z.string()).min(1).max(20),
906
1255
  excludePages: z.array(z.string()).max(10).optional(),
@@ -952,7 +1301,7 @@ export function createServer() {
952
1301
  };
953
1302
  server.registerTool('domain_visibility', {
954
1303
  title: 'Domain visibility over time (DataForSEO Labs)',
955
- description: 'Monthly organic visibility for ANY domain or subdomain — no Search Console access needed: ranking-keyword totals, position distribution (1–3 / 4–10 / 11–20 / 21–100) and estimated traffic value (ETV) per month, plus a trend verdict. The Semrush-style "organic overview / visibility over time" for you or a competitor. Labs call, cached 20 days. Pass location (name or code) for the right market. Grain: month × domain. Joins: domain → Labs/backlinks tools; period → rank_history.',
1304
+ description: '[Paid: Labs, cheap, cached 20d | Use for: visibility-over-time trend, any domain, no GSC needed] Monthly organic visibility for ANY domain or subdomain — no Search Console access needed: ranking-keyword totals, position distribution (1–3 / 4–10 / 11–20 / 21–100) and estimated traffic value (ETV) per month, plus a trend verdict. The Semrush-style "organic overview / visibility over time" for you or a competitor. Labs call, cached 20 days. Pass location (name or code) for the right market. Grain: month × domain. Joins: domain → Labs/backlinks tools; period → rank_history.',
956
1305
  inputSchema: {
957
1306
  target: z.string().describe('Domain or subdomain, no scheme (e.g. example.com or blog.example.com)'),
958
1307
  location: z.union([z.string(), z.number()]).optional(),
@@ -1008,7 +1357,7 @@ export function createServer() {
1008
1357
  });
1009
1358
  server.registerTool('top_pages', {
1010
1359
  title: 'Top ranking pages on a domain (DataForSEO Labs)',
1011
- description: 'The top organic pages of ANY domain or subdomain, ranked by estimated traffic value (ETV): page, ranking-keyword count, ETV and top-3 / top-10 keyword counts. The Semrush-style "top pages" view — works on competitors, no crawl or GSC needed. Labs call, cached 20 days. Pass location for the right market. Grain: page × domain. Joins: domain → Labs tools; URL → pages.url_key on your own property.',
1360
+ description: '[Paid: Labs, cheap, cached 20d | Use for: where rival traffic actually lands] The top organic pages of ANY domain or subdomain, ranked by estimated traffic value (ETV): page, ranking-keyword count, ETV and top-3 / top-10 keyword counts. The Semrush-style "top pages" view — works on competitors, no crawl or GSC needed. Labs call, cached 20 days. Pass location for the right market. Grain: page × domain. Joins: domain → Labs tools; URL → pages.url_key on your own property.',
1012
1361
  inputSchema: {
1013
1362
  target: z.string().describe('Domain or subdomain, no scheme'),
1014
1363
  location: z.union([z.string(), z.number()]).optional(),
@@ -1055,7 +1404,7 @@ export function createServer() {
1055
1404
  });
1056
1405
  server.registerTool('ranked_keywords', {
1057
1406
  title: 'Ranked keywords for a domain / URL / folder (DataForSEO Labs)',
1058
- description: 'Every keyword a target ranks for in Google organic — scope it to a whole domain, a subdomain, ONE page (scope:url with the full URL), or a subfolder (scope:folder + folder:"/blog/"). Returns keyword, position, search volume, ETV, the ranking URL, plus keyword difficulty, search intent and SERP features where the response carries them. Set aioOnly:true to list only keywords whose SERP shows a Google AI Overview (your AIO exposure list; per-keyword CITATION checking uses the SERP tools — see the cookbook). The Semrush-style "keywords a page or site ranks for" view — works on any site. Labs call, cached 20 days. Pass location for the right market. Grain: keyword × target. Joins: keyword → GSC search_analytics.query; URL → pages.url_key.',
1407
+ description: '[Paid: Labs, cheap, cached 20d - ONE call replaces any keyword loop | Use for: full keyword footprint of a domain/page/folder] Every keyword a target ranks for in Google organic — scope it to a whole domain, a subdomain, ONE page (scope:url with the full URL), or a subfolder (scope:folder + folder:"/blog/"). Returns keyword, position, search volume, ETV, the ranking URL, plus keyword difficulty, search intent and SERP features where the response carries them. Set aioOnly:true to list only keywords whose SERP shows a Google AI Overview (your AIO exposure list; per-keyword CITATION checking uses the SERP tools — see the cookbook). The Semrush-style "keywords a page or site ranks for" view — works on any site. Labs call, cached 20 days. Pass location for the right market. Grain: keyword × target. Joins: keyword → GSC search_analytics.query; URL → pages.url_key.',
1059
1408
  inputSchema: {
1060
1409
  target: z.string().describe('Domain, subdomain, or full URL (full URL required for scope:url)'),
1061
1410
  scope: z.enum(['domain', 'subdomain', 'url', 'folder']).optional(),
@@ -1151,9 +1500,112 @@ export function createServer() {
1151
1500
  structuredContent: { target: dfsTarget, scope: mode, aioOnly: aioOnly ?? false, totalCount, rowsTotal: kws.length, keywords: kws.slice(0, 100), cached: r.cached, cost: r.cost },
1152
1501
  };
1153
1502
  });
1503
+ // serp_features — the SERP-feature footprint: "how much of my market do AI Overviews,
1504
+ // snippets and other features sit on, and do I already rank page 1 there?" One cached
1505
+ // Labs pull, volume-weighted; persisted so the dashboard can chart it.
1506
+ server.registerTool('serp_features', {
1507
+ title: 'SERP-feature footprint (AI Overviews, snippets, PAA)',
1508
+ description: '[Paid: Labs, ONE cached call | Use for: AI-Overview / zero-click exposure with real numbers] How much of your keyword universe carries each SERP feature - AI Overviews, featured snippets, People Also Ask, shopping, video - weighted by search volume, and how much of that volume you already rank page 1 for. ONE DataForSEO Labs ranked_keywords pull (top-volume sample, cached 20 days, never per-keyword SERP loops). Answers "how exposed are we to AI Overviews / zero-click?" with real numbers. Deterministic: feature PRESENCE (from the Labs index). Judgement proxy: page-1 rank stands in for feature ownership - true ownership needs per-keyword SERP calls (see ranked_keywords aioOnly + the cookbook). Persists to the property DB so the dashboard charts it.',
1509
+ inputSchema: {
1510
+ siteUrl: z.string().describe('The GSC property - results persist to its database'),
1511
+ target: z.string().optional().describe('Override the analysed domain (defaults to the property host)'),
1512
+ location: z.union([z.string(), z.number()]).optional(),
1513
+ languageCode: z.string().optional(),
1514
+ limit: z.number().int().min(100).max(1000).optional().describe('Keyword sample size (default 1000, ordered by volume)'),
1515
+ },
1516
+ }, async ({ siteUrl, target, location, languageCode, limit }) => {
1517
+ const client = requireDfs(dfs);
1518
+ const domain = dfsHost(target ?? siteUrl);
1519
+ const r = await client.rankedKeywords(domain, location, languageCode ?? 'en', limit ?? 1000, 'keyword_data.keyword_info.search_volume,desc');
1520
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1521
+ const sample = items.map((it) => {
1522
+ const kd = it.keyword_data ?? {};
1523
+ const serp = it.ranked_serp_element?.serp_item ?? {};
1524
+ return {
1525
+ keyword: kd.keyword,
1526
+ position: Number(serp.rank_absolute) || null,
1527
+ searchVolume: kd.keyword_info?.search_volume ?? null,
1528
+ serpFeatures: Array.isArray(kd.serp_info?.serp_item_types) ? kd.serp_info.serp_item_types : null,
1529
+ };
1530
+ }).filter(s => s.keyword);
1531
+ const fp = computeSerpFootprint(domain, typeof location === 'string' ? location : location != null ? String(location) : null, sample);
1532
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1533
+ try {
1534
+ persistSerpFootprint(db.db, fp);
1535
+ }
1536
+ finally {
1537
+ db.close();
1538
+ }
1539
+ const lines = fp.features.map(f => `| ${f.label} | ${f.volumeSharePct}% | ${fmtNum(f.volume)} | ${f.keywords} | ${f.page1SharePct}% |`);
1540
+ const md = `**SERP-feature footprint for ${domain}** - sample: top ${fp.sampleKeywords} keywords by volume (${fmtNum(fp.sampleVolume)} searches/mo)\n\n` +
1541
+ `| Feature | % of sample volume | Volume on feature SERPs | Keywords | You rank page 1 (share of that volume) |\n|---|---|---|---|---|\n` +
1542
+ lines.join('\n') +
1543
+ `\n\nPresence is deterministic (Labs index); "page 1" is an ownership PROXY - per-keyword citation/ownership checks are the SERP tools. ${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}).`} Persisted for the dashboard.` +
1544
+ browserLink(siteUrl);
1545
+ return {
1546
+ content: [{ type: 'text', text: md }],
1547
+ structuredContent: fp,
1548
+ };
1549
+ });
1550
+ // market_sizing — Market Sizing and Prioritisation: the organic market read that opens
1551
+ // an engagement. Top-down Labs pulls only (client + <=4 rivals), cached, ~$0.65 worst case.
1552
+ server.registerTool('market_sizing', {
1553
+ title: 'Market Sizing and Prioritisation (share of voice vs competitors)',
1554
+ description: '[Paid: Labs, one cached call per domain (<=5) | Use for: sizing the organic market and who owns it] Build the organic market map: your domain plus up to 4 named competitors, ONE cached Labs ranked_keywords pull each, unioned into a keyword universe. Returns total monthly demand (deduplicated search volume), each domain\'s share of voice (ETV share) overall and per topic cluster, and the leader per cluster - the "here is the market, here is who owns it, here is where to attack" table that opens an engagement. Deterministic: the competitor set and ranked keywords (Labs index). Judgement: ETV is DataForSEO\'s CTR-curve traffic estimate - the SoV percentages inherit that. Persists to the property DB for the dashboard chart. Get the competitor set from competitors_domain first if unsure.',
1555
+ inputSchema: {
1556
+ siteUrl: z.string(),
1557
+ competitors: z.array(z.string()).min(1).max(4).describe('Competitor domains, e.g. ["rival.com", "other.co.uk"]'),
1558
+ location: z.union([z.string(), z.number()]).optional(),
1559
+ languageCode: z.string().optional(),
1560
+ limitPerDomain: z.number().int().min(100).max(1000).optional().describe('Keywords sampled per domain (default 1000, by volume)'),
1561
+ },
1562
+ }, async ({ siteUrl, competitors, location, languageCode, limitPerDomain }) => {
1563
+ const client = requireDfs(dfs);
1564
+ const own = dfsHost(siteUrl);
1565
+ const domains = [own, ...competitors.map(c => dfsHost(c)).filter(c => c && c !== own)];
1566
+ const inputs = [];
1567
+ let cost = 0;
1568
+ let cachedAll = true;
1569
+ for (const domain of domains) { // sequential - the client serialises anyway
1570
+ const r = await client.rankedKeywords(domain, location, languageCode ?? 'en', limitPerDomain ?? 1000, 'keyword_data.keyword_info.search_volume,desc');
1571
+ cost += r.cost;
1572
+ cachedAll = cachedAll && r.cached;
1573
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1574
+ inputs.push({
1575
+ domain,
1576
+ keywords: items.map((it) => ({
1577
+ keyword: it.keyword_data?.keyword,
1578
+ volume: it.keyword_data?.keyword_info?.search_volume ?? null,
1579
+ position: Number(it.ranked_serp_element?.serp_item?.rank_absolute) || null,
1580
+ etv: it.ranked_serp_element?.serp_item?.etv != null ? Number(it.ranked_serp_element.serp_item.etv) : null,
1581
+ })).filter((k) => k.keyword),
1582
+ });
1583
+ }
1584
+ const m = computeMarketSizing(inputs, typeof location === 'string' ? location : location != null ? String(location) : null);
1585
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1586
+ try {
1587
+ persistMarketSizing(db.db, m);
1588
+ }
1589
+ finally {
1590
+ db.close();
1591
+ }
1592
+ const sovLine = m.domains.map(d => `${d === own ? '**' + d + '**' : d} ${m.sovByDomain[d]}%`).join(' · ');
1593
+ const clusterRows = m.clusters.map(c => `| ${c.head.replace(/\|/g, '\\|')} | ${fmtNum(c.volume)} | ${c.keywords} | ${c.leader === own ? '**you**' : (c.leader ?? '-')} | ${m.domains.map(d => `${c.sov[d]}%`).join(' / ')} |`);
1594
+ const md = `**Market Sizing and Prioritisation** - ${m.domains.join(' vs ')}\n\n` +
1595
+ `Universe: ${fmtNum(m.universeKeywords)} unique keywords, **${fmtNum(m.universeVolume)} searches/mo** total demand.\n` +
1596
+ `Share of voice (ETV share): ${sovLine}\n\n` +
1597
+ `| Topic cluster | Volume/mo | Keywords | Leader | SoV (${m.domains.join(' / ')}) |\n|---|---|---|---|---|\n` +
1598
+ clusterRows.join('\n') +
1599
+ `\n\nD: competitor set + ranked keywords (Labs index). N: ETV is a CTR-curve estimate - SoV inherits it. ${cachedAll ? 'All cached.' : `Live cost $${cost.toFixed(4)}.`} Persisted for the dashboard.` +
1600
+ browserLink(siteUrl);
1601
+ return {
1602
+ content: [{ type: 'text', text: md }],
1603
+ structuredContent: m,
1604
+ };
1605
+ });
1154
1606
  server.registerTool('topic_gaps', {
1155
1607
  title: 'Topic gaps — what to cover to be expert in your space (Labs + GSC)',
1156
- description: 'What related topics should this site cover to be seen as expert in its space? Pulls competitor keyword footprints (DataForSEO Labs ranked_keywords, ≤4 bounded + 20-day-cached calls), subtracts everything YOU already surface for (every GSC query with impressions) or already have a page about (title/H1/slug near-match), clusters the surviving gap keywords lexically, and scores each topic by summed search volume × how many competitors rank there. Each topic names the owning competitor with an example URL, your nearest existing page to build from, and (when resolve_entities has run) whether it sits beside entities you already cover. Pass competitors explicitly (max 3) or let it derive the top 2 from ranking overlap. Needs synced GSC data + DataForSEO credentials.',
1608
+ description: '[Paid: Labs, <=4 cached calls | Use for: topic-level gap plan vs competitors] What related topics should this site cover to be seen as expert in its space? Pulls competitor keyword footprints (DataForSEO Labs ranked_keywords, ≤4 bounded + 20-day-cached calls), subtracts everything YOU already surface for (every GSC query with impressions) or already have a page about (title/H1/slug near-match), clusters the surviving gap keywords lexically, and scores each topic by summed search volume × how many competitors rank there. Each topic names the owning competitor with an example URL, your nearest existing page to build from, and (when resolve_entities has run) whether it sits beside entities you already cover. Pass competitors explicitly (max 3) or let it derive the top 2 from ranking overlap. Needs synced GSC data + DataForSEO credentials.',
1157
1609
  inputSchema: {
1158
1610
  siteUrl: z.string(),
1159
1611
  competitors: z.array(z.string()).max(3).optional().describe('Competitor domains (max 3, e.g. ["rival.com"]). Omitted → derived via competitors_domain'),
@@ -1286,7 +1738,7 @@ export function createServer() {
1286
1738
  }
1287
1739
  const c = data.summary?.current;
1288
1740
  const summary = `Dashboard opened for ${siteUrl} — ${c?.clicks ?? 0} clicks / ${c?.impressions ?? 0} impressions (last 28d)` +
1289
- `${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.`;
1741
+ `${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.` + browserLink(siteUrl);
1290
1742
  return { content: [{ type: 'text', text: summary }], structuredContent: { siteUrl } };
1291
1743
  });
1292
1744
  // App-only data tool: the dashboard widget calls this via app.callServerTool to fetch its
@@ -1316,21 +1768,73 @@ export function createServer() {
1316
1768
  const tpl = readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8');
1317
1769
  const json = JSON.stringify(data).replace(/</g, '\\u003c'); // prevent </script> breakout
1318
1770
  const inject = `<script>window.__DASH_FIXTURE__=${json};window.__DASH_THEME__=${JSON.stringify(theme ?? 'light')};</script>`;
1319
- const html = tpl.replace(/<head([^>]*)>/i, `<head$1>${inject}`);
1771
+ // Replacement FUNCTION, not string — crawl data containing $& / $' would otherwise
1772
+ // be interpreted as String.replace substitution patterns and corrupt the report.
1773
+ const html = tpl.replace(/<head([^>]*)>/i, (_m, attrs) => `<head${attrs}>${inject}`);
1320
1774
  const dir = path.join(dataDir(), 'reports');
1321
1775
  mkdirSync(dir, { recursive: true });
1322
1776
  const file = path.join(dir, `${sanitizeProperty(siteUrl)}-dashboard.html`);
1323
1777
  writeFileSync(file, html);
1324
1778
  return {
1325
- content: [{ type: 'text', text: `Report saved: ${file}\nOpen it in any browser for the full interactive dashboard (${data.findings?.total ?? 0} findings). Shareable — send it to a client as-is.` }],
1779
+ content: [{ type: 'text', text: `Report saved: ${file}\nOpen it in any browser for the full interactive dashboard (${data.findings?.total ?? 0} findings). Shareable — send it to a client as-is.` + browserLink(siteUrl) }],
1326
1780
  structuredContent: { path: file, siteUrl, findings: data.findings?.total ?? 0, bytes: html.length },
1327
1781
  };
1328
1782
  });
1783
+ // serve_dashboard — the local webserver delivery surface: the full dashboard in a real
1784
+ // browser tab (live data, property switcher, native downloads), no MCP-App host needed.
1785
+ server.registerTool('serve_dashboard', {
1786
+ title: 'Serve the dashboard on a local webserver',
1787
+ description: 'Start a localhost-only webserver and return a URL that opens the full interactive dashboard in your browser — live data straight from the local database (always current, unlike export_report snapshots), a property switcher, working CSV downloads, and no host widget limits. The server stays up while the MCP server runs; call again with stop=true to shut it down. Localhost only — nothing is exposed to the network.',
1788
+ inputSchema: { siteUrl: z.string().optional(), port: z.number().int().min(1024).max(65535).optional(), stop: z.boolean().optional() },
1789
+ }, async ({ siteUrl, port, stop }) => {
1790
+ if (stop) {
1791
+ const was = dashboardServerUrl();
1792
+ const stopped = await stopDashboardServer();
1793
+ return {
1794
+ content: [{ type: 'text', text: stopped ? `Dashboard server stopped (was ${was}).` : 'No dashboard server running.' }],
1795
+ structuredContent: { stopped },
1796
+ };
1797
+ }
1798
+ const { url } = await startDashboardServer({
1799
+ dataDir,
1800
+ uiHtml: () => readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8'),
1801
+ call: {
1802
+ get_dashboard_data: async (a) => {
1803
+ const want = String(a.siteUrl ?? '');
1804
+ // Only serve properties that actually exist locally — getDashboardData would
1805
+ // otherwise CREATE an empty DB file for any bogus siteUrl posted at the API.
1806
+ if (!listLocalProperties(dataDir()).some(p => p.siteUrl === want))
1807
+ throw new Error(`unknown property ${want}`);
1808
+ return getDashboardData(dataDir(), want);
1809
+ },
1810
+ related_terms: async (a) => {
1811
+ const r = await requireDfs(dfs).relatedTerms(String(a.keyword ?? ''), a.location, a.languageCode);
1812
+ return r;
1813
+ },
1814
+ keyword_volume: async (a) => {
1815
+ const r = await requireDfs(dfs).searchVolume(a.keywords ?? [], a.location, a.languageCode);
1816
+ const items = (r.tasks[0]?.result ?? []).map((k) => ({
1817
+ keyword: k.keyword, searchVolume: k.search_volume, cpc: k.cpc, competition: k.competition,
1818
+ }));
1819
+ return { keywords: items, cached: r.cached, cost: r.cost };
1820
+ },
1821
+ },
1822
+ ...(port != null ? { port } : {}),
1823
+ });
1824
+ const open = siteUrl ? `${url}/dashboard?siteUrl=${encodeURIComponent(siteUrl)}` : `${url}/`;
1825
+ const portNote = port != null && !url.endsWith(`:${port}`)
1826
+ ? ` (already running on its original port — requested port ${port} ignored; stop=true first to move it)`
1827
+ : '';
1828
+ return {
1829
+ content: [{ type: 'text', text: `Dashboard server running — open ${open} in your browser${portNote}. Live data, all charts, CSV downloads, property switcher. Stays up while this MCP server runs; serve_dashboard stop=true to stop it.` }],
1830
+ structuredContent: { url: open, base: url },
1831
+ };
1832
+ });
1329
1833
  // pull_backlinks — on-demand backlink profile (DataForSEO, paid + 20-day cached). Powers
1330
1834
  // backlinks-to-404 (the big quick win), top-linked pages, and true-orphan detection.
1331
1835
  server.registerTool('pull_backlinks', {
1332
1836
  title: 'Pull backlink profile (DataForSEO)',
1333
- description: 'Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
1837
+ description: '[Paid: Backlinks subscription (separate - 40204 = not activated), cached 20d | Use for: authority + dead-backlink recovery] Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
1334
1838
  inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(1000).optional(), statusLimit: z.number().int().min(0).max(1000).optional() },
1335
1839
  }, async ({ siteUrl, limit, statusLimit }) => {
1336
1840
  const bl = requireDfs(backlinks);