@houtini/seo-audit-console 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js CHANGED
@@ -8,6 +8,17 @@ import path from 'node:path';
8
8
  import { fileURLToPath } from 'node:url';
9
9
  import { z } from 'zod';
10
10
  import { getDashboardData } from './core/dashboardData.js';
11
+ import { startDashboardServer, stopDashboardServer, dashboardServerUrl, listLocalProperties } from './core/webServer.js';
12
+ import { computeSerpFootprint, persistSerpFootprint } from './core/serpFootprint.js';
13
+ import { computeMarketSizing, persistMarketSizing } from './core/marketSizing.js';
14
+ /** A clickable browser-dashboard link appended to tool outputs — the user should always
15
+ * know the full interactive report is one click away (or one serve_dashboard call away). */
16
+ function browserLink(siteUrl) {
17
+ const base = dashboardServerUrl();
18
+ if (base)
19
+ return `\n\nBrowser dashboard: ${base}/dashboard${siteUrl ? `?siteUrl=${encodeURIComponent(siteUrl)}` : ''}`;
20
+ return `\n\nTip: run serve_dashboard to open the full interactive dashboard in your browser.`;
21
+ }
11
22
  import { runAudit, runSingleCheck, listChecks } from './audit/engine.js';
12
23
  import { buildAuditMarkdown } from './audit/report.js';
13
24
  import { diffLatest, buildDriftMarkdown } from './audit/drift.js';
@@ -34,6 +45,8 @@ import { Backlinks } from './core/Backlinks.js';
34
45
  import { WikidataClient } from './core/WikidataClient.js';
35
46
  import { Entities } from './core/Entities.js';
36
47
  import { JobManager } from './core/JobManager.js';
48
+ import { gscFreshness } from './core/gscFreshness.js';
49
+ import { clusterKeywordList } from './audit/keywordList.js';
37
50
  const SERVER_NAME = 'seo-audit-console';
38
51
  const SERVER_VERSION = JSON.parse(readFileSync(new URL('../package.json', import.meta.url), 'utf8')).version;
39
52
  // Where per-property crawl/audit DBs live. Resolution order (computed once per run):
@@ -56,8 +69,10 @@ export function dataDir() {
56
69
  return RESOLVED_DATA_DIR;
57
70
  }
58
71
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
59
- const DASHBOARD_URI = 'ui://dashboard/main.html';
60
- const SYNC_PROGRESS_URI = 'ui://sync-progress/main.html';
72
+ // Version the widget URIs: hosts may cache ui:// resources by URI indefinitely, so an
73
+ // unversioned URI can pin users to a stale (or broken) cached bundle across releases.
74
+ const DASHBOARD_URI = `ui://dashboard/main-${SERVER_VERSION}.html`;
75
+ const SYNC_PROGRESS_URI = `ui://sync-progress/main-${SERVER_VERSION}.html`;
61
76
  // Must match the categories actually used by CHECKS (src/audit/checks.ts) so category
62
77
  // filters never silently return empty. (Was listing performance/agentic/integrity/war-stories
63
78
  // which no check uses, and omitting content/security which checks do use.)
@@ -89,6 +104,18 @@ ARCHETYPE CHAINS (compose along these lines):
89
104
  2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
90
105
  3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
91
106
 
107
+ COST DISCIPLINE (behave like a strategist who knows the margins):
108
+ - Free and instant, use liberally: everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
109
+ - Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
110
+ - Paid per-keyword (SERP): related_terms and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
111
+ - Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
112
+
113
+ AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
114
+ 1. Baseline: refresh_property → run_audit → serve_dashboard (share the URL).
115
+ 2. Market: serp_features (feature exposure) + domain_visibility + competitors_domain → topic_gaps vs the named rivals.
116
+ 3. Content plan: suggest_pages (demand you already have) + topic_gaps (demand rivals own) → draft_content briefs.
117
+ 4. Fix cycle: fix_finding per top finding → re-crawl → detect_changes to prove the fix landed.
118
+
92
119
  Before planning ANY complex multi-source question, call composition_cookbook — it returns the full data-surface map and worked recipes using these exact tool and table names.`;
93
120
  const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
94
121
 
@@ -132,6 +159,14 @@ Raw access: query_audit runs any single check with full evidence; every table ab
132
159
  - **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
133
160
  - **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
134
161
 
162
+ ## Agency engagement recipes (the deliverable arc)
163
+
164
+ - **Week-one baseline:** refresh_property → run_audit → serve_dashboard. Share the dashboard URL; the ranked findings ARE the technical workstream.
165
+ - **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
166
+ - **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
167
+ - **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
168
+ - **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
169
+
135
170
  Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
136
171
  // The check catalogue rendered as markdown — single source of truth is listChecks();
137
172
  // shared by the seo-audit://checks-reference resource (and buildable for any category subset).
@@ -334,7 +369,7 @@ export function createServer() {
334
369
  }, async ({ siteUrl, scope, categories, includeJudgement }) => {
335
370
  const result = runAudit(dataDir(), siteUrl, { scope, categories, includeJudgement });
336
371
  return {
337
- content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) }],
372
+ content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) + browserLink(siteUrl) }],
338
373
  structuredContent: result,
339
374
  };
340
375
  });
@@ -738,7 +773,7 @@ export function createServer() {
738
773
  // ── DataForSEO (cached 20 days, single-worker) ──────────────────────────
739
774
  server.registerTool('keyword_volume', {
740
775
  title: 'Keyword search volume (DataForSEO)',
741
- description: 'True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
776
+ description: '[Paid: Keywords API, cheap, cached 20d | Use for: demand sizing] True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
742
777
  inputSchema: {
743
778
  keywords: z.array(z.string()).min(1).max(700),
744
779
  location: z.union([z.string(), z.number()]).optional(),
@@ -757,7 +792,7 @@ export function createServer() {
757
792
  });
758
793
  server.registerTool('related_terms', {
759
794
  title: 'Related terms (People Also Ask + related searches)',
760
- description: 'People Also Ask questions and related searches for a keyword (DataForSEO SERP). Powers click-through "related terms" on the keyword charts. SERP call — cached 20 days.',
795
+ description: '[Paid: SERP call PER KEYWORD - only for clicked/explicit keywords, never lists | Use for: the SERP context of a single keyword] People Also Ask questions and related searches for a keyword (DataForSEO SERP). Powers click-through "related terms" on the keyword charts. SERP call — cached 20 days.',
761
796
  inputSchema: { keyword: z.string(), location: z.union([z.string(), z.number()]).optional(), languageCode: z.string().optional() },
762
797
  }, async ({ keyword, location, languageCode }) => {
763
798
  const client = requireDfs(dfs);
@@ -786,9 +821,95 @@ export function createServer() {
786
821
  db.close();
787
822
  }
788
823
  });
824
+ // content_opportunities — the content marketer's report: everything the stored data
825
+ // says about what to WRITE, REFRESH and REWRITE, in one free composition.
826
+ server.registerTool('content_opportunities', {
827
+ title: 'Content opportunity report (write / refresh / rewrite)',
828
+ description: 'The content marketer\'s report, composed entirely from stored data - NO paid calls. Four sections: WRITE NEXT (new pages proposed from queries you already earn impressions for but have no winning page - suggest_pages), REFRESH NOW (pages that lost 20%+ of their clicks vs the prior period - content decay), REWRITE SNIPPETS (page-1 rankings earning far below expected CTR - title/meta rewrites, the fastest wins), and STRENGTHEN (keyword clusters where you rank 4-20 - one push from the money positions). Every line traces to real Search Console data. Chain into draft_content for a brief, or keyword_volume to size a cluster against the market.',
829
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(3).max(30).optional().describe('Rows per section (default 10)') },
830
+ }, async ({ siteUrl, limit }) => {
831
+ const n = limit ?? 10;
832
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
833
+ let proposals;
834
+ let weak;
835
+ try {
836
+ proposals = suggestPages(db.db, { maxProposals: n }).proposals;
837
+ const fresh = gscFreshness(db.db);
838
+ weak = clusterKeywordList(db.db, undefined, { maxDate: fresh.effectiveMax }).clusters
839
+ .filter(c => c.verdict === 'weak' && c.bestPosition != null && c.bestPosition >= 4).slice(0, n);
840
+ }
841
+ finally {
842
+ db.close();
843
+ }
844
+ const dash = getDashboardData(dataDir(), siteUrl);
845
+ const decay = (dash.contentDecay ?? []).slice(0, n);
846
+ const snippets = (dash.quickWins ?? []).filter(q => q.type === 'snippet').sort((a, b) => b.potential - a.potential).slice(0, n);
847
+ const sect = [];
848
+ sect.push(`## 1. Write next - demand you already have, no winning page\n` + (proposals.length
849
+ ? `| Proposed page (head term) | Impressions/mo | Best pos today | Queries |\n|---|---|---|---|\n` +
850
+ proposals.map(p => `| ${p.headTerm.replace(/\|/g, '\\|')} | ${fmtNum(p.totalImpressions)} | ${p.bestPosition} | ${p.queries.length} |`).join('\n')
851
+ : '_No unserved-demand gaps found._'));
852
+ sect.push(`## 2. Refresh now - pages losing clicks\n` + (decay.length
853
+ ? `| Page | Clicks were | Now | Drop | Clicks lost |\n|---|---|---|---|---|\n` +
854
+ decay.map(d => `| ${d.urlKey.replace(/\|/g, '\\|')} | ${fmtNum(d.prevClicks)} | ${fmtNum(d.clicks)} | ${d.dropPct}% | ${fmtNum(d.lost)} |`).join('\n')
855
+ : '_No significant decay - nothing lost 20%+ of its clicks._'));
856
+ sect.push(`## 3. Rewrite snippets - ranking well, under-clicked\n` + (snippets.length
857
+ ? `| Query | Position | CTR | Expected | Clicks recoverable/mo |\n|---|---|---|---|---|\n` +
858
+ snippets.map(q => `| ${q.query.replace(/\|/g, '\\|')} | ${q.position} | ${(q.ctr * 100).toFixed(1)}% | ${(q.expectedCtr * 100).toFixed(1)}% | ${fmtNum(Math.round(q.potential))} |`).join('\n')
859
+ : '_No page-1 CTR gaps found._'));
860
+ sect.push(`## 4. Strengthen - clusters one push from the money positions\n` + (weak.length
861
+ ? `| Cluster | Best pos | Impressions 90d | Keywords | Ranking URL |\n|---|---|---|---|---|\n` +
862
+ weak.map(c => `| ${c.head.replace(/\|/g, '\\|')} | ${c.bestPosition} | ${fmtNum(c.impressions)} | ${c.keywords.length} | ${c.url ?? '-'} |`).join('\n')
863
+ : '_No weak clusters - your rankings are polarised own/absent._'));
864
+ const md = `# Content opportunity report - ${siteUrl}\n\nEverything below is from your own Search Console + crawl data (no paid calls, no invented keywords).\n\n` +
865
+ sect.join('\n\n') +
866
+ `\n\nNext moves: draft_content for a brief on any row · keyword_volume to size a cluster against the market · topic_gaps to add what competitors own.` +
867
+ browserLink(siteUrl);
868
+ return {
869
+ content: [{ type: 'text', text: md }],
870
+ structuredContent: {
871
+ siteUrl,
872
+ writeNext: proposals, refreshNow: decay, rewriteSnippets: snippets,
873
+ strengthen: weak.map(c => ({ head: c.head, bestPosition: c.bestPosition, impressions: c.impressions, url: c.url, keywords: c.keywords.length })),
874
+ },
875
+ };
876
+ });
877
+ // keyword_list — demand-first clustering ("list mode"): a keyword list becomes topics
878
+ // with own/weak/absent verdicts, clustered by the URL Google already answers them with.
879
+ server.registerTool('keyword_list', {
880
+ title: 'Cluster a keyword list into topics with own/weak/absent verdicts',
881
+ description: 'Demand-first keyword clustering from stored data - NO paid calls. Give it a keyword list (or omit keywords to use your top 500 GSC queries) and it clusters them by the page Google ALREADY answers them with (two keywords that rank via the same URL belong together - the strongest clustering signal, free from your own GSC data), then groups non-ranking keywords lexically. Each cluster gets a deterministic verdict: OWN (best position <=3), WEAK (4-20), ABSENT (no ranking page), plus the ranking URL, summed 90-day impressions and clicks. The keyword-research workhorse: paste a client keyword list, get the topic map and where you stand. Chain with keyword_volume for market volumes on the interesting clusters, or draft_content for the absent ones.',
882
+ inputSchema: {
883
+ siteUrl: z.string(),
884
+ keywords: z.array(z.string()).max(2000).optional().describe('The keyword list. Omit to derive from your top 500 GSC queries by impressions'),
885
+ limit: z.number().int().min(1).max(200).optional().describe('Clusters to show in the table (default 40)'),
886
+ },
887
+ }, async ({ siteUrl, keywords, limit }) => {
888
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
889
+ try {
890
+ const fresh = gscFreshness(db.db);
891
+ const r = clusterKeywordList(db.db, keywords, { maxDate: fresh.effectiveMax });
892
+ const show = r.clusters.slice(0, limit ?? 40);
893
+ const counts = { own: 0, weak: 0, absent: 0 };
894
+ for (const c of r.clusters)
895
+ counts[c.verdict]++;
896
+ const rowsMd = show.map(c => `| ${c.head.replace(/\|/g, '\\|')} | ${c.verdict.toUpperCase()} | ${c.bestPosition ?? '-'} | ${fmtNum(c.impressions)} | ${c.keywords.length} | ${c.url ? c.url.replace(/\|/g, '\\|') : '-'} |`);
897
+ const md = `**Keyword list clustered for ${siteUrl}** - ${r.totalKeywords} keywords (${r.derived ? 'derived from your GSC demand' : 'your list'}, ${r.inGsc} with GSC data) → ${r.clusters.length} clusters: ${counts.own} own · ${counts.weak} weak · ${counts.absent} absent\n\n` +
898
+ `| Cluster (head term) | Verdict | Best pos | Impressions 90d | Keywords | Ranking URL |\n|---|---|---|---|---|---|\n` +
899
+ rowsMd.join('\n') +
900
+ `\n\nVerdicts are deterministic from your GSC positions. ABSENT clusters are content opportunities - chain into keyword_volume (market volume) or draft_content (a brief).`;
901
+ return {
902
+ content: [{ type: 'text', text: md }],
903
+ structuredContent: { ...r, clusters: r.clusters.slice(0, 200) },
904
+ };
905
+ }
906
+ finally {
907
+ db.close();
908
+ }
909
+ });
789
910
  server.registerTool('search_intent', {
790
911
  title: 'Search intent classification (DataForSEO Labs)',
791
- description: 'Classify the search intent (informational / navigational / commercial / transactional) of keywords — primary label + probability + secondary intents. Use it to spot intent mismatch: e.g. a transactional product page ranking for an informational query (a common cause of high impressions / low CTR). Pass siteUrl to persist the intents so run_audit can surface intent-vs-pagetype-mismatch. Labs call, cached 20 days, up to 1000 keywords. Language-only (no location).',
912
+ description: '[Paid: Labs, cheap, cached 20d, up to 1000 kw/call | Use for: intent mapping before content planning] Classify the search intent (informational / navigational / commercial / transactional) of keywords — primary label + probability + secondary intents. Use it to spot intent mismatch: e.g. a transactional product page ranking for an informational query (a common cause of high impressions / low CTR). Pass siteUrl to persist the intents so run_audit can surface intent-vs-pagetype-mismatch. Labs call, cached 20 days, up to 1000 keywords. Language-only (no location).',
792
913
  inputSchema: { keywords: z.array(z.string()).min(1).max(1000), languageCode: z.string().optional(), siteUrl: z.string().optional() },
793
914
  }, async ({ keywords, languageCode, siteUrl }) => {
794
915
  const client = requireDfs(dfs);
@@ -822,7 +943,7 @@ export function createServer() {
822
943
  });
823
944
  server.registerTool('page_lighthouse', {
824
945
  title: 'Page Lighthouse — lab Core Web Vitals (DataForSEO On-Page)',
825
- description: 'Run a live Lighthouse audit for ONE url: lab Core Web Vitals (LCP, CLS, TBT, FCP, Speed Index), category scores (performance/SEO/best-practices/accessibility), and the top time-saving opportunities. Complements GSC/CrUX field data (aggregate + delayed) with on-demand lab data. Pass siteUrl to persist the CWV so run_audit can surface high-yield-cwv-fail. Paid On-Page call (~2000 credits), slow (~20–120s), cached 20 days.',
946
+ description: '[Paid: On-Page, ONE URL per call | Use for: CWV evidence on a page that earns clicks] Run a live Lighthouse audit for ONE url: lab Core Web Vitals (LCP, CLS, TBT, FCP, Speed Index), category scores (performance/SEO/best-practices/accessibility), and the top time-saving opportunities. Complements GSC/CrUX field data (aggregate + delayed) with on-demand lab data. Pass siteUrl to persist the CWV so run_audit can surface high-yield-cwv-fail. Paid On-Page call (~2000 credits), slow (~20–120s), cached 20 days.',
826
947
  inputSchema: { url: z.string().url(), forMobile: z.boolean().optional(), siteUrl: z.string().optional() },
827
948
  }, async ({ url, forMobile, siteUrl }) => {
828
949
  const client = requireDfs(dfs);
@@ -876,7 +997,7 @@ export function createServer() {
876
997
  });
877
998
  server.registerTool('competitors_domain', {
878
999
  title: 'Competitor domains (DataForSEO Labs)',
879
- description: 'Discover the domains competing with a target for the same organic keywords (ranked by keyword overlap), with intersection counts and organic traffic estimates. The seed list for content-gap analysis (feed these into page_intersection). Labs call, cached 20 days. Pass location + language for the right market.',
1000
+ description: '[Paid: Labs, cheap, cached 20d | Use for: naming the real competitor set - the opening move of competitive work] Discover the domains competing with a target for the same organic keywords (ranked by keyword overlap), with intersection counts and organic traffic estimates. The seed list for content-gap analysis (feed these into page_intersection). Labs call, cached 20 days. Pass location + language for the right market.',
880
1001
  inputSchema: { target: z.string(), location: z.union([z.string(), z.number()]).optional(), languageCode: z.string().optional(), limit: z.number().int().min(1).max(100).optional() },
881
1002
  }, async ({ target, location, languageCode, limit }) => {
882
1003
  const client = requireDfs(dfs);
@@ -898,7 +1019,7 @@ export function createServer() {
898
1019
  });
899
1020
  server.registerTool('page_intersection', {
900
1021
  title: 'Content gap — page intersection (DataForSEO Labs)',
901
- description: 'Find keywords that competitor pages rank for but your page does NOT (the content gap). Pass competitor URLs as `competitorUrls` (max 20; wildcards like https://site.com/blog/* allowed) and your own URL(s) in `excludePages` (max 10) to subtract. Returns gap keywords sorted by search volume, with each competitor’s rank. Labs call, cached 20 days. Pass location + language.',
1022
+ description: '[Paid: Labs, cheap, cached 20d | Use for: page-level content gap vs named rival URLs] Find keywords that competitor pages rank for but your page does NOT (the content gap). Pass competitor URLs as `competitorUrls` (max 20; wildcards like https://site.com/blog/* allowed) and your own URL(s) in `excludePages` (max 10) to subtract. Returns gap keywords sorted by search volume, with each competitor’s rank. Labs call, cached 20 days. Pass location + language.',
902
1023
  inputSchema: {
903
1024
  competitorUrls: z.array(z.string()).min(1).max(20),
904
1025
  excludePages: z.array(z.string()).max(10).optional(),
@@ -950,7 +1071,7 @@ export function createServer() {
950
1071
  };
951
1072
  server.registerTool('domain_visibility', {
952
1073
  title: 'Domain visibility over time (DataForSEO Labs)',
953
- description: 'Monthly organic visibility for ANY domain or subdomain — no Search Console access needed: ranking-keyword totals, position distribution (1–3 / 4–10 / 11–20 / 21–100) and estimated traffic value (ETV) per month, plus a trend verdict. The Semrush-style "organic overview / visibility over time" for you or a competitor. Labs call, cached 20 days. Pass location (name or code) for the right market. Grain: month × domain. Joins: domain → Labs/backlinks tools; period → rank_history.',
1074
+ description: '[Paid: Labs, cheap, cached 20d | Use for: visibility-over-time trend, any domain, no GSC needed] Monthly organic visibility for ANY domain or subdomain — no Search Console access needed: ranking-keyword totals, position distribution (1–3 / 4–10 / 11–20 / 21–100) and estimated traffic value (ETV) per month, plus a trend verdict. The Semrush-style "organic overview / visibility over time" for you or a competitor. Labs call, cached 20 days. Pass location (name or code) for the right market. Grain: month × domain. Joins: domain → Labs/backlinks tools; period → rank_history.',
954
1075
  inputSchema: {
955
1076
  target: z.string().describe('Domain or subdomain, no scheme (e.g. example.com or blog.example.com)'),
956
1077
  location: z.union([z.string(), z.number()]).optional(),
@@ -1006,7 +1127,7 @@ export function createServer() {
1006
1127
  });
1007
1128
  server.registerTool('top_pages', {
1008
1129
  title: 'Top ranking pages on a domain (DataForSEO Labs)',
1009
- description: 'The top organic pages of ANY domain or subdomain, ranked by estimated traffic value (ETV): page, ranking-keyword count, ETV and top-3 / top-10 keyword counts. The Semrush-style "top pages" view — works on competitors, no crawl or GSC needed. Labs call, cached 20 days. Pass location for the right market. Grain: page × domain. Joins: domain → Labs tools; URL → pages.url_key on your own property.',
1130
+ description: '[Paid: Labs, cheap, cached 20d | Use for: where rival traffic actually lands] The top organic pages of ANY domain or subdomain, ranked by estimated traffic value (ETV): page, ranking-keyword count, ETV and top-3 / top-10 keyword counts. The Semrush-style "top pages" view — works on competitors, no crawl or GSC needed. Labs call, cached 20 days. Pass location for the right market. Grain: page × domain. Joins: domain → Labs tools; URL → pages.url_key on your own property.',
1010
1131
  inputSchema: {
1011
1132
  target: z.string().describe('Domain or subdomain, no scheme'),
1012
1133
  location: z.union([z.string(), z.number()]).optional(),
@@ -1053,7 +1174,7 @@ export function createServer() {
1053
1174
  });
1054
1175
  server.registerTool('ranked_keywords', {
1055
1176
  title: 'Ranked keywords for a domain / URL / folder (DataForSEO Labs)',
1056
- description: 'Every keyword a target ranks for in Google organic — scope it to a whole domain, a subdomain, ONE page (scope:url with the full URL), or a subfolder (scope:folder + folder:"/blog/"). Returns keyword, position, search volume, ETV, the ranking URL, plus keyword difficulty, search intent and SERP features where the response carries them. Set aioOnly:true to list only keywords whose SERP shows a Google AI Overview (your AIO exposure list; per-keyword CITATION checking uses the SERP tools — see the cookbook). The Semrush-style "keywords a page or site ranks for" view — works on any site. Labs call, cached 20 days. Pass location for the right market. Grain: keyword × target. Joins: keyword → GSC search_analytics.query; URL → pages.url_key.',
1177
+ description: '[Paid: Labs, cheap, cached 20d - ONE call replaces any keyword loop | Use for: full keyword footprint of a domain/page/folder] Every keyword a target ranks for in Google organic — scope it to a whole domain, a subdomain, ONE page (scope:url with the full URL), or a subfolder (scope:folder + folder:"/blog/"). Returns keyword, position, search volume, ETV, the ranking URL, plus keyword difficulty, search intent and SERP features where the response carries them. Set aioOnly:true to list only keywords whose SERP shows a Google AI Overview (your AIO exposure list; per-keyword CITATION checking uses the SERP tools — see the cookbook). The Semrush-style "keywords a page or site ranks for" view — works on any site. Labs call, cached 20 days. Pass location for the right market. Grain: keyword × target. Joins: keyword → GSC search_analytics.query; URL → pages.url_key.',
1057
1178
  inputSchema: {
1058
1179
  target: z.string().describe('Domain, subdomain, or full URL (full URL required for scope:url)'),
1059
1180
  scope: z.enum(['domain', 'subdomain', 'url', 'folder']).optional(),
@@ -1149,9 +1270,112 @@ export function createServer() {
1149
1270
  structuredContent: { target: dfsTarget, scope: mode, aioOnly: aioOnly ?? false, totalCount, rowsTotal: kws.length, keywords: kws.slice(0, 100), cached: r.cached, cost: r.cost },
1150
1271
  };
1151
1272
  });
1273
+ // serp_features — the SERP-feature footprint: "how much of my market do AI Overviews,
1274
+ // snippets and other features sit on, and do I already rank page 1 there?" One cached
1275
+ // Labs pull, volume-weighted; persisted so the dashboard can chart it.
1276
+ server.registerTool('serp_features', {
1277
+ title: 'SERP-feature footprint (AI Overviews, snippets, PAA)',
1278
+ description: '[Paid: Labs, ONE cached call | Use for: AI-Overview / zero-click exposure with real numbers] How much of your keyword universe carries each SERP feature - AI Overviews, featured snippets, People Also Ask, shopping, video - weighted by search volume, and how much of that volume you already rank page 1 for. ONE DataForSEO Labs ranked_keywords pull (top-volume sample, cached 20 days, never per-keyword SERP loops). Answers "how exposed are we to AI Overviews / zero-click?" with real numbers. Deterministic: feature PRESENCE (from the Labs index). Judgement proxy: page-1 rank stands in for feature ownership - true ownership needs per-keyword SERP calls (see ranked_keywords aioOnly + the cookbook). Persists to the property DB so the dashboard charts it.',
1279
+ inputSchema: {
1280
+ siteUrl: z.string().describe('The GSC property - results persist to its database'),
1281
+ target: z.string().optional().describe('Override the analysed domain (defaults to the property host)'),
1282
+ location: z.union([z.string(), z.number()]).optional(),
1283
+ languageCode: z.string().optional(),
1284
+ limit: z.number().int().min(100).max(1000).optional().describe('Keyword sample size (default 1000, ordered by volume)'),
1285
+ },
1286
+ }, async ({ siteUrl, target, location, languageCode, limit }) => {
1287
+ const client = requireDfs(dfs);
1288
+ const domain = dfsHost(target ?? siteUrl);
1289
+ const r = await client.rankedKeywords(domain, location, languageCode ?? 'en', limit ?? 1000, 'keyword_data.keyword_info.search_volume,desc');
1290
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1291
+ const sample = items.map((it) => {
1292
+ const kd = it.keyword_data ?? {};
1293
+ const serp = it.ranked_serp_element?.serp_item ?? {};
1294
+ return {
1295
+ keyword: kd.keyword,
1296
+ position: Number(serp.rank_absolute) || null,
1297
+ searchVolume: kd.keyword_info?.search_volume ?? null,
1298
+ serpFeatures: Array.isArray(kd.serp_info?.serp_item_types) ? kd.serp_info.serp_item_types : null,
1299
+ };
1300
+ }).filter(s => s.keyword);
1301
+ const fp = computeSerpFootprint(domain, typeof location === 'string' ? location : location != null ? String(location) : null, sample);
1302
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1303
+ try {
1304
+ persistSerpFootprint(db.db, fp);
1305
+ }
1306
+ finally {
1307
+ db.close();
1308
+ }
1309
+ const lines = fp.features.map(f => `| ${f.label} | ${f.volumeSharePct}% | ${fmtNum(f.volume)} | ${f.keywords} | ${f.page1SharePct}% |`);
1310
+ const md = `**SERP-feature footprint for ${domain}** - sample: top ${fp.sampleKeywords} keywords by volume (${fmtNum(fp.sampleVolume)} searches/mo)\n\n` +
1311
+ `| Feature | % of sample volume | Volume on feature SERPs | Keywords | You rank page 1 (share of that volume) |\n|---|---|---|---|---|\n` +
1312
+ lines.join('\n') +
1313
+ `\n\nPresence is deterministic (Labs index); "page 1" is an ownership PROXY - per-keyword citation/ownership checks are the SERP tools. ${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}).`} Persisted for the dashboard.` +
1314
+ browserLink(siteUrl);
1315
+ return {
1316
+ content: [{ type: 'text', text: md }],
1317
+ structuredContent: fp,
1318
+ };
1319
+ });
1320
+ // market_sizing — Market Sizing and Prioritisation: the organic market read that opens
1321
+ // an engagement. Top-down Labs pulls only (client + <=4 rivals), cached, ~$0.65 worst case.
1322
+ server.registerTool('market_sizing', {
1323
+ title: 'Market Sizing and Prioritisation (share of voice vs competitors)',
1324
+ description: '[Paid: Labs, one cached call per domain (<=5) | Use for: sizing the organic market and who owns it] Build the organic market map: your domain plus up to 4 named competitors, ONE cached Labs ranked_keywords pull each, unioned into a keyword universe. Returns total monthly demand (deduplicated search volume), each domain\'s share of voice (ETV share) overall and per topic cluster, and the leader per cluster - the "here is the market, here is who owns it, here is where to attack" table that opens an engagement. Deterministic: the competitor set and ranked keywords (Labs index). Judgement: ETV is DataForSEO\'s CTR-curve traffic estimate - the SoV percentages inherit that. Persists to the property DB for the dashboard chart. Get the competitor set from competitors_domain first if unsure.',
1325
+ inputSchema: {
1326
+ siteUrl: z.string(),
1327
+ competitors: z.array(z.string()).min(1).max(4).describe('Competitor domains, e.g. ["rival.com", "other.co.uk"]'),
1328
+ location: z.union([z.string(), z.number()]).optional(),
1329
+ languageCode: z.string().optional(),
1330
+ limitPerDomain: z.number().int().min(100).max(1000).optional().describe('Keywords sampled per domain (default 1000, by volume)'),
1331
+ },
1332
+ }, async ({ siteUrl, competitors, location, languageCode, limitPerDomain }) => {
1333
+ const client = requireDfs(dfs);
1334
+ const own = dfsHost(siteUrl);
1335
+ const domains = [own, ...competitors.map(c => dfsHost(c)).filter(c => c && c !== own)];
1336
+ const inputs = [];
1337
+ let cost = 0;
1338
+ let cachedAll = true;
1339
+ for (const domain of domains) { // sequential - the client serialises anyway
1340
+ const r = await client.rankedKeywords(domain, location, languageCode ?? 'en', limitPerDomain ?? 1000, 'keyword_data.keyword_info.search_volume,desc');
1341
+ cost += r.cost;
1342
+ cachedAll = cachedAll && r.cached;
1343
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1344
+ inputs.push({
1345
+ domain,
1346
+ keywords: items.map((it) => ({
1347
+ keyword: it.keyword_data?.keyword,
1348
+ volume: it.keyword_data?.keyword_info?.search_volume ?? null,
1349
+ position: Number(it.ranked_serp_element?.serp_item?.rank_absolute) || null,
1350
+ etv: it.ranked_serp_element?.serp_item?.etv != null ? Number(it.ranked_serp_element.serp_item.etv) : null,
1351
+ })).filter((k) => k.keyword),
1352
+ });
1353
+ }
1354
+ const m = computeMarketSizing(inputs, typeof location === 'string' ? location : location != null ? String(location) : null);
1355
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1356
+ try {
1357
+ persistMarketSizing(db.db, m);
1358
+ }
1359
+ finally {
1360
+ db.close();
1361
+ }
1362
+ const sovLine = m.domains.map(d => `${d === own ? '**' + d + '**' : d} ${m.sovByDomain[d]}%`).join(' · ');
1363
+ const clusterRows = m.clusters.map(c => `| ${c.head.replace(/\|/g, '\\|')} | ${fmtNum(c.volume)} | ${c.keywords} | ${c.leader === own ? '**you**' : (c.leader ?? '-')} | ${m.domains.map(d => `${c.sov[d]}%`).join(' / ')} |`);
1364
+ const md = `**Market Sizing and Prioritisation** - ${m.domains.join(' vs ')}\n\n` +
1365
+ `Universe: ${fmtNum(m.universeKeywords)} unique keywords, **${fmtNum(m.universeVolume)} searches/mo** total demand.\n` +
1366
+ `Share of voice (ETV share): ${sovLine}\n\n` +
1367
+ `| Topic cluster | Volume/mo | Keywords | Leader | SoV (${m.domains.join(' / ')}) |\n|---|---|---|---|---|\n` +
1368
+ clusterRows.join('\n') +
1369
+ `\n\nD: competitor set + ranked keywords (Labs index). N: ETV is a CTR-curve estimate - SoV inherits it. ${cachedAll ? 'All cached.' : `Live cost $${cost.toFixed(4)}.`} Persisted for the dashboard.` +
1370
+ browserLink(siteUrl);
1371
+ return {
1372
+ content: [{ type: 'text', text: md }],
1373
+ structuredContent: m,
1374
+ };
1375
+ });
1152
1376
  server.registerTool('topic_gaps', {
1153
1377
  title: 'Topic gaps — what to cover to be expert in your space (Labs + GSC)',
1154
- description: 'What related topics should this site cover to be seen as expert in its space? Pulls competitor keyword footprints (DataForSEO Labs ranked_keywords, ≤4 bounded + 20-day-cached calls), subtracts everything YOU already surface for (every GSC query with impressions) or already have a page about (title/H1/slug near-match), clusters the surviving gap keywords lexically, and scores each topic by summed search volume × how many competitors rank there. Each topic names the owning competitor with an example URL, your nearest existing page to build from, and (when resolve_entities has run) whether it sits beside entities you already cover. Pass competitors explicitly (max 3) or let it derive the top 2 from ranking overlap. Needs synced GSC data + DataForSEO credentials.',
1378
+ description: '[Paid: Labs, <=4 cached calls | Use for: topic-level gap plan vs competitors] What related topics should this site cover to be seen as expert in its space? Pulls competitor keyword footprints (DataForSEO Labs ranked_keywords, ≤4 bounded + 20-day-cached calls), subtracts everything YOU already surface for (every GSC query with impressions) or already have a page about (title/H1/slug near-match), clusters the surviving gap keywords lexically, and scores each topic by summed search volume × how many competitors rank there. Each topic names the owning competitor with an example URL, your nearest existing page to build from, and (when resolve_entities has run) whether it sits beside entities you already cover. Pass competitors explicitly (max 3) or let it derive the top 2 from ranking overlap. Needs synced GSC data + DataForSEO credentials.',
1155
1379
  inputSchema: {
1156
1380
  siteUrl: z.string(),
1157
1381
  competitors: z.array(z.string()).max(3).optional().describe('Competitor domains (max 3, e.g. ["rival.com"]). Omitted → derived via competitors_domain'),
@@ -1284,7 +1508,7 @@ export function createServer() {
1284
1508
  }
1285
1509
  const c = data.summary?.current;
1286
1510
  const summary = `Dashboard opened for ${siteUrl} — ${c?.clicks ?? 0} clicks / ${c?.impressions ?? 0} impressions (last 28d)` +
1287
- `${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.`;
1511
+ `${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.` + browserLink(siteUrl);
1288
1512
  return { content: [{ type: 'text', text: summary }], structuredContent: { siteUrl } };
1289
1513
  });
1290
1514
  // App-only data tool: the dashboard widget calls this via app.callServerTool to fetch its
@@ -1314,21 +1538,73 @@ export function createServer() {
1314
1538
  const tpl = readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8');
1315
1539
  const json = JSON.stringify(data).replace(/</g, '\\u003c'); // prevent </script> breakout
1316
1540
  const inject = `<script>window.__DASH_FIXTURE__=${json};window.__DASH_THEME__=${JSON.stringify(theme ?? 'light')};</script>`;
1317
- const html = tpl.replace(/<head([^>]*)>/i, `<head$1>${inject}`);
1541
+ // Replacement FUNCTION, not string — crawl data containing $& / $' would otherwise
1542
+ // be interpreted as String.replace substitution patterns and corrupt the report.
1543
+ const html = tpl.replace(/<head([^>]*)>/i, (_m, attrs) => `<head${attrs}>${inject}`);
1318
1544
  const dir = path.join(dataDir(), 'reports');
1319
1545
  mkdirSync(dir, { recursive: true });
1320
1546
  const file = path.join(dir, `${sanitizeProperty(siteUrl)}-dashboard.html`);
1321
1547
  writeFileSync(file, html);
1322
1548
  return {
1323
- content: [{ type: 'text', text: `Report saved: ${file}\nOpen it in any browser for the full interactive dashboard (${data.findings?.total ?? 0} findings). Shareable — send it to a client as-is.` }],
1549
+ content: [{ type: 'text', text: `Report saved: ${file}\nOpen it in any browser for the full interactive dashboard (${data.findings?.total ?? 0} findings). Shareable — send it to a client as-is.` + browserLink(siteUrl) }],
1324
1550
  structuredContent: { path: file, siteUrl, findings: data.findings?.total ?? 0, bytes: html.length },
1325
1551
  };
1326
1552
  });
1553
+ // serve_dashboard — the local webserver delivery surface: the full dashboard in a real
1554
+ // browser tab (live data, property switcher, native downloads), no MCP-App host needed.
1555
+ server.registerTool('serve_dashboard', {
1556
+ title: 'Serve the dashboard on a local webserver',
1557
+ description: 'Start a localhost-only webserver and return a URL that opens the full interactive dashboard in your browser — live data straight from the local database (always current, unlike export_report snapshots), a property switcher, working CSV downloads, and no host widget limits. The server stays up while the MCP server runs; call again with stop=true to shut it down. Localhost only — nothing is exposed to the network.',
1558
+ inputSchema: { siteUrl: z.string().optional(), port: z.number().int().min(1024).max(65535).optional(), stop: z.boolean().optional() },
1559
+ }, async ({ siteUrl, port, stop }) => {
1560
+ if (stop) {
1561
+ const was = dashboardServerUrl();
1562
+ const stopped = await stopDashboardServer();
1563
+ return {
1564
+ content: [{ type: 'text', text: stopped ? `Dashboard server stopped (was ${was}).` : 'No dashboard server running.' }],
1565
+ structuredContent: { stopped },
1566
+ };
1567
+ }
1568
+ const { url } = await startDashboardServer({
1569
+ dataDir,
1570
+ uiHtml: () => readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8'),
1571
+ call: {
1572
+ get_dashboard_data: async (a) => {
1573
+ const want = String(a.siteUrl ?? '');
1574
+ // Only serve properties that actually exist locally — getDashboardData would
1575
+ // otherwise CREATE an empty DB file for any bogus siteUrl posted at the API.
1576
+ if (!listLocalProperties(dataDir()).some(p => p.siteUrl === want))
1577
+ throw new Error(`unknown property ${want}`);
1578
+ return getDashboardData(dataDir(), want);
1579
+ },
1580
+ related_terms: async (a) => {
1581
+ const r = await requireDfs(dfs).relatedTerms(String(a.keyword ?? ''), a.location, a.languageCode);
1582
+ return r;
1583
+ },
1584
+ keyword_volume: async (a) => {
1585
+ const r = await requireDfs(dfs).searchVolume(a.keywords ?? [], a.location, a.languageCode);
1586
+ const items = (r.tasks[0]?.result ?? []).map((k) => ({
1587
+ keyword: k.keyword, searchVolume: k.search_volume, cpc: k.cpc, competition: k.competition,
1588
+ }));
1589
+ return { keywords: items, cached: r.cached, cost: r.cost };
1590
+ },
1591
+ },
1592
+ ...(port != null ? { port } : {}),
1593
+ });
1594
+ const open = siteUrl ? `${url}/dashboard?siteUrl=${encodeURIComponent(siteUrl)}` : `${url}/`;
1595
+ const portNote = port != null && !url.endsWith(`:${port}`)
1596
+ ? ` (already running on its original port — requested port ${port} ignored; stop=true first to move it)`
1597
+ : '';
1598
+ return {
1599
+ content: [{ type: 'text', text: `Dashboard server running — open ${open} in your browser${portNote}. Live data, all charts, CSV downloads, property switcher. Stays up while this MCP server runs; serve_dashboard stop=true to stop it.` }],
1600
+ structuredContent: { url: open, base: url },
1601
+ };
1602
+ });
1327
1603
  // pull_backlinks — on-demand backlink profile (DataForSEO, paid + 20-day cached). Powers
1328
1604
  // backlinks-to-404 (the big quick win), top-linked pages, and true-orphan detection.
1329
1605
  server.registerTool('pull_backlinks', {
1330
1606
  title: 'Pull backlink profile (DataForSEO)',
1331
- description: 'Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
1607
+ description: '[Paid: Backlinks subscription (separate - 40204 = not activated), cached 20d | Use for: authority + dead-backlink recovery] Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
1332
1608
  inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(1000).optional(), statusLimit: z.number().int().min(0).max(1000).optional() },
1333
1609
  }, async ({ siteUrl, limit, statusLimit }) => {
1334
1610
  const bl = requireDfs(backlinks);