@houtini/seo-audit-console 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. package/README.md +16 -14
  2. package/dist/audit/checks.d.ts +0 -7
  3. package/dist/audit/checks.d.ts.map +1 -1
  4. package/dist/audit/checks.js +22 -282
  5. package/dist/audit/checks.js.map +1 -1
  6. package/dist/audit/drift.d.ts +0 -12
  7. package/dist/audit/drift.d.ts.map +1 -1
  8. package/dist/audit/drift.js +0 -15
  9. package/dist/audit/drift.js.map +1 -1
  10. package/dist/audit/engine.d.ts +0 -3
  11. package/dist/audit/engine.d.ts.map +1 -1
  12. package/dist/audit/engine.js +7 -42
  13. package/dist/audit/engine.js.map +1 -1
  14. package/dist/audit/keywordList.d.ts +0 -14
  15. package/dist/audit/keywordList.d.ts.map +1 -1
  16. package/dist/audit/keywordList.js +0 -5
  17. package/dist/audit/keywordList.js.map +1 -1
  18. package/dist/audit/opportunities.js +3 -26
  19. package/dist/audit/opportunities.js.map +1 -1
  20. package/dist/audit/recon.d.ts +0 -31
  21. package/dist/audit/recon.d.ts.map +1 -1
  22. package/dist/audit/recon.js +0 -27
  23. package/dist/audit/recon.js.map +1 -1
  24. package/dist/audit/report.d.ts +0 -4
  25. package/dist/audit/report.d.ts.map +1 -1
  26. package/dist/audit/report.js +0 -7
  27. package/dist/audit/report.js.map +1 -1
  28. package/dist/audit/schema-validate.d.ts +0 -26
  29. package/dist/audit/schema-validate.d.ts.map +1 -1
  30. package/dist/audit/schema-validate.js +8 -64
  31. package/dist/audit/schema-validate.js.map +1 -1
  32. package/dist/audit/templates.d.ts +0 -24
  33. package/dist/audit/templates.d.ts.map +1 -1
  34. package/dist/audit/templates.js +1 -21
  35. package/dist/audit/templates.js.map +1 -1
  36. package/dist/audit/topicGaps.d.ts +0 -1
  37. package/dist/audit/topicGaps.d.ts.map +1 -1
  38. package/dist/audit/topicGaps.js +3 -24
  39. package/dist/audit/topicGaps.js.map +1 -1
  40. package/dist/core/AuditDatabase.d.ts +0 -14
  41. package/dist/core/AuditDatabase.d.ts.map +1 -1
  42. package/dist/core/AuditDatabase.js +8 -61
  43. package/dist/core/AuditDatabase.js.map +1 -1
  44. package/dist/core/Backlinks.d.ts +0 -7
  45. package/dist/core/Backlinks.d.ts.map +1 -1
  46. package/dist/core/Backlinks.js +1 -23
  47. package/dist/core/Backlinks.js.map +1 -1
  48. package/dist/core/Crawler.d.ts +0 -1
  49. package/dist/core/Crawler.d.ts.map +1 -1
  50. package/dist/core/Crawler.js +24 -119
  51. package/dist/core/Crawler.js.map +1 -1
  52. package/dist/core/DataForSeoClient.d.ts +25 -74
  53. package/dist/core/DataForSeoClient.d.ts.map +1 -1
  54. package/dist/core/DataForSeoClient.js +65 -90
  55. package/dist/core/DataForSeoClient.js.map +1 -1
  56. package/dist/core/Entities.d.ts +0 -6
  57. package/dist/core/Entities.d.ts.map +1 -1
  58. package/dist/core/Entities.js +0 -8
  59. package/dist/core/Entities.js.map +1 -1
  60. package/dist/core/FirecrawlClient.d.ts +0 -18
  61. package/dist/core/FirecrawlClient.d.ts.map +1 -1
  62. package/dist/core/FirecrawlClient.js +1 -13
  63. package/dist/core/FirecrawlClient.js.map +1 -1
  64. package/dist/core/GscClient.d.ts +0 -7
  65. package/dist/core/GscClient.d.ts.map +1 -1
  66. package/dist/core/GscClient.js +1 -10
  67. package/dist/core/GscClient.js.map +1 -1
  68. package/dist/core/GscSync.d.ts +0 -4
  69. package/dist/core/GscSync.d.ts.map +1 -1
  70. package/dist/core/GscSync.js +2 -33
  71. package/dist/core/GscSync.js.map +1 -1
  72. package/dist/core/JobManager.d.ts +0 -5
  73. package/dist/core/JobManager.d.ts.map +1 -1
  74. package/dist/core/JobManager.js +0 -8
  75. package/dist/core/JobManager.js.map +1 -1
  76. package/dist/core/LinkIntersect.d.ts +0 -1
  77. package/dist/core/LinkIntersect.d.ts.map +1 -1
  78. package/dist/core/LinkIntersect.js +1 -37
  79. package/dist/core/LinkIntersect.js.map +1 -1
  80. package/dist/core/MajesticClient.d.ts +0 -26
  81. package/dist/core/MajesticClient.d.ts.map +1 -1
  82. package/dist/core/MajesticClient.js +0 -15
  83. package/dist/core/MajesticClient.js.map +1 -1
  84. package/dist/core/RankTracker.d.ts +0 -4
  85. package/dist/core/RankTracker.d.ts.map +1 -1
  86. package/dist/core/RankTracker.js +0 -9
  87. package/dist/core/RankTracker.js.map +1 -1
  88. package/dist/core/Refresh.d.ts +0 -6
  89. package/dist/core/Refresh.d.ts.map +1 -1
  90. package/dist/core/Refresh.js +0 -6
  91. package/dist/core/Refresh.js.map +1 -1
  92. package/dist/core/SupadataClient.d.ts +0 -11
  93. package/dist/core/SupadataClient.d.ts.map +1 -1
  94. package/dist/core/SupadataClient.js +0 -5
  95. package/dist/core/SupadataClient.js.map +1 -1
  96. package/dist/core/UrlInspector.d.ts +0 -5
  97. package/dist/core/UrlInspector.d.ts.map +1 -1
  98. package/dist/core/UrlInspector.js +0 -6
  99. package/dist/core/UrlInspector.js.map +1 -1
  100. package/dist/core/WikidataClient.d.ts +0 -2
  101. package/dist/core/WikidataClient.d.ts.map +1 -1
  102. package/dist/core/WikidataClient.js +0 -7
  103. package/dist/core/WikidataClient.js.map +1 -1
  104. package/dist/core/agentReadiness.d.ts +0 -9
  105. package/dist/core/agentReadiness.d.ts.map +1 -1
  106. package/dist/core/agentReadiness.js +0 -13
  107. package/dist/core/agentReadiness.js.map +1 -1
  108. package/dist/core/ctrModel.js +0 -3
  109. package/dist/core/ctrModel.js.map +1 -1
  110. package/dist/core/dashboardData.d.ts +0 -1
  111. package/dist/core/dashboardData.d.ts.map +1 -1
  112. package/dist/core/dashboardData.js +13 -95
  113. package/dist/core/dashboardData.js.map +1 -1
  114. package/dist/core/dataStorage.d.ts +0 -2
  115. package/dist/core/dataStorage.d.ts.map +1 -1
  116. package/dist/core/dataStorage.js +6 -19
  117. package/dist/core/dataStorage.js.map +1 -1
  118. package/dist/core/draftBrief.d.ts +0 -7
  119. package/dist/core/draftBrief.d.ts.map +1 -1
  120. package/dist/core/draftBrief.js +0 -10
  121. package/dist/core/draftBrief.js.map +1 -1
  122. package/dist/core/extract.d.ts +0 -11
  123. package/dist/core/extract.d.ts.map +1 -1
  124. package/dist/core/extract.js +1 -38
  125. package/dist/core/extract.js.map +1 -1
  126. package/dist/core/googleNews.d.ts +0 -11
  127. package/dist/core/googleNews.d.ts.map +1 -1
  128. package/dist/core/googleNews.js +0 -7
  129. package/dist/core/googleNews.js.map +1 -1
  130. package/dist/core/gscFreshness.d.ts +0 -10
  131. package/dist/core/gscFreshness.d.ts.map +1 -1
  132. package/dist/core/gscFreshness.js +0 -11
  133. package/dist/core/gscFreshness.js.map +1 -1
  134. package/dist/core/linkGraph.d.ts +0 -15
  135. package/dist/core/linkGraph.d.ts.map +1 -1
  136. package/dist/core/linkGraph.js +1 -23
  137. package/dist/core/linkGraph.js.map +1 -1
  138. package/dist/core/marketSizing.d.ts +0 -13
  139. package/dist/core/marketSizing.d.ts.map +1 -1
  140. package/dist/core/marketSizing.js +1 -1
  141. package/dist/core/marketSizing.js.map +1 -1
  142. package/dist/core/passageScore.d.ts +0 -13
  143. package/dist/core/passageScore.d.ts.map +1 -1
  144. package/dist/core/passageScore.js +1 -16
  145. package/dist/core/passageScore.js.map +1 -1
  146. package/dist/core/paths.d.ts +0 -3
  147. package/dist/core/paths.d.ts.map +1 -1
  148. package/dist/core/paths.js +0 -0
  149. package/dist/core/paths.js.map +1 -1
  150. package/dist/core/queryData.d.ts +0 -1
  151. package/dist/core/queryData.d.ts.map +1 -1
  152. package/dist/core/queryData.js +0 -18
  153. package/dist/core/queryData.js.map +1 -1
  154. package/dist/core/reconFetch.d.ts +0 -6
  155. package/dist/core/reconFetch.d.ts.map +1 -1
  156. package/dist/core/reconFetch.js +1 -4
  157. package/dist/core/reconFetch.js.map +1 -1
  158. package/dist/core/reconResearch.d.ts +0 -20
  159. package/dist/core/reconResearch.d.ts.map +1 -1
  160. package/dist/core/reconResearch.js +1 -16
  161. package/dist/core/reconResearch.js.map +1 -1
  162. package/dist/core/reranker.d.ts +0 -2
  163. package/dist/core/reranker.d.ts.map +1 -1
  164. package/dist/core/reranker.js +1 -11
  165. package/dist/core/reranker.js.map +1 -1
  166. package/dist/core/robots.d.ts +0 -6
  167. package/dist/core/robots.d.ts.map +1 -1
  168. package/dist/core/robots.js +1 -7
  169. package/dist/core/robots.js.map +1 -1
  170. package/dist/core/serpFootprint.d.ts +0 -12
  171. package/dist/core/serpFootprint.d.ts.map +1 -1
  172. package/dist/core/serpFootprint.js +0 -1
  173. package/dist/core/serpFootprint.js.map +1 -1
  174. package/dist/core/serpRecon.d.ts +0 -14
  175. package/dist/core/serpRecon.d.ts.map +1 -1
  176. package/dist/core/serpRecon.js +1 -13
  177. package/dist/core/serpRecon.js.map +1 -1
  178. package/dist/core/sitemap.js +3 -16
  179. package/dist/core/sitemap.js.map +1 -1
  180. package/dist/core/sql.d.ts +0 -4
  181. package/dist/core/sql.d.ts.map +1 -1
  182. package/dist/core/sql.js +0 -4
  183. package/dist/core/sql.js.map +1 -1
  184. package/dist/core/url-key.d.ts +0 -31
  185. package/dist/core/url-key.d.ts.map +1 -1
  186. package/dist/core/url-key.js +1 -38
  187. package/dist/core/url-key.js.map +1 -1
  188. package/dist/core/webHandlers.d.ts +0 -6
  189. package/dist/core/webHandlers.d.ts.map +1 -1
  190. package/dist/core/webHandlers.js +2 -5
  191. package/dist/core/webHandlers.js.map +1 -1
  192. package/dist/core/webServer.d.ts +0 -16
  193. package/dist/core/webServer.d.ts.map +1 -1
  194. package/dist/core/webServer.js +3 -17
  195. package/dist/core/webServer.js.map +1 -1
  196. package/dist/dashboard.js +0 -23
  197. package/dist/dashboard.js.map +1 -1
  198. package/dist/generators/index.d.ts +0 -7
  199. package/dist/generators/index.d.ts.map +1 -1
  200. package/dist/generators/index.js +1 -26
  201. package/dist/generators/index.js.map +1 -1
  202. package/dist/index.js +0 -2
  203. package/dist/index.js.map +1 -1
  204. package/dist/server.d.ts.map +1 -1
  205. package/dist/server.js +190 -312
  206. package/dist/server.js.map +1 -1
  207. package/package.json +1 -1
  208. package/server.json +85 -85
package/dist/server.js CHANGED
@@ -18,15 +18,10 @@ import { fetchCompetitorContent, routeFor } from './core/reconResearch.js';
18
18
  import { fetchOwnPage } from './core/reconFetch.js';
19
19
  import { parseSerpForRecon, reconVerdict } from './core/serpRecon.js';
20
20
  import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos, pageState, crawlRealityOverride, clusterCannibalisation, opportunityBasis } from './audit/recon.js';
21
- /** A clickable browser-dashboard link appended to tool outputs — the user should always
22
- * know the full interactive report is one click away (or one serve_dashboard call away). */
23
21
  function browserLink(siteUrl) {
24
22
  const base = dashboardServerUrl();
25
23
  if (base)
26
24
  return `\n\n📊 Browser dashboard: ${base}/dashboard${siteUrl ? `?siteUrl=${encodeURIComponent(siteUrl)}` : ''}`;
27
- // No live server: point at the dashboard surfaces that always work. get_dashboard renders the
28
- // interactive dashboard in chat (works through the Docker gateway where a served port would not);
29
- // serve_dashboard opens a browser tab locally; export_report writes a shareable HTML file.
30
25
  return `\n\n📊 See it in the dashboard — run get_dashboard${siteUrl ? ` for ${siteUrl}` : ''} (interactive, in chat), serve_dashboard (browser tab), or export_report (shareable HTML).`;
31
26
  }
32
27
  import { runAudit, runSingleCheck, listChecks } from './audit/engine.js';
@@ -62,9 +57,6 @@ import { gscFreshness } from './core/gscFreshness.js';
62
57
  import { clusterKeywordList } from './audit/keywordList.js';
63
58
  const SERVER_NAME = 'seo-audit-console';
64
59
  const SERVER_VERSION = JSON.parse(readFileSync(new URL('../package.json', import.meta.url), 'utf8')).version;
65
- // Where per-property crawl/audit DBs live. Resolution order (computed once per run):
66
- // SAC_DATA_DIR env > persisted choice (~/.seo-audit-console.json) > Documents default.
67
- // The user can set the persisted choice through the `data_location` tool (no JSON editing).
68
60
  const CONFIG_PATH = path.join(homedir(), '.seo-audit-console.json');
69
61
  const DEFAULT_DATA_DIR = path.join(homedir(), 'Documents', 'seo-audit-console');
70
62
  function readConfigDataDir() {
@@ -82,111 +74,102 @@ export function dataDir() {
82
74
  return RESOLVED_DATA_DIR;
83
75
  }
84
76
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
85
- // Version the widget URIs: hosts may cache ui:// resources by URI indefinitely, so an
86
- // unversioned URI can pin users to a stale (or broken) cached bundle across releases.
87
77
  const DASHBOARD_URI = `ui://dashboard/main-${SERVER_VERSION}.html`;
88
78
  const SYNC_PROGRESS_URI = `ui://sync-progress/main-${SERVER_VERSION}.html`;
89
- // Must match the categories actually used by CHECKS (src/audit/checks.ts) so category
90
- // filters never silently return empty. (Was listing performance/agentic/integrity/war-stories
91
- // which no check uses, and omitting content/security which checks do use.)
92
79
  const CHECK_CATEGORIES = [
93
80
  'crawlability', 'indexation', 'onpage', 'content', 'schema', 'security', 'performance', 'merged',
94
81
  ];
95
82
  function isoDaysAgo(days) {
96
83
  return new Date(Date.now() - days * 86400000).toISOString().slice(0, 10);
97
84
  }
98
- // Server-level instructions (MCP `initialize` result): teach the assistant how the data
99
- // sources JOIN so it can COMPOSE bespoke multi-source analyses, not just run presets.
100
- const SERVER_INSTRUCTIONS = `SEO Audit Console fuses four data sources into one SQLite database per property: Google Search Console history, a first-party site crawl, GSC URL Inspection, and on-demand DataForSEO (SERP/Labs/Backlinks). Its real power is COMPOSITION — joining sources to answer questions no single tool answers.
101
-
102
- JOIN KEYS (memorise these):
103
- - url_key — one normalised URL form (https, unified www/apex, sorted params, no tracking params/fragments). Joins the crawl (pages, links) ↔ GSC (search_analytics.page_key) ↔ url_inspection ↔ page_backlinks ↔ page_cwv ↔ page_entity. normalize_url shows the key for any URL.
104
- - query — the literal search term. Joins GSC search_analytics ↔ DataForSEO keyword tools (keyword_volume, search_intent, ranked_keywords rows) ↔ keyword_intent.
105
- - domain — a bare host (no scheme, no www). Joins the Labs tools (ranked_keywords, domain_visibility, top_pages, competitors_domain, topic_gaps) ↔ backlinks summary.
106
-
107
- GRAIN (one line per source):
108
- - search_analytics: date × query × page (plus device/country when synced with segments). Rows are additive; position must be impression-weighted when aggregated.
109
- - pages / links: the LATEST crawl only, one row per url_key (history lives in page_snapshots, diffed by detect_changes). Carries title/H1/meta, canonical, schema (json_ld), body_chunks, iPR, click_depth, inlink_count.
110
- - url_inspection: one row per inspected URL — Google's own view (coverage_state, google_canonical vs user_canonical, last_crawl_time, crawled_as, rich_results).
111
- - rank_history: month × domain (DataForSEO rank distribution + ETV).
112
- - page_backlinks: one row per backlinked URL (counts + live HTTP status).
113
- - Labs tools: keyword × target, or month × target; cached 20 days; each call costs money — never loop them in bulk.
114
-
115
- ARCHETYPE CHAINS (compose along these lines):
116
- 1. Demand → reality: a GSC query with impressions but weak rank → the ranking page's crawl fields (title/H1/body_chunks) → does the page actually say what the query asks? → fix on-page or draft content.
117
- 2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
118
- 3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
119
-
120
- COST DISCIPLINE (behave like a strategist who knows the margins):
121
- - Free and instant, use liberally: everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
122
- - Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, topic_trend, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
123
- - Paid per-keyword (SERP): related_terms, youtube_discovery (ranking videos for a topic — pair with a transcript tool), news_discovery (recent coverage / freshness), and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
124
- - Content research (what to write / what changed): topic_trend (is a topic rising/seasonal) → youtube_discovery + news_discovery (what the winning videos/articles cover) → draft_content.
125
- - Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
126
-
127
- AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
128
- 1. Baseline: refresh_property → run_audit → serve_dashboard (share the URL).
129
- 2. Market: serp_features (feature exposure) + domain_visibility + competitors_domain → topic_gaps vs the named rivals.
130
- 3. Content plan: suggest_pages (demand you already have) + topic_gaps (demand rivals own) → draft_content briefs.
131
- 4. Fix cycle: fix_finding per top finding → re-crawl → detect_changes to prove the fix landed.
132
-
85
+ const SERVER_INSTRUCTIONS = `SEO Audit Console fuses four data sources into one SQLite database per property: Google Search Console history, a first-party site crawl, GSC URL Inspection, and on-demand DataForSEO (SERP/Labs/Backlinks). Its real power is COMPOSITION — joining sources to answer questions no single tool answers.
86
+
87
+ JOIN KEYS (memorise these):
88
+ - url_key — one normalised URL form (https, unified www/apex, sorted params, no tracking params/fragments). Joins the crawl (pages, links) ↔ GSC (search_analytics.page_key) ↔ url_inspection ↔ page_backlinks ↔ page_cwv ↔ page_entity. normalize_url shows the key for any URL.
89
+ - query — the literal search term. Joins GSC search_analytics ↔ DataForSEO keyword tools (keyword_volume, search_intent, ranked_keywords rows) ↔ keyword_intent.
90
+ - domain — a bare host (no scheme, no www). Joins the Labs tools (ranked_keywords, domain_visibility, top_pages, competitors_domain, topic_gaps) ↔ backlinks summary.
91
+
92
+ GRAIN (one line per source):
93
+ - search_analytics: date × query × page (plus device/country when synced with segments). Rows are additive; position must be impression-weighted when aggregated.
94
+ - pages / links: the LATEST crawl only, one row per url_key (history lives in page_snapshots, diffed by detect_changes). Carries title/H1/meta, canonical, schema (json_ld), body_chunks, iPR, click_depth, inlink_count.
95
+ - url_inspection: one row per inspected URL — Google's own view (coverage_state, google_canonical vs user_canonical, last_crawl_time, crawled_as, rich_results).
96
+ - rank_history: month × domain (DataForSEO rank distribution + ETV).
97
+ - page_backlinks: one row per backlinked URL (counts + live HTTP status).
98
+ - Labs tools: keyword × target, or month × target; cached 20 days; each call costs money — never loop them in bulk.
99
+
100
+ ARCHETYPE CHAINS (compose along these lines):
101
+ 1. Demand → reality: a GSC query with impressions but weak rank → the ranking page's crawl fields (title/H1/body_chunks) → does the page actually say what the query asks? → fix on-page or draft content.
102
+ 2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
103
+ 3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
104
+
105
+ COST DISCIPLINE (behave like a strategist who knows the margins):
106
+ - Free and instant, use liberally: trend_categories, and everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
107
+ - Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, topic_trend, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
108
+ - Paid per-keyword (SERP): related_terms, youtube_discovery (ranking videos for a topic — pair with a transcript tool), news_discovery (recent coverage / freshness), and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
109
+ - Content research (what to write / what changed): topic_trend (is a topic rising/seasonal; with a categoryCode from the free trend_categories, is a whole market growing and what is breaking out in it) → youtube_discovery + news_discovery (what the winning videos/articles cover) → draft_content.
110
+ - Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
111
+
112
+ AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
113
+ 1. Baseline: refresh_property → run_audit → serve_dashboard (share the URL).
114
+ 2. Market: serp_features (feature exposure) + domain_visibility + competitors_domain → topic_gaps vs the named rivals.
115
+ 3. Content plan: suggest_pages (demand you already have) + topic_gaps (demand rivals own) → draft_content briefs.
116
+ 4. Fix cycle: fix_finding per top finding → re-crawl → detect_changes to prove the fix landed.
117
+
133
118
  Before planning ANY complex multi-source question, call composition_cookbook — it returns the full data-surface map and worked recipes using these exact tool and table names.`;
134
- const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
135
-
136
- This server's value is composition: joining Search Console, the crawl, URL Inspection and DataForSEO to answer questions no preset check covers. This page is static (no API calls) — use it to PLAN, then run the tools.
137
-
138
- ## The data surface
139
-
140
- | Source (table) | Grain | Key dimensions | Join keys | Freshness | Cost |
141
- |---|---|---|---|---|---|
142
- | search_analytics (GSC) | date × query × page (+device/country with segments) | clicks, impressions, ctr, position | page_key (url_key), query, date | sync_gsc / refresh_property — incremental, GSC lags ~2–3 days | free |
143
- | pages + links (crawl) | one row per url_key, LATEST crawl | status, title, H1, meta, canonical_key, robots, json_ld, hreflang, redirects, body_chunks, word_count, ipr, click_depth, inlink_count, conditional_304 | url_key | start_crawl / refresh_property (on demand) | free |
144
- | page_snapshots (drift) | url_key × crawl | field-level SEO snapshot per crawl | url_key, captured_at | every crawl, automatically | free |
145
- | url_inspection | one row per inspected URL (top pages by clicks, quota-limited) | coverage_state, page_fetch_state, google_canonical, user_canonical, last_crawl_time, crawled_as, rich_results | url_key | inspect_urls | free (GSC quota) |
146
- | sitemap_urls | one row per sitemap URL | lastmod | url_key | captured at crawl time | free |
147
- | rank_history | month × domain | rank distribution (1–3/4–10/11–20/21–100), ETV | period | track_ranks | paid, 20-day cache |
148
- | page_backlinks | one row per backlinked URL | backlinks, referring_domains, live status_code | url_key, domain | pull_backlinks (needs the DataForSEO Backlinks subscription) | paid, 20-day cache |
149
- | link_prospects | one row per prospect domain (links to competitors, not you) | intersections, domain_trust, spam_score, dofollow, trust_flow, topical_trust_flow | domain | link_intersect (needs the DataForSEO Backlinks subscription; Majestic optional) | paid, 20-day cache |
150
- | keyword_intent | one row per keyword | intent + probability | query | search_intent (pass siteUrl to persist) | paid (cheap), cached |
151
- | page_cwv | one row per audited URL | performance, LCP, CLS, TBT | url_key | page_lighthouse (pass siteUrl to persist) | paid, cached |
152
- | page_entity + entity_edge | one row per page / edge per relation | QID, label, subclass-of / part-of | url_key, qid | resolve_entities (free Wikidata) | free |
153
- | Labs (ranked_keywords, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps) | keyword × target, or month × target | volume, ETV, position, KD, intent, SERP features, AIO citations | domain, query | on demand | paid, 20-day cache |
154
- | findings (audit_runs) | finding per check × URL | priority, evidence JSON | url_key, check_id | run_audit | free |
155
-
156
- Raw access: query_audit runs any single check with full evidence; every table above lives in one SQLite file per property (path via data_location) if you need direct SQL.
157
-
158
- ## Worked recipes
159
-
160
- 1. **AI Overview exposure & citation loss.** ranked_keywords target:<your domain> aioOnly:true → keywords you rank for where the SERP shows an AI Overview (exposure). For citation (are YOU a source?) check specific keywords with related_terms/serpOrganic — the ai_overview item's references[] carries domain/url/quoted text. For each exposed keyword's ranking page, pull its per-day clicks from search_analytics (page_key × date). Pages whose clicks fell while the AIO citation appeared = you are feeding the answer without earning the visit.
161
- 2. **Striking distance without body coverage.** query_audit check:striking-distance (GSC rank 11–20) → for each page, check pages.body_chunks for the query's terms. The automated versions: body-missing-top-query, rag-answer-gap, and score_passages for the dense-answer test. Pages ranking 11–20 that never answer the query in one passage are the highest-yield rewrites.
162
- 3. **Not indexed + no equity.** url_inspection.coverage_state ~ 'not indexed' joined to pages.ipr + inlink_count. Low-iPR unindexed pages need internal links, not resubmission; high-iPR unindexed pages are the real anomalies. (Checks: coverage-not-indexed, underlinked-high-demand.)
163
- 4. **Cannibalisation with semantic overlap.** run_audit → keyword-cannibalisation evidence lists the competing URLs per query → compare those pages' body_chunks: heavy chunk overlap = consolidate (301 the loser); light overlap = differentiate the titles/H1s and interlink with distinct anchors.
164
- 5. **Stable rank, falling CTR → SERP feature shift.** In search_analytics find queries where weekly position is flat but ctr declines → related_terms / ranked_keywords serpFeatures for that keyword shows what now sits above you (AIO, featured snippet, shopping). ctr-below-expected is the deterministic starting list.
165
- 6. **Schema vs rich-result reality.** pages.json_ld (declared @types) joined to url_inspection.rich_results (what Google actually detected + issues). The rich-result-issues check automates the per-URL diff; the composition question is per-TEMPLATE (list_templates): which template's schema never earns its rich result?
166
- 7. **Migration signal transfer.** pages.redirects (recorded chains) → url_inspection.google_canonical of the target (has Google accepted the move?) → search_analytics clicks by page_key before/after the migration date. Equity that didn't follow the 301 shows up as a target with no canonical adoption and no click recovery.
167
- 8. **404s with backlinks.** pages.status_code = 404 joined to page_backlinks.backlinks (run pull_backlinks first) → run_audit surfaces backlinks-to-404; fix_finding generates the 301 that recovers the equity.
168
- 9. **Competitor topic gap.** topic_gaps (bounded + cached: competitor ranked_keywords minus your GSC queries and page titles/H1s, clustered and scored) — or do it manually with ranked_keywords per competitor when you want the raw rows.
169
- 10. **Link gap (what links do rivals have that we don't).** link_intersect over the competitor set (or a single company) → link_prospects: domains linking to them but not you, followed-first and sorted by domain trust. DataForSEO domain rank surfaces spam directories at the top; MAJESTIC_API_KEY re-sorts by Trust Flow (a rank-227 domain is often TF 0) and Topical Trust Flow shows whether that authority is on-topic. data_storage flags when a property's prospect set is going stale (competitors keep earning links).
170
-
171
- ## Novel combinations (nothing else surfaces these)
172
-
173
- - **Crawl budget vs equity:** url_inspection.last_crawl_time × pages.ipr — your highest-iPR pages should be recrawled often; a high-iPR page Google rarely revisits is a freshness/priority problem (and vice versa: junk crawled daily = wasted budget).
174
- - **AIO text vs your content:** the ai_overview items in a SERP call (related_terms' underlying serpOrganic) × pages.body_chunks — is the text Google quotes actually on your page, and in one extractable chunk?
175
- - **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
176
- - **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
177
-
178
- ## Agency engagement recipes (the deliverable arc)
179
-
180
- - **Week-one baseline:** refresh_property → run_audit → serve_dashboard. Share the dashboard URL; the ranked findings ARE the technical workstream.
181
- - **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
182
- - **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
183
- - **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
184
- - **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
185
- - **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
186
-
119
+ const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
120
+
121
+ This server's value is composition: joining Search Console, the crawl, URL Inspection and DataForSEO to answer questions no preset check covers. This page is static (no API calls) — use it to PLAN, then run the tools.
122
+
123
+ ## The data surface
124
+
125
+ | Source (table) | Grain | Key dimensions | Join keys | Freshness | Cost |
126
+ |---|---|---|---|---|---|
127
+ | search_analytics (GSC) | date × query × page (+device/country with segments) | clicks, impressions, ctr, position | page_key (url_key), query, date | sync_gsc / refresh_property — incremental, GSC lags ~2–3 days | free |
128
+ | pages + links (crawl) | one row per url_key, LATEST crawl | status, title, H1, meta, canonical_key, robots, json_ld, hreflang, redirects, body_chunks, word_count, ipr, click_depth, inlink_count, conditional_304 | url_key | start_crawl / refresh_property (on demand) | free |
129
+ | page_snapshots (drift) | url_key × crawl | field-level SEO snapshot per crawl | url_key, captured_at | every crawl, automatically | free |
130
+ | url_inspection | one row per inspected URL (top pages by clicks, quota-limited) | coverage_state, page_fetch_state, google_canonical, user_canonical, last_crawl_time, crawled_as, rich_results | url_key | inspect_urls | free (GSC quota) |
131
+ | sitemap_urls | one row per sitemap URL | lastmod | url_key | captured at crawl time | free |
132
+ | rank_history | month × domain | rank distribution (1–3/4–10/11–20/21–100), ETV | period | track_ranks | paid, 20-day cache |
133
+ | page_backlinks | one row per backlinked URL | backlinks, referring_domains, live status_code | url_key, domain | pull_backlinks (needs the DataForSEO Backlinks subscription) | paid, 20-day cache |
134
+ | link_prospects | one row per prospect domain (links to competitors, not you) | intersections, domain_trust, spam_score, dofollow, trust_flow, topical_trust_flow | domain | link_intersect (needs the DataForSEO Backlinks subscription; Majestic optional) | paid, 20-day cache |
135
+ | keyword_intent | one row per keyword | intent + probability | query | search_intent (pass siteUrl to persist) | paid (cheap), cached |
136
+ | page_cwv | one row per audited URL | performance, LCP, CLS, TBT | url_key | page_lighthouse (pass siteUrl to persist) | paid, cached |
137
+ | page_entity + entity_edge | one row per page / edge per relation | QID, label, subclass-of / part-of | url_key, qid | resolve_entities (free Wikidata) | free |
138
+ | Labs (ranked_keywords, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps) | keyword × target, or month × target | volume, ETV, position, KD, intent, SERP features, AIO citations | domain, query | on demand | paid, 20-day cache |
139
+ | findings (audit_runs) | finding per check × URL | priority, evidence JSON | url_key, check_id | run_audit | free |
140
+
141
+ Raw access: query_audit runs any single check with full evidence; every table above lives in one SQLite file per property (path via data_location) if you need direct SQL.
142
+
143
+ ## Worked recipes
144
+
145
+ 1. **AI Overview exposure & citation loss.** ranked_keywords target:<your domain> aioOnly:true → keywords you rank for where the SERP shows an AI Overview (exposure). For citation (are YOU a source?) check specific keywords with related_terms/serpOrganic — the ai_overview item's references[] carries domain/url/quoted text. For each exposed keyword's ranking page, pull its per-day clicks from search_analytics (page_key × date). Pages whose clicks fell while the AIO citation appeared = you are feeding the answer without earning the visit.
146
+ 2. **Striking distance without body coverage.** query_audit check:striking-distance (GSC rank 11–20) → for each page, check pages.body_chunks for the query's terms. The automated versions: body-missing-top-query, rag-answer-gap, and score_passages for the dense-answer test. Pages ranking 11–20 that never answer the query in one passage are the highest-yield rewrites.
147
+ 3. **Not indexed + no equity.** url_inspection.coverage_state ~ 'not indexed' joined to pages.ipr + inlink_count. Low-iPR unindexed pages need internal links, not resubmission; high-iPR unindexed pages are the real anomalies. (Checks: coverage-not-indexed, underlinked-high-demand.)
148
+ 4. **Cannibalisation with semantic overlap.** run_audit → keyword-cannibalisation evidence lists the competing URLs per query → compare those pages' body_chunks: heavy chunk overlap = consolidate (301 the loser); light overlap = differentiate the titles/H1s and interlink with distinct anchors.
149
+ 5. **Stable rank, falling CTR → SERP feature shift.** In search_analytics find queries where weekly position is flat but ctr declines → related_terms / ranked_keywords serpFeatures for that keyword shows what now sits above you (AIO, featured snippet, shopping). ctr-below-expected is the deterministic starting list.
150
+ 6. **Schema vs rich-result reality.** pages.json_ld (declared @types) joined to url_inspection.rich_results (what Google actually detected + issues). The rich-result-issues check automates the per-URL diff; the composition question is per-TEMPLATE (list_templates): which template's schema never earns its rich result?
151
+ 7. **Migration signal transfer.** pages.redirects (recorded chains) → url_inspection.google_canonical of the target (has Google accepted the move?) → search_analytics clicks by page_key before/after the migration date. Equity that didn't follow the 301 shows up as a target with no canonical adoption and no click recovery.
152
+ 8. **404s with backlinks.** pages.status_code = 404 joined to page_backlinks.backlinks (run pull_backlinks first) → run_audit surfaces backlinks-to-404; fix_finding generates the 301 that recovers the equity.
153
+ 9. **Competitor topic gap.** topic_gaps (bounded + cached: competitor ranked_keywords minus your GSC queries and page titles/H1s, clustered and scored) — or do it manually with ranked_keywords per competitor when you want the raw rows.
154
+ 10. **Link gap (what links do rivals have that we don't).** link_intersect over the competitor set (or a single company) → link_prospects: domains linking to them but not you, followed-first and sorted by domain trust. DataForSEO domain rank surfaces spam directories at the top; MAJESTIC_API_KEY re-sorts by Trust Flow (a rank-227 domain is often TF 0) and Topical Trust Flow shows whether that authority is on-topic. data_storage flags when a property's prospect set is going stale (competitors keep earning links).
155
+
156
+ ## Novel combinations (nothing else surfaces these)
157
+
158
+ - **Crawl budget vs equity:** url_inspection.last_crawl_time × pages.ipr — your highest-iPR pages should be recrawled often; a high-iPR page Google rarely revisits is a freshness/priority problem (and vice versa: junk crawled daily = wasted budget).
159
+ - **AIO text vs your content:** the ai_overview items in a SERP call (related_terms' underlying serpOrganic) × pages.body_chunks — is the text Google quotes actually on your page, and in one extractable chunk?
160
+ - **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
161
+ - **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
162
+
163
+ ## Agency engagement recipes (the deliverable arc)
164
+
165
+ - **Week-one baseline:** refresh_property → run_audit → serve_dashboard. Share the dashboard URL; the ranked findings ARE the technical workstream.
166
+ - **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
167
+ - **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
168
+ - **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
169
+ - **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
170
+ - **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
171
+
187
172
  Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
188
- // The check catalogue rendered as markdown — single source of truth is listChecks();
189
- // shared by the seo-audit://checks-reference resource (and buildable for any category subset).
190
173
  function buildChecksMarkdown(checks) {
191
174
  const byCategory = new Map();
192
175
  for (const c of checks) {
@@ -200,53 +183,54 @@ function buildChecksMarkdown(checks) {
200
183
  list.map(c => `| \`${c.id}\` | ${c.severity} | ${c.labels.join(',')} | ${c.certainty} | ${c.fixType} | ${esc(c.title)} | ${esc(c.fix)} |`).join('\n'));
201
184
  return `# Check registry — ${checks.length} checks\n\nLabels: D = deterministic (cites bytes), G = evidence from Google's own data (Search Console / URL Inspection), N = judgement (heuristic, gated behind includeJudgement). Certainty < 1 discounts a finding's priority.\n\n${sections.join('\n\n')}\n`;
202
185
  }
203
- const HELP_TEXT = `# SEO Audit Console — what it can do
204
- A technical-SEO audit that fuses **Search Console + a site crawl + DataForSEO**, joined on a normalised URL, with evidence on every finding. Typical flow: **refresh → audit → fix → report**.
205
-
206
- ## 1. Sync the data
207
- - **refresh_property** — sync everything (GSC → crawl → URL inspection → rank history). _"Refresh sc-domain:example.com"_ · add \`segments:true\` for device/country, \`maxPages\`, \`startDate\`.
208
- - **sync_gsc / start_crawl / inspect_urls / track_ranks** — run just one part. _"Crawl example.com, 500 pages"_
209
- - **check_sync_status / check_crawl_status** — poll a job. _"Check sync status"_
210
- - **list_properties** — _"List my Search Console properties"_
211
-
212
- ## 2. Audit
213
- - **run_audit** — score all checks; returns a prioritised markdown report. _"Run an SEO audit on sc-domain:example.com"_ · \`scope:full\`, \`categories\`, \`includeJudgement:true\`.
214
- - **query_audit** — one named check with evidence (\`columns\`/\`offset\` for big sets). _"Show striking-distance for example.com"_
215
- - **query_data** — read-only queries over the raw tables; aggregates in the database (counts/percentages/sums), answers not rows. _"How do status codes break down on example.com?"_
216
- - **list_checks** — _"What does the audit check for?"_
217
-
218
- ## 3. Fix (the moat)
219
- - **fix_finding** — paste-ready remediation from your own data: JSON-LD for missing/invalid schema, a 301 rule for broken links, iPR-ranked internal-link suggestions. _"Generate the fix for finding 12"_ or _"fix_finding check:missing-required-fields url:https://example.com/x"_
220
- - **detect_changes** — what changed since the last crawl (status, canonical, noindex, title, schema), severity-ranked — the regression monitor. _"Detect changes on example.com"_
221
- - **check_agent_readiness** — is your site ready for AI agents? Scores llms.txt / agents.md / AI-bot rules / Content Signals / MCP server card / Agent Skills / API Catalog / OAuth signals (0–100 + level) with copy-paste fixes. _"Check agent readiness for example.com"_
222
-
223
- ## 4. Backlinks, keywords & competitive (DataForSEO, on-demand, cached 20 days)
224
- - **pull_backlinks** — backlink profile + per-page counts + live status → unlocks **backlinks-to-404** (recover lost equity), top-linked pages, true orphans. _"Pull backlinks for example.com"_
225
- - **link_intersect** — the links your competitors have that you don't - a prioritised outreach prospect list (followed-first, then domain trust, spam filtered). Also answers "what links does company X have that we don't?" for a single company. Set MAJESTIC_API_KEY to re-sort by Trust Flow + Topical Trust Flow (kills directory noise). _"Link intersect for example.com vs rival1.com, rival2.com"_
226
- - **keyword_volume / related_terms** — volume/CPC, and People-Also-Ask + related searches. _"Search volume for [\\"best widgets\\"]"_
227
- - **search_intent** — informational/navigational/commercial/transactional per keyword → spot intent mismatch behind low CTR. _"Classify intent for [\\"buy running shoes\\", \\"how to clean shoes\\"]"_
228
- - **page_lighthouse** — lab Core Web Vitals + opportunities for one URL (~20–120s). _"Run Lighthouse on https://example.com/slow-page"_
229
- - **competitors_domain** — domains competing for your organic keywords. _"Find competitors for example.com in the UK"_
230
- - **page_intersection** — keywords competitor pages rank for but yours doesn’t (content gap). _"Content gap: competitorUrls [\\"https://rival.com/guide\\"], excludePages [\\"https://example.com/guide\\"]"_
231
- - **domain_visibility** — monthly ranking-keyword distribution + ETV trend for ANY domain/subdomain (Semrush-style organic overview). _"Show visibility over time for competitor.com"_
232
- - **top_pages** — a domain's top organic pages by estimated traffic. _"Top pages on competitor.com"_
233
- - **ranked_keywords** — keywords a domain / subdomain / URL / subfolder ranks for (+ difficulty, intent, SERP features). _"What does competitor.com/blog/ rank for?"_ · \`scope:url|folder\` · \`aioOnly:true\` = keywords where the target is cited in AI Overviews
234
- - **topic_gaps** — what related topics should this site cover to be expert in its space: competitor keyword footprints minus everything you already rank or have a page for, clustered into ranked topics with volumes, the owning competitor and your nearest existing page. _"What topics should example.com cover? Compare against rival.com"_
235
-
236
- ## 5. Templates & opportunities
237
- - **list_templates** — cluster pages into templates (one fix → N pages) with a representative exemplar. _"List page templates for example.com"_
238
- - **suggest_pages** — new-page ideas grounded in real GSC demand, minus what you already cover. _"Suggest new pages for example.com"_
239
- - **resolve_entities** — map pages to Wikidata entities → unlocks entity-internal-link-gap (judgement). _"Resolve entities for example.com"_
240
-
241
- ## 6. Reports & dashboard
242
- - **get_dashboard** — interactive dashboard (renders in chat). _"Show the dashboard for example.com"_
243
- - **export_report** — self-contained shareable HTML to send a client. _"Export the report for example.com"_
244
-
245
- ## 7. Utilities
246
- - **data_location** — where DBs are stored (set with a path). · **normalize_url** — the join key for a URL.
247
- - **data_storage** — per-property disk usage + row counts; prune (vacuum / clear-crawl-history / delete-property, destructive ones need \`confirm:true\`). _"How much disk is my audit data using?"_
248
-
249
- _Tip: first time on a property → \`refresh_property\` then \`run_audit\`._
186
+ const HELP_TEXT = `# SEO Audit Console — what it can do
187
+ A technical-SEO audit that fuses **Search Console + a site crawl + DataForSEO**, joined on a normalised URL, with evidence on every finding. Typical flow: **refresh → audit → fix → report**.
188
+
189
+ ## 1. Sync the data
190
+ - **refresh_property** — sync everything (GSC → crawl → URL inspection → rank history). _"Refresh sc-domain:example.com"_ · add \`segments:true\` for device/country, \`maxPages\`, \`startDate\`.
191
+ - **sync_gsc / start_crawl / inspect_urls / track_ranks** — run just one part. _"Crawl example.com, 500 pages"_
192
+ - **check_sync_status / check_crawl_status** — poll a job. _"Check sync status"_
193
+ - **list_properties** — _"List my Search Console properties"_
194
+
195
+ ## 2. Audit
196
+ - **run_audit** — score all checks; returns a prioritised markdown report. _"Run an SEO audit on sc-domain:example.com"_ · \`scope:full\`, \`categories\`, \`includeJudgement:true\`.
197
+ - **query_audit** — one named check with evidence (\`columns\`/\`offset\` for big sets). _"Show striking-distance for example.com"_
198
+ - **query_data** — read-only queries over the raw tables; aggregates in the database (counts/percentages/sums), answers not rows. _"How do status codes break down on example.com?"_
199
+ - **list_checks** — _"What does the audit check for?"_
200
+
201
+ ## 3. Fix (the moat)
202
+ - **fix_finding** — paste-ready remediation from your own data: JSON-LD for missing/invalid schema, a 301 rule for broken links, iPR-ranked internal-link suggestions. _"Generate the fix for finding 12"_ or _"fix_finding check:missing-required-fields url:https://example.com/x"_
203
+ - **detect_changes** — what changed since the last crawl (status, canonical, noindex, title, schema), severity-ranked — the regression monitor. _"Detect changes on example.com"_
204
+ - **check_agent_readiness** — is your site ready for AI agents? Scores llms.txt / agents.md / AI-bot rules / Content Signals / MCP server card / Agent Skills / API Catalog / OAuth signals (0–100 + level) with copy-paste fixes. _"Check agent readiness for example.com"_
205
+
206
+ ## 4. Backlinks, keywords & competitive (DataForSEO, on-demand, cached 20 days)
207
+ - **pull_backlinks** — backlink profile + per-page counts + live status → unlocks **backlinks-to-404** (recover lost equity), top-linked pages, true orphans. _"Pull backlinks for example.com"_
208
+ - **link_intersect** — the links your competitors have that you don't - a prioritised outreach prospect list (followed-first, then domain trust, spam filtered). Also answers "what links does company X have that we don't?" for a single company. Set MAJESTIC_API_KEY to re-sort by Trust Flow + Topical Trust Flow (kills directory noise). _"Link intersect for example.com vs rival1.com, rival2.com"_
209
+ - **keyword_volume / related_terms** — volume/CPC, and People-Also-Ask + related searches. _"Search volume for [\\"best widgets\\"]"_
210
+ - **topic_trend / trend_categories** — Google Trends interest over time for keywords, a whole category (no keyword needed), or a keyword inside a category; \`related:true\` adds rising queries and topics. trend_categories (free) finds the category code. _"Is the Software category rising in the UK? What's breaking out in it?"_
211
+ - **search_intent** — informational/navigational/commercial/transactional per keyword → spot intent mismatch behind low CTR. _"Classify intent for [\\"buy running shoes\\", \\"how to clean shoes\\"]"_
212
+ - **page_lighthouse** — lab Core Web Vitals + opportunities for one URL (~20–120s). _"Run Lighthouse on https://example.com/slow-page"_
213
+ - **competitors_domain** — domains competing for your organic keywords. _"Find competitors for example.com in the UK"_
214
+ - **page_intersection** — keywords competitor pages rank for but yours doesn’t (content gap). _"Content gap: competitorUrls [\\"https://rival.com/guide\\"], excludePages [\\"https://example.com/guide\\"]"_
215
+ - **domain_visibility** — monthly ranking-keyword distribution + ETV trend for ANY domain/subdomain (Semrush-style organic overview). _"Show visibility over time for competitor.com"_
216
+ - **top_pages** — a domain's top organic pages by estimated traffic. _"Top pages on competitor.com"_
217
+ - **ranked_keywords** — keywords a domain / subdomain / URL / subfolder ranks for (+ difficulty, intent, SERP features). _"What does competitor.com/blog/ rank for?"_ · \`scope:url|folder\` · \`aioOnly:true\` = keywords where the target is cited in AI Overviews
218
+ - **topic_gaps** — what related topics should this site cover to be expert in its space: competitor keyword footprints minus everything you already rank or have a page for, clustered into ranked topics with volumes, the owning competitor and your nearest existing page. _"What topics should example.com cover? Compare against rival.com"_
219
+
220
+ ## 5. Templates & opportunities
221
+ - **list_templates** — cluster pages into templates (one fix → N pages) with a representative exemplar. _"List page templates for example.com"_
222
+ - **suggest_pages** — new-page ideas grounded in real GSC demand, minus what you already cover. _"Suggest new pages for example.com"_
223
+ - **resolve_entities** — map pages to Wikidata entities → unlocks entity-internal-link-gap (judgement). _"Resolve entities for example.com"_
224
+
225
+ ## 6. Reports & dashboard
226
+ - **get_dashboard** — interactive dashboard (renders in chat). _"Show the dashboard for example.com"_
227
+ - **export_report** — self-contained shareable HTML to send a client. _"Export the report for example.com"_
228
+
229
+ ## 7. Utilities
230
+ - **data_location** — where DBs are stored (set with a path). · **normalize_url** — the join key for a URL.
231
+ - **data_storage** — per-property disk usage + row counts; prune (vacuum / clear-crawl-history / delete-property, destructive ones need \`confirm:true\`). _"How much disk is my audit data using?"_
232
+
233
+ _Tip: first time on a property → \`refresh_property\` then \`run_audit\`._
250
234
  _Composing your own analysis? Call \`composition_cookbook\` first — the data-surface map (grain + join keys per source) and worked multi-source recipes._`;
251
235
  export function createServer() {
252
236
  const server = new McpServer({ name: SERVER_NAME, version: SERVER_VERSION }, { instructions: SERVER_INSTRUCTIONS });
@@ -255,7 +239,7 @@ export function createServer() {
255
239
  const gsc = credPath ? new GscClient(credPath) : null;
256
240
  const sync = gsc ? new GscSync(gsc, dataDir()) : null;
257
241
  const inspector = gsc ? new UrlInspector(gsc, dataDir()) : null;
258
- const crawler = new Crawler(dataDir()); // no GSC credentials required
242
+ const crawler = new Crawler(dataDir());
259
243
  const dfsUser = process.env.DATAFORSEO_USERNAME;
260
244
  const dfsPass = process.env.DATAFORSEO_PASSWORD;
261
245
  const dfsCacheDays = Number(process.env.DATAFORSEO_CACHE_DAYS) || 7;
@@ -264,16 +248,12 @@ export function createServer() {
264
248
  : null;
265
249
  const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
266
250
  const linkIntersect = dfs ? new LinkIntersect(dfs, dataDir()) : null;
267
- // Majestic (Trust Flow / Topical Trust Flow) — optional link_intersect + trapped-authority tier.
268
251
  const majesticKey = process.env.MAJESTIC_API_KEY;
269
252
  const majesticCacheDays = Number(process.env.MAJESTIC_CACHE_DAYS) || 30;
270
253
  const majestic = majesticKey ? new MajesticClient(majesticKey, path.join(dataDir(), 'majestic-cache.db'), majesticCacheDays) : null;
271
- // Backlinks after Majestic, so pull_backlinks can enrich per-URL pages with Trust Flow.
272
254
  const backlinks = dfs ? new Backlinks(dfs, dataDir(), majestic) : null;
273
- // Firecrawl (competitor-page scraping for content recon) — optional; degrades gracefully.
274
255
  const firecrawlKey = process.env.FIRECRAWL_API_KEY;
275
256
  const firecrawl = firecrawlKey ? new FirecrawlClient(firecrawlKey, path.join(dataDir(), 'firecrawl-cache.db')) : null;
276
- // Supadata (transcribes the ranking videos for content recon) — optional; degrades gracefully.
277
257
  const supadataKey = process.env.SUPADATA_API_KEY;
278
258
  const supadata = supadataKey ? new SupadataClient(supadataKey, path.join(dataDir(), 'supadata-cache.db')) : null;
279
259
  const entities = new Entities(new WikidataClient(path.join(dataDir(), 'wikidata-cache.db')), dataDir());
@@ -288,33 +268,21 @@ export function createServer() {
288
268
  throw new Error('DATAFORSEO_USERNAME / DATAFORSEO_PASSWORD not set — required for DataForSEO.');
289
269
  return v;
290
270
  };
291
- // Which optional integrations have a key set — drives the dashboard's key-gated tabs
292
- // (Links, Content research) and their affiliate-linked upsell states. The data layer has
293
- // no env access, so it's injected here onto every dashboard payload surface.
294
271
  const apiKeysStatus = () => ({ dataforseo: !!dfs, majestic: !!majestic, firecrawl: !!firecrawl, supadata: !!supadata });
295
- // The dashboard webserver's options — one definition shared by serve_dashboard AND the
296
- // auto-start at the end of refresh_property / run_audit, so a populated property always has
297
- // a live browser link (browserLink() surfaces dashboardServerUrl() once this is running).
298
272
  const buildWebOpts = (port) => ({
299
273
  dataDir,
300
274
  uiHtml: () => readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8'),
301
275
  call: buildWebCallHandlers({ dataDir, dfs, apiKeys: apiKeysStatus }),
302
276
  ...(port != null ? { port } : {}),
303
277
  });
304
- // Idempotent auto-start used by refresh_property / run_audit. Best-effort: a served port that
305
- // can't bind (e.g. Docker without a published port) must never fail the populate/audit — the
306
- // in-chat get_dashboard and export_report still work, and browserLink() falls back to those.
307
- // Opt-out with SAC_AUTOSERVE=0 for headless/CI/sandboxed runs where binding a localhost port
308
- // is unwanted (default: on, since the whole point is to hand the user a link).
309
278
  const autoServeDashboard = async () => {
310
279
  if (/^(0|false|no|off)$/i.test(process.env.SAC_AUTOSERVE ?? ''))
311
280
  return;
312
281
  try {
313
282
  await startDashboardServer(buildWebOpts());
314
283
  }
315
- catch { /* dashboard is a convenience, not a dependency */ }
284
+ catch { }
316
285
  };
317
- // ── Introspection (no data / creds required) ────────────────────────────
318
286
  server.registerTool('seo_audit_help', {
319
287
  title: 'Help — what this audit can do',
320
288
  description: 'Overview of every tool/feature with an example prompt for each. Start here.',
@@ -358,8 +326,6 @@ export function createServer() {
358
326
  inputSchema: { path: z.string().optional() },
359
327
  }, async ({ path: newPath }) => {
360
328
  if (newPath) {
361
- // Persist an ABSOLUTE path — a relative one resolves against the host process's
362
- // cwd, which differs across Claude Desktop launches, so DBs would "disappear".
363
329
  newPath = path.resolve(newPath);
364
330
  mkdirSync(newPath, { recursive: true });
365
331
  writeFileSync(CONFIG_PATH, JSON.stringify({ dataDir: newPath }, null, 2));
@@ -405,10 +371,6 @@ export function createServer() {
405
371
  }
406
372
  const rows = s.properties.map(p => `| ${p.siteUrl ?? p.file} | ${fmtBytes(p.bytes)} | ${p.searchAnalytics.toLocaleString('en-US')} | ${p.pages.toLocaleString('en-US')} | ${p.links.toLocaleString('en-US')} | ${p.pageSnapshots.toLocaleString('en-US')} | ${p.findings.toLocaleString('en-US')} | ${p.lastSynced?.slice(0, 10) ?? '—'} | ${p.lastCrawl?.slice(0, 10) ?? '—'} |`).join('\n');
407
373
  const cacheLines = s.caches.map(c => `- ${c.file}: ${fmtBytes(c.bytes)}`).join('\n');
408
- // Link-intersect freshness: prospect data ages as competitors keep earning links, so
409
- // flag properties that HAVE link_prospects and roughly how stale it is — a re-run of
410
- // link_intersect updates it. (The DataForSEO call is 20-day cached; older than that a
411
- // re-run genuinely refetches.)
412
374
  const nowMs = Date.now();
413
375
  const liProps = s.properties.filter(p => (p.linkProspects ?? 0) > 0);
414
376
  const liLines = liProps.map(p => {
@@ -425,7 +387,6 @@ export function createServer() {
425
387
  `Prune with data_storage prune:{siteUrl, action:"vacuum" | "clear-crawl-history" | "delete-property"} (destructive actions need confirm:true).`;
426
388
  return { content: [{ type: 'text', text: md }], structuredContent: s };
427
389
  });
428
- // ── Audit engine ────────────────────────────────────────────────────────
429
390
  server.registerTool('run_audit', {
430
391
  title: 'Run SEO audit',
431
392
  description: 'Run the technical-SEO checks against synced data and return scored findings, ranked by expected clicks per dev-hour — Priority = (T × Y × C) / E (T = clicks at stake from real GSC data, Y = expected yield, C = certainty, E = effort hours). Crawl + GSC + URL-inspection checks. Set includeJudgement=true to include heuristic (N) checks.',
@@ -437,7 +398,7 @@ export function createServer() {
437
398
  },
438
399
  }, async ({ siteUrl, scope, categories, includeJudgement }) => {
439
400
  const result = runAudit(dataDir(), siteUrl, { scope, categories, includeJudgement });
440
- await autoServeDashboard(); // spin up the browser dashboard so browserLink() hands over a live URL
401
+ await autoServeDashboard();
441
402
  return {
442
403
  content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) + browserLink(siteUrl) }],
443
404
  structuredContent: { ...result, dashboardUrl: dashboardServerUrl() },
@@ -455,8 +416,6 @@ export function createServer() {
455
416
  },
456
417
  }, async ({ siteUrl, check, limit, offset, columns }) => {
457
418
  const r = runSingleCheck(dataDir(), siteUrl, check, limit, offset ?? 0);
458
- // Token discipline (mirrors query_data): optional evidence-column selection + loud
459
- // 120-char cell truncation, and an explicit offset/total footer.
460
419
  const shape = (f) => {
461
420
  const src = (f.evidence ?? {});
462
421
  const keys = columns?.length ? columns.filter(k => k in src) : Object.keys(src);
@@ -467,9 +426,6 @@ export function createServer() {
467
426
  };
468
427
  let sc = { ...r, findingsTotal: r.total };
469
428
  let findings = r.findings.map(shape);
470
- // Hosts cap the model-facing result (~60k chars). On evidence-heavy checks a large
471
- // `limit` can blow that ceiling and error the whole call — trim findings (keeping the
472
- // true total) instead of failing. Same guard pattern as detect_changes.
473
429
  while (findings.length > 25 && JSON.stringify({ ...sc, findings }).length > 45000) {
474
430
  findings = findings.slice(0, Math.floor(findings.length / 2));
475
431
  }
@@ -540,7 +496,6 @@ export function createServer() {
540
496
  const text = `WRITING BRIEF — ${r.url}\nTarget query: “${r.topQuery ?? '(unknown)'}”\nGap: ${r.gap}\n\nTASK\n${r.brief}\n\nVOICE (match this — the page's own writing):\n${voice || ' (no substantial passages)'}\n\nFACTS (ground in these — the page's most query-relevant content; invent nothing beyond them):\n${facts || ' (none)'}`;
541
497
  return { content: [{ type: 'text', text }], structuredContent: r };
542
498
  });
543
- // detect_changes — change-detection / drift: what changed on the site since the last crawl.
544
499
  server.registerTool('detect_changes', {
545
500
  title: 'Detect changes since the last crawl',
546
501
  description: 'Compare the two most recent crawls and report what changed per URL — status code, indexability, canonical, robots/noindex, title, meta, H1, schema, large content swings — each classified by severity (critical → info). This is the monitor: run refresh_property on different days to build history, then this surfaces regressions (a page that went noindex, a canonical that flipped, a 200 that became a 404). Needs at least two crawls.',
@@ -549,9 +504,6 @@ export function createServer() {
549
504
  const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
550
505
  try {
551
506
  const d = diffLatest(adb.db);
552
- // Cap the changes array in structuredContent to `limit` — the full diff can be tens of
553
- // thousands of rows (e.g. a crawl-methodology change) and blow the MCP token ceiling. The
554
- // summary keeps the TRUE totals; markdown is already limited.
555
507
  const lim = limit ?? 50;
556
508
  const sc = { ...d, changes: d.changes.slice(0, lim), changesReturned: Math.min(lim, d.changes.length), changesTotal: d.changes.length };
557
509
  return {
@@ -563,18 +515,16 @@ export function createServer() {
563
515
  adb.close();
564
516
  }
565
517
  });
566
- // check_agent_readiness — is the site ready for AI agents? (the agentic-SEO / GEO frontier)
567
518
  server.registerTool('check_agent_readiness', {
568
519
  title: 'Check agent readiness (AI-agent / GEO signals)',
569
520
  description: 'Probe a site for AI-agent readiness — the signals agents use to discover and use it: robots.txt AI-bot rules + Content Signals, sitemap, Link headers, llms.txt, agents.md, Markdown content negotiation, Web Bot Auth, MCP server card, Agent Skills, API Catalog, OAuth discovery. Returns a 0–100 score, a level (Basic web presence → Agent-native), and a per-check checklist with copy-paste fixes. Live HTTP probes of the property origin (no crawl/GSC needed). Modelled on Cloudflare\'s isitagentready.com.',
570
521
  inputSchema: { siteUrl: z.string() },
571
522
  }, async ({ siteUrl }) => {
572
523
  const r = await checkAgentReadiness(siteUrl);
573
- // Persist so the dashboard/export can show it (live probe, not crawl-derived).
574
524
  try {
575
525
  const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
576
526
  try {
577
- adb.db.prepare(`INSERT INTO agent_readiness (origin,score,level,checks,by_category,checked_at) VALUES (?,?,?,?,?,datetime('now'))
527
+ adb.db.prepare(`INSERT INTO agent_readiness (origin,score,level,checks,by_category,checked_at) VALUES (?,?,?,?,?,datetime('now'))
578
528
  ON CONFLICT(origin) DO UPDATE SET score=excluded.score, level=excluded.level, checks=excluded.checks, by_category=excluded.by_category, checked_at=excluded.checked_at`)
579
529
  .run(r.origin, r.score, r.level, JSON.stringify(r.checks), JSON.stringify(r.byCategory));
580
530
  }
@@ -582,13 +532,12 @@ export function createServer() {
582
532
  adb.close();
583
533
  }
584
534
  }
585
- catch { /* persistence is best-effort */ }
535
+ catch { }
586
536
  return {
587
537
  content: [{ type: 'text', text: buildAgentReadinessMarkdown(r) }],
588
538
  structuredContent: r,
589
539
  };
590
540
  });
591
- // fix_finding — the moat: turn a finding into a concrete, paste-ready remediation.
592
541
  server.registerTool('fix_finding', {
593
542
  title: 'Generate a fix for a finding',
594
543
  description: 'Turn an audit finding into a concrete, paste-ready remediation: JSON-LD for missing structured data, a 301 rule for broken / redirecting internal links, or internal-link suggestions for orphan / striking-distance pages. Other checks return their deterministic fix guidance. Dry-run — returns artifacts, never writes to your site. Identify the finding by findingId (from run_audit) or by check + url.',
@@ -602,7 +551,6 @@ export function createServer() {
602
551
  }, async ({ siteUrl, findingId, check, url, redirectFormat }) => {
603
552
  const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
604
553
  try {
605
- // Resolve the finding → checkId, affected url_key, evidence.
606
554
  let checkId = check;
607
555
  let affectedKey = null;
608
556
  let evidence = {};
@@ -641,14 +589,12 @@ export function createServer() {
641
589
  break;
642
590
  }
643
591
  case 'redirect-chain': {
644
- // The finding's url_key IS the chain's final page — its row holds the recorded
645
- // hops. Collapse each hop straight to the final URL; never fuzzy-match here.
646
592
  const page = db.db.prepare('SELECT url, redirects FROM pages WHERE url_key = ?').get(affectedKey);
647
593
  let hops = [];
648
594
  try {
649
595
  hops = page?.redirects ? JSON.parse(page.redirects) : [];
650
596
  }
651
- catch { /* malformed chain */ }
597
+ catch { }
652
598
  kind = 'redirect';
653
599
  fix = page && hops.length
654
600
  ? { ...collapseChainRules(page.url, hops, redirectFormat ?? 'htaccess'), chain: hops }
@@ -656,8 +602,6 @@ export function createServer() {
656
602
  break;
657
603
  }
658
604
  case 'internal-links-to-redirects': {
659
- // The flagged key is a redirect SOURCE; the crawler recorded where it actually
660
- // goes — find the chain containing it and use that real destination.
661
605
  const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
662
606
  let knownTarget = null;
663
607
  for (const row of db.db.prepare(`SELECT url, redirects FROM pages WHERE redirects IS NOT NULL AND redirects <> ''`).iterate()) {
@@ -668,15 +612,13 @@ export function createServer() {
668
612
  break;
669
613
  }
670
614
  }
671
- catch { /* skip malformed */ }
615
+ catch { }
672
616
  }
673
617
  kind = 'redirect';
674
618
  fix = suggestRedirect(affectedKey ?? '', [], redirectFormat ?? 'htaccess', knownTarget);
675
619
  break;
676
620
  }
677
621
  case 'broken-internal-links': {
678
- // Genuinely dead target (4xx/5xx) — no recorded destination exists, so fall back
679
- // to token-overlap fuzzy matching against live pages.
680
622
  const livePages = db.db
681
623
  .prepare('SELECT url FROM pages WHERE status_code = 200 AND indexable = 1')
682
624
  .all();
@@ -687,11 +629,8 @@ export function createServer() {
687
629
  case 'orphan-with-impressions':
688
630
  case 'striking-distance': {
689
631
  const page = db.db.prepare('SELECT h1, title FROM pages WHERE url_key = ?').get(affectedKey);
690
- // Prefer the GSC query as anchor, but fall back to H1/title when it's a
691
- // boolean/over-long search string (common on job boards) — not usable anchor text.
692
632
  const q = evidence.query;
693
633
  const cleanQuery = q && q.length <= 60 && !/["()]|\bor\b|\bnot\b|\s-\w/i.test(q) ? q : undefined;
694
- // Never fall back to the raw q — it was just rejected as unusable anchor text.
695
634
  const anchor = cleanQuery ?? page?.h1 ?? page?.title ?? '';
696
635
  kind = 'internal-links';
697
636
  fix = suggestInternalLinks(db.db, affectedKey ?? '', anchor);
@@ -721,7 +660,6 @@ export function createServer() {
721
660
  const key = urlKey(url, { hostForm });
722
661
  return { content: [{ type: 'text', text: key }], structuredContent: { url, key, hostForm } };
723
662
  });
724
- // ── GSC ─────────────────────────────────────────────────────────────────
725
663
  server.registerTool('list_properties', {
726
664
  title: 'List GSC properties',
727
665
  description: 'List Google Search Console properties accessible to the service account.',
@@ -733,8 +671,6 @@ export function createServer() {
733
671
  structuredContent: { properties },
734
672
  };
735
673
  });
736
- // refresh_property — the "sync everything" verb (GSC + crawl + inspection in one job).
737
- // Skip flags let the same tool do "just update X".
738
674
  registerAppTool(server, 'refresh_property', {
739
675
  title: 'Refresh a property (sync + crawl + inspect)',
740
676
  description: 'Full refresh for a property in one async job: GSC sync → site crawl → URL inspection → DataForSEO rank history. Opens a live progress widget (phases + counts). Set gsc/crawl/inspect/ranks=false to run just part. GSC sync is "lite" (date×query×page) by default — set segments=true to also pull device/country breakdowns (much heavier on large sites). GSC sync is INCREMENTAL: the first sync pulls the window (default last 90 days; pass startDate to go deeper), later syncs only fetch new days — set full=true to force a full re-pull. Use this for "sync everything"; use the single-purpose tools to update just one thing.',
@@ -756,8 +692,6 @@ export function createServer() {
756
692
  }, async ({ siteUrl, gsc: doGsc, crawl, inspect, ranks, segments, full, location, startDate, endDate, maxPages, inspectLimit }) => {
757
693
  const jobId = jobs.start('refresh', async (update, signal) => {
758
694
  const r = await refresh.run(siteUrl, { gsc: doGsc, crawl, inspect, ranks, segments, full, location, startDate, endDate, maxPages, inspectLimit }, update, signal);
759
- // Property is now populated — spin up the browser dashboard and hand back its URL so the
760
- // finished job (polled via check_sync_status) carries a one-click link, no extra tool call.
761
695
  await autoServeDashboard();
762
696
  const dashboardUrl = dashboardServerUrl();
763
697
  return dashboardUrl ? { ...r, dashboardUrl, dashboard: `${dashboardUrl}/dashboard?siteUrl=${encodeURIComponent(siteUrl)}` } : r;
@@ -820,7 +754,6 @@ export function createServer() {
820
754
  structuredContent: { jobId, status: 'running', siteUrl },
821
755
  };
822
756
  });
823
- // ── Crawl ─────────────────────────────────────────────────────────────────
824
757
  server.registerTool('start_crawl', {
825
758
  title: 'Crawl a site (just the crawl)',
826
759
  description: 'Update just the crawl: fetch the site into the local database (async job — poll with check_crawl_status). HTTP crawl; respects robots.txt; asset file-types are HEAD-only (no body download); internal-search / cart / wp-json / builder junk URLs are skipped by default. For a full refresh use refresh_property. excludePatterns adds extra URL regexes to skip (e.g. ["/author/","/tag/","/page/"]) to keep big crawls light. Grain: one row per url_key, latest crawl. Joins: url_key → GSC/inspection/backlinks.',
@@ -854,7 +787,6 @@ export function createServer() {
854
787
  const all = jobs.list();
855
788
  return { content: [{ type: 'text', text: `${all.length} jobs` }], structuredContent: { jobs: all } };
856
789
  });
857
- // ── DataForSEO (cached 20 days, single-worker) ──────────────────────────
858
790
  server.registerTool('keyword_volume', {
859
791
  title: 'Keyword search volume (DataForSEO)',
860
792
  description: '[Paid: Keywords API, cheap, cached 20d | Use for: demand sizing] True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
@@ -924,7 +856,6 @@ export function createServer() {
924
856
  articles.push(a);
925
857
  } };
926
858
  let cost = 0, cached = true, googleCount = 0, dfsCount = 0, googleError = null;
927
- // Free Google News RSS.
928
859
  if (src === 'google' || src === 'both') {
929
860
  try {
930
861
  const g = await fetchGoogleNews(keyword, { limit: n });
@@ -936,7 +867,6 @@ export function createServer() {
936
867
  googleError = e instanceof Error ? e.message : 'failed';
937
868
  }
938
869
  }
939
- // Paid DataForSEO Google News SERP (only when explicitly asked, or 'both' AND a key is set).
940
870
  if (src === 'dataforseo' || (src === 'both' && !!dfs)) {
941
871
  const r = await requireDfs(dfs).serpNews(keyword, location, languageCode, n);
942
872
  cost += r.cost;
@@ -945,8 +875,6 @@ export function createServer() {
945
875
  for (const a of r.articles)
946
876
  add({ ...a, via: 'dataforseo' });
947
877
  }
948
- // Top sources: which publishers are covering this topic, ranked by article count — the
949
- // "who is talking about X" read. Powers "trending news + sources for X" in one call.
950
878
  const sourceCount = new Map();
951
879
  for (const a of articles) {
952
880
  const s = String(a.source ?? '').trim();
@@ -964,17 +892,22 @@ export function createServer() {
964
892
  });
965
893
  server.registerTool('topic_trend', {
966
894
  title: 'Google Trends interest over time (DataForSEO)',
967
- description: '[Paid: Keywords API, cheap, cached 20d | Use for: seasonality + "is this rising or fading" — should we update now, when to publish] Google Trends relative interest (0–100) over time for up to 5 keywords (DataForSEO KEYWORDS_DATA / google_trends). Returns a per-keyword time series plus a rising/falling/flat read, so you can see direction and seasonality. timeRange: past_7_days | past_30_days | past_90_days | past_12_months (default) | past_5_years | 2004_present. type: web (default) | news | youtube | images. Default location: United States.',
895
+ description: '[Paid: Keywords API, cheap, cached 20d | Use for: seasonality + "is this rising or fading" — should we update now, when to publish; category monitoring — is a whole market growing] Google Trends relative interest (0–100) over time (DataForSEO KEYWORDS_DATA / google_trends). Give up to 5 keywords, a categoryCode (find it with trend_categories), or both: keywords alone search all categories; keywords + categoryCode narrow a term to one industry (e.g. "jaguar" in Autos); categoryCode alone returns interest in the WHOLE category, no keyword needed. related:true adds the top + rising related topics and queries (at most 1 keyword) — the category-only form surfaces what is breaking out across a market. Returns the time series plus a rising/falling/flat read. timeRange: past_7_days | past_30_days | past_90_days | past_12_months (default) | past_5_years | 2004_present. type: web (default) | news | youtube | images | froogle. Default location: United States.',
968
896
  inputSchema: {
969
- keywords: z.array(z.string()).min(1).max(5),
897
+ keywords: z.array(z.string()).max(5).optional(),
898
+ categoryCode: z.number().int().min(1).optional(),
899
+ related: z.boolean().optional(),
970
900
  location: z.union([z.string(), z.number()]).optional(),
971
901
  languageCode: z.string().optional(),
972
902
  timeRange: z.enum(['past_7_days', 'past_30_days', 'past_90_days', 'past_12_months', 'past_5_years', '2004_present']).optional(),
973
903
  type: z.enum(['web', 'news', 'youtube', 'images', 'froogle']).optional(),
974
904
  },
975
- }, async ({ keywords, location, languageCode, timeRange, type }) => {
905
+ }, async ({ keywords, categoryCode, related, location, languageCode, timeRange, type }) => {
976
906
  const client = requireDfs(dfs);
977
- const r = await client.googleTrends(keywords, location, languageCode, { ...(timeRange ? { timeRange } : {}), ...(type ? { type } : {}) });
907
+ const r = await client.googleTrends(keywords ?? [], location, languageCode, {
908
+ ...(timeRange ? { timeRange } : {}), ...(type ? { type } : {}),
909
+ ...(categoryCode ? { categoryCode } : {}), ...(related ? { related } : {}),
910
+ });
978
911
  const summarise = (k) => {
979
912
  const vals = r.series.map(s => s.values[k]).filter((v) => typeof v === 'number');
980
913
  if (!vals.length)
@@ -986,9 +919,31 @@ export function createServer() {
986
919
  return `${k}: ${dir} (${Math.round(first)}→${Math.round(last)}, peak ${Math.max(...vals)})`;
987
920
  };
988
921
  const txt = r.keywords.map(summarise).join('\n');
922
+ const fmt = (xs, rising) => xs.slice(0, 10).map(x => ` ${'title' in x ? x.title : x.query} (${x.value == null ? '?' : rising ? `+${x.value}%` : x.value})`).join('\n') || ' none';
923
+ const relatedTxt = r.queries && r.topics
924
+ ? `\n\nRising queries:\n${fmt(r.queries.rising, true)}\nTop queries:\n${fmt(r.queries.top, false)}\nRising topics:\n${fmt(r.topics.rising, true)}`
925
+ : '';
926
+ const scope = r.categoryCode ? ` in category ${r.categoryCode}` : '';
927
+ return {
928
+ content: [{ type: 'text', text: `Trend${scope}, ${r.series.length} points${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}:\n${txt}${relatedTxt}` }],
929
+ structuredContent: { keywords: r.keywords, categoryCode: r.categoryCode, series: r.series, topics: r.topics, queries: r.queries, cached: r.cached, cost: r.cost },
930
+ };
931
+ });
932
+ server.registerTool('trend_categories', {
933
+ title: 'Google Trends category codes (DataForSEO)',
934
+ description: '[Free | Use before topic_trend with categoryCode] Search the Google Trends category tree (~1,400 categories) by name and get the codes topic_trend takes, with each match\'s parent so you can pick the right level. Omit query to list the top-level categories.',
935
+ inputSchema: { query: z.string().optional(), limit: z.number().int().min(1).max(100).optional() },
936
+ }, async ({ query, limit }) => {
937
+ const client = requireDfs(dfs);
938
+ const cats = await client.googleTrendsCategories();
939
+ const byCode = new Map(cats.map(c => [c.code, c]));
940
+ const q = query?.trim().toLowerCase();
941
+ const hits = (q ? cats.filter(c => c.name.toLowerCase().includes(q)) : cats.filter(c => c.parent === 0)).slice(0, limit ?? 25);
942
+ const withParent = hits.map(c => ({ ...c, parentName: c.parent ? byCode.get(c.parent)?.name ?? null : null }));
943
+ const lines = withParent.map(c => `${c.code} ${c.parentName ? `${c.parentName} > ` : ''}${c.name}`).join('\n');
989
944
  return {
990
- content: [{ type: 'text', text: `Trend, ${r.series.length} points${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}:\n${txt}` }],
991
- structuredContent: { keywords: r.keywords, series: r.series, cached: r.cached, cost: r.cost },
945
+ content: [{ type: 'text', text: `${withParent.length} of ${cats.length} categories${q ? ` matching "${query}"` : ' (top level)'}:\n${lines || 'none'}\n\nPass a code to topic_trend as categoryCode.` }],
946
+ structuredContent: { categories: withParent, total: cats.length },
992
947
  };
993
948
  });
994
949
  server.registerTool('suggest_pages', {
@@ -1010,8 +965,6 @@ export function createServer() {
1010
965
  db.close();
1011
966
  }
1012
967
  });
1013
- // content_opportunities — the content marketer's report: everything the stored data
1014
- // says about what to WRITE, REFRESH and REWRITE, in one free composition.
1015
968
  server.registerTool('content_opportunities', {
1016
969
  title: 'Content opportunity report (write / refresh / rewrite)',
1017
970
  description: 'The content marketer\'s report, composed entirely from stored data - NO paid calls. Four sections: WRITE NEXT (new pages proposed from queries you already earn impressions for but have no winning page - suggest_pages), REFRESH NOW (pages that lost 20%+ of their clicks vs the prior period - content decay), REWRITE SNIPPETS (page-1 rankings earning far below expected CTR - title/meta rewrites, the fastest wins), and STRENGTHEN (keyword clusters where you rank 4-20 - one push from the money positions). Every line traces to real Search Console data. Chain into draft_content for a brief, or keyword_volume to size a cluster against the market.',
@@ -1063,7 +1016,6 @@ export function createServer() {
1063
1016
  },
1064
1017
  };
1065
1018
  });
1066
- // ── Content recon (recon_targets) — the data-intensive "why are we losing, what to do" mission ──
1067
1019
  server.registerTool('recon_targets', {
1068
1020
  title: 'Content recon: why a page is losing, and what to do about it',
1069
1021
  description: '[Paid: DataForSEO SERP per page (~$0.004 each, plus a small refundable surcharge for loading async AI Overviews), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced, depth 20 organic), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, competitive-gap, or page-cannot-rank (the crawl says the URL is noindex/canonicalised away, so its GSC history is legacy and the SERP read belongs to another page). Honesty rules baked in: every GSC figure carries its 28-day window; organicRank:null means "absent from the top 20 ORGANIC results", never a position; and an AI Overview whose citations could not be resolved reports aioCitesUs:null (UNKNOWN) rather than "not cited". To-dos are prioritised by OPPORTUNITY (impressions x the CTR gap between where you rank and a realistic target, damped by the verdict), not by raw impressions, and cannibalisation is counted across the whole query cluster. Set scrapeCompetitors:true to also pull the top competitors as content, routed by host: YouTube/video → Supadata transcript (SUPADATA_API_KEY), Reddit → its .json, other pages → Firecrawl (FIRECRAWL_API_KEY) with a free HTTP fallback; competitorLimit (default 5) caps how many are fetched and everything above the cap is listed as skipped. Cloudflare-challenge sites (e.g. PCMag) still can\'t be fetched from a server and come back as a per-URL error — the SERP still tells you they rank; if one matters, ask the user to paste its copy or supply a text file and diff that in. Transcribe the videos (usually what wins these SERPs) and use the reachable pages. Then write findings back with save_recon_todo; track with recon_todos. **Async job:** returns a jobId immediately - poll check_sync_status; the finished job carries the per-page verdicts, to-dos and a summary. (Each page is persisted to the ledger as it completes, so recon_todos shows results even mid-run.)',
@@ -1078,10 +1030,8 @@ export function createServer() {
1078
1030
  crawlAs: z.enum(['browser', 'googlebot']).optional().describe('UA for the free HTTP fetch: browser (default, mimics a visit from Google — gets Reddit + mid-tier) or googlebot'),
1079
1031
  },
1080
1032
  }, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, competitorLimit, crawlAs }) => {
1081
- const client = requireDfs(dfs); // fail fast if no DataForSEO creds, before starting the job
1033
+ const client = requireDfs(dfs);
1082
1034
  const count = urls?.length ?? limit ?? 5;
1083
- // Async job: N live page-fetches + N serialised SERP calls exceed the ~60s MCP ceiling
1084
- // past a handful of pages, so return a jobId and poll (like refresh_property).
1085
1035
  const jobId = jobs.start('recon', async (update, signal) => {
1086
1036
  const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1087
1037
  try {
@@ -1092,8 +1042,6 @@ export function createServer() {
1092
1042
  const ownDomain = dfsHost(siteUrl);
1093
1043
  const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
1094
1044
  const SERP_DEPTH = 20;
1095
- // Every GSC number below is this window and only this window. Undeclared, a 28-day figure
1096
- // read against an all-time baseline looks exactly like a decline.
1097
1045
  const windowStart = db.db.prepare(`SELECT date(?, '-27 days') d`).get(fresh.effectiveMax);
1098
1046
  const gscWindow = { start: windowStart.d, end: fresh.effectiveMax, days: 28, metric: 'Search Console, last 28 days of synced data (not all-time)' };
1099
1047
  let targets;
@@ -1101,7 +1049,7 @@ export function createServer() {
1101
1049
  targets = [];
1102
1050
  for (const u of urls) {
1103
1051
  const key = urlKey(u, { hostForm });
1104
- const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
1052
+ const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
1105
1053
  FROM search_analytics WHERE page_key=? AND query IS NOT NULL AND date > date(?, '-28 days') AND date <= ? GROUP BY query ORDER BY imp DESC LIMIT 1`)
1106
1054
  .get(key, fresh.effectiveMax, fresh.effectiveMax);
1107
1055
  if (row?.query)
@@ -1124,12 +1072,10 @@ export function createServer() {
1124
1072
  try {
1125
1073
  own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
1126
1074
  }
1127
- catch { /* page unreachable — classify on SERP alone */ }
1075
+ catch { }
1128
1076
  const serpResp = await client.serpOrganic(t.query, location, 'en', SERP_DEPTH, { loadAsyncAiOverview: true });
1129
1077
  cost += serpResp.cost;
1130
1078
  const serp = parseSerpForRecon(serpResp, ownDomain, SERP_DEPTH);
1131
- // The crawl row overrides the SERP/GSC read: GSC keeps reporting impressions for URLs
1132
- // that have since been canonicalised away or set noindex.
1133
1079
  const state = pageState(db.db, t.urlKey);
1134
1080
  const serpVerdict = reconVerdict(serp);
1135
1081
  const verdict = crawlRealityOverride(serpVerdict, state) ?? serpVerdict;
@@ -1148,10 +1094,6 @@ export function createServer() {
1148
1094
  let competitorContent = undefined;
1149
1095
  let competitorFetch = undefined;
1150
1096
  if (scrapeCompetitors && (firecrawl || supadata)) {
1151
- // Build a routed candidate set: organic-above + AI-Overview references + one ranking
1152
- // video, deduped, our own domain removed. Each URL routes by HOST (YouTube→supadata,
1153
- // Reddit→.json, else→firecrawl). What we don't fetch is REPORTED, not silently dropped
1154
- // — an undeclared cap of 3 is what made this look like it returned nothing.
1155
1097
  const seen = new Set();
1156
1098
  const candidates = [];
1157
1099
  const add = (url, from) => {
@@ -1164,8 +1106,6 @@ export function createServer() {
1164
1106
  serp.aioReferences.slice(0, 6).forEach(r => add(r.url, 'aio-reference'));
1165
1107
  serp.videoItems.slice(0, 2).forEach(v => add(v.url, 'video-pack'));
1166
1108
  const lim = competitorLimit ?? 5;
1167
- // Organic-above first (the editorial pages actually worth diffing), but keep one slot
1168
- // for a ranking video — video is usually what wins these SERPs.
1169
1109
  const video = candidates.find(c => c.route === 'video');
1170
1110
  const rest = candidates.filter(c => c !== video);
1171
1111
  const reserve = video && lim >= 3 ? 1 : 0;
@@ -1235,12 +1175,6 @@ export function createServer() {
1235
1175
  return note;
1236
1176
  })() +
1237
1177
  `\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
1238
- // The poll (check_sync_status) injects this result into the model's context, which the
1239
- // host caps (~25k tokens). The full per-page detail (heading lists, competitor URL arrays,
1240
- // full to-do objects, cannibalisation clusters) is already persisted to the ledger and read
1241
- // back via recon_todos, so the job result returns the readable `summary` plus a SLIM per-page
1242
- // view — verdict, ranks, opportunity — and keeps only competitorContent (the scraped research
1243
- // payload, which lives nowhere else). This keeps a 10-page poll well under the cap.
1244
1178
  const slimTargets = results.map(r => ({
1245
1179
  urlKey: r.urlKey, query: r.query, window: r.window,
1246
1180
  impressions: r.impressions, gscPosition: r.gscPosition, priorPosition: r.priorPosition, slipped: r.slipped,
@@ -1262,8 +1196,6 @@ export function createServer() {
1262
1196
  structuredContent: { jobId, status: 'running', siteUrl },
1263
1197
  };
1264
1198
  });
1265
- // save_recon_todo — the research session writes its content-gap / originality findings back
1266
- // into the ledger against a page (source: 'research').
1267
1199
  server.registerTool('save_recon_todo', {
1268
1200
  title: 'Save content-recon to-dos (research writeback)',
1269
1201
  description: 'Write content-recon findings back into the trackable ledger for a page — the gaps and originality the research session found by diffing competitors (firecrawl) and videos (supadata) against our content. Each to-do is an action with a type (content-gap / originality / schema / freshness / format), rationale and evidence. Snapshots the page baseline so the fix\'s effect on rank/AIO-citation is measurable later. Run recon_targets first (it classifies the page and seeds the deterministic to-dos); this adds the judgement ones. Track everything with recon_todos.',
@@ -1283,11 +1215,7 @@ export function createServer() {
1283
1215
  try {
1284
1216
  const key = urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' });
1285
1217
  const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position, gsc_impressions, verdict FROM recon_page WHERE url_key=?`).get(key);
1286
- // aio_cites_us is NULL when the AI Overview never resolved — keep that as unknown rather
1287
- // than coercing it to false, or the outcome diff reads "no → yes" off a missing fetch.
1288
1218
  const baseline = page ? { organicRank: page.organic_rank, aioCitesUs: page.aio_cites_us == null ? null : !!page.aio_cites_us, gscPosition: page.gsc_position, at: new Date().toISOString().slice(0, 10) } : {};
1289
- // Default priority to the PAGE's opportunity so research to-dos sort on the same scale as
1290
- // the deterministic ones (a default of 0 buried every judgement finding at the bottom).
1291
1219
  const basis = page ? opportunityBasis(page.gsc_impressions ?? 0, page.organic_rank, page.gsc_position ?? 20, page.verdict) : null;
1292
1220
  const fallbackPriority = basis ? Math.max(1, Math.round(basis.opportunityClicks * basis.verdictFactor)) : 0;
1293
1221
  const drafts = todos.map(t => ({ action: t.action, type: t.type ?? 'content-gap', rationale: t.rationale ?? '', evidence: t.evidence ?? {}, priority: t.priority ?? fallbackPriority }));
@@ -1298,8 +1226,6 @@ export function createServer() {
1298
1226
  db.close();
1299
1227
  }
1300
1228
  });
1301
- // recon_todos — list, track and annotate the ledger. No id → list (optionally filtered);
1302
- // id → update status and/or append a dated annotation, and optionally re-measure the outcome.
1303
1229
  server.registerTool('recon_todos', {
1304
1230
  title: 'List, track and annotate content-recon to-dos',
1305
1231
  description: 'The content-recon to-do board. With no id: list to-dos (optionally filter by page or status), grouped by page with each page\'s verdict — the pick-a-page-to-work-on surface, and the hand-off to content-machine. With id: update one to-do — set status (open → researching → drafted → shipped → dismissed) and/or append a dated annotation note (your own observations, the history). On status:shipped with remeasure:true it re-fetches the SERP and records the outcome, so you can see whether the fix moved you from AIO-uncited to cited, or up the organic ranks.',
@@ -1325,7 +1251,6 @@ export function createServer() {
1325
1251
  let outcome = row.outcome;
1326
1252
  let outcomeNote = '';
1327
1253
  if (status === 'shipped' && remeasure && dfs) {
1328
- // Same depth + async-AIO handling as recon_targets, so before/after are comparable.
1329
1254
  const serpResp = await dfs.serpOrganic(row.query, location, 'en', 20, { loadAsyncAiOverview: true });
1330
1255
  const serp = parseSerpForRecon(serpResp, dfsHost(siteUrl), 20);
1331
1256
  const cited = (v) => (v == null ? 'unknown' : v ? 'yes' : 'no');
@@ -1349,7 +1274,7 @@ export function createServer() {
1349
1274
  where.push('status=?');
1350
1275
  args.push(status);
1351
1276
  }
1352
- const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
1277
+ const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
1353
1278
  FROM recon_todo ${where.length ? 'WHERE ' + where.join(' AND ') : ''} ORDER BY url_key, priority DESC`).all(...args);
1354
1279
  if (!rows.length)
1355
1280
  return { content: [{ type: 'text', text: `No recon to-dos${key ? ` for ${key}` : ''}${status ? ` with status ${status}` : ''}. Run recon_targets to generate some.` }], structuredContent: { todos: [] } };
@@ -1371,8 +1296,6 @@ export function createServer() {
1371
1296
  db.close();
1372
1297
  }
1373
1298
  });
1374
- // keyword_list — demand-first clustering ("list mode"): a keyword list becomes topics
1375
- // with own/weak/absent verdicts, clustered by the URL Google already answers them with.
1376
1299
  server.registerTool('keyword_list', {
1377
1300
  title: 'Cluster a keyword list into topics with own/weak/absent verdicts',
1378
1301
  description: 'Demand-first keyword clustering from stored data - NO paid calls. Give it a keyword list (or omit keywords to use your top 500 GSC queries) and it clusters them by the page Google ALREADY answers them with (two keywords that rank via the same URL belong together - the strongest clustering signal, free from your own GSC data), then groups non-ranking keywords lexically. Each cluster gets a deterministic verdict: OWN (best position <=3), WEAK (4-20), ABSENT (no ranking page), plus the ranking URL, summed 90-day impressions and clicks. The keyword-research workhorse: paste a client keyword list, get the topic map and where you stand. Chain with keyword_volume for market volumes on the interesting clusters, or draft_content for the absent ones.',
@@ -1421,7 +1344,7 @@ export function createServer() {
1421
1344
  if (siteUrl) {
1422
1345
  const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1423
1346
  try {
1424
- const up = db.db.prepare(`INSERT INTO keyword_intent (keyword,intent,probability,fetched_at) VALUES (?,?,?,datetime('now'))
1347
+ const up = db.db.prepare(`INSERT INTO keyword_intent (keyword,intent,probability,fetched_at) VALUES (?,?,?,datetime('now'))
1425
1348
  ON CONFLICT(keyword) DO UPDATE SET intent=excluded.intent, probability=excluded.probability, fetched_at=datetime('now')`);
1426
1349
  db.db.transaction(() => { for (const it of items)
1427
1350
  if (it.keyword && it.intent) {
@@ -1475,9 +1398,9 @@ export function createServer() {
1475
1398
  if (siteUrl && cats.performance) {
1476
1399
  const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1477
1400
  try {
1478
- db.db.prepare(`INSERT INTO page_cwv (url_key,url,for_mobile,performance,lcp_ms,cls,tbt_ms,fetched_at)
1479
- VALUES (?,?,?,?,?,?,?,datetime('now'))
1480
- ON CONFLICT(url_key) DO UPDATE SET url=excluded.url, for_mobile=excluded.for_mobile, performance=excluded.performance,
1401
+ db.db.prepare(`INSERT INTO page_cwv (url_key,url,for_mobile,performance,lcp_ms,cls,tbt_ms,fetched_at)
1402
+ VALUES (?,?,?,?,?,?,?,datetime('now'))
1403
+ ON CONFLICT(url_key) DO UPDATE SET url=excluded.url, for_mobile=excluded.for_mobile, performance=excluded.performance,
1481
1404
  lcp_ms=excluded.lcp_ms, cls=excluded.cls, tbt_ms=excluded.tbt_ms, fetched_at=datetime('now')`)
1482
1405
  .run(urlKey(url, { hostForm: hostFormForProperty(siteUrl) }), url, mobile ? 1 : 0, out.scores.performance, numeric('largest-contentful-paint'), numeric('cumulative-layout-shift'), numeric('total-blocking-time'));
1483
1406
  persisted = true;
@@ -1501,7 +1424,7 @@ export function createServer() {
1501
1424
  const cleaned = target.replace(/^sc-domain:/, '').replace(/^https?:\/\//, '').replace(/^www\./, '').replace(/\/+$/, '');
1502
1425
  const r = await client.competitorsDomain(cleaned, location, languageCode, limit ?? 20);
1503
1426
  const items = (r.tasks[0]?.result?.[0]?.items ?? [])
1504
- .filter((it) => it.domain && it.domain.replace(/^www\./, '') !== cleaned) // drop the target itself
1427
+ .filter((it) => it.domain && it.domain.replace(/^www\./, '') !== cleaned)
1505
1428
  .map((it) => ({
1506
1429
  domain: it.domain,
1507
1430
  intersections: it.intersections ?? null,
@@ -1529,7 +1452,7 @@ export function createServer() {
1529
1452
  const r = await client.pageIntersection(competitorUrls, excludePages ?? [], location, languageCode, limit ?? 100);
1530
1453
  const items = (r.tasks[0]?.result?.[0]?.items ?? [])
1531
1454
  .map((it) => {
1532
- const kd = it.keyword_data ?? {}; // actual shape: item.keyword_data.{keyword,keyword_info}
1455
+ const kd = it.keyword_data ?? {};
1533
1456
  return {
1534
1457
  keyword: kd.keyword ?? null,
1535
1458
  searchVolume: kd.keyword_info?.search_volume ?? null,
@@ -1540,16 +1463,13 @@ export function createServer() {
1540
1463
  })),
1541
1464
  };
1542
1465
  })
1543
- .sort((a, b) => (b.searchVolume ?? 0) - (a.searchVolume ?? 0)); // order_by unsupported on this endpoint
1466
+ .sort((a, b) => (b.searchVolume ?? 0) - (a.searchVolume ?? 0));
1544
1467
  return {
1545
1468
  content: [{ type: 'text', text: `${items.length} gap keywords${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
1546
1469
  structuredContent: { gaps: items, cached: r.cached, cost: r.cost },
1547
1470
  };
1548
1471
  });
1549
- // Strip any scheme/sc-domain:/path down to a bare host for Labs domain targets.
1550
- // Always strips leading www. — Labs treats www.example.com as a subdomain, not the domain.
1551
1472
  const dfsHost = (t) => t.replace(/^sc-domain:/, '').replace(/^https?:\/\//, '').replace(/\/.*$/, '').replace(/^www\./, '').trim();
1552
- // Cap a markdown table well under the ~40k model-facing ceiling: keep whole rows, note the rest.
1553
1473
  const capMdRows = (header, rows, footer = '', maxChars = 30000) => {
1554
1474
  let out = header;
1555
1475
  let used = 0;
@@ -1581,7 +1501,6 @@ export function createServer() {
1581
1501
  const r = await client.historicalRankOverview(cleaned, location, 'en', languageName);
1582
1502
  const items = r.tasks[0]?.result?.[0]?.items ?? [];
1583
1503
  const n = (v) => Number(v) || 0;
1584
- // Guard year/month — a malformed item would emit an "undefined-NaN" period row.
1585
1504
  const series = items
1586
1505
  .filter((it) => Number.isInteger(it?.year) && Number.isInteger(it?.month))
1587
1506
  .map((it) => {
@@ -1641,7 +1560,7 @@ export function createServer() {
1641
1560
  .map((it) => {
1642
1561
  const page = it.page_address ?? it.relative_url ?? null;
1643
1562
  if (!page)
1644
- return null; // skip malformed rows
1563
+ return null;
1645
1564
  const o = it.metrics?.organic ?? {};
1646
1565
  const top3 = n(o.pos_1) + n(o.pos_2_3);
1647
1566
  return {
@@ -1698,14 +1617,9 @@ export function createServer() {
1698
1617
  const folderPath = mode === 'folder'
1699
1618
  ? (folder.trim().startsWith('/') ? folder.trim() : '/' + folder.trim())
1700
1619
  : null;
1701
- // aioOnly: only rows where the target's ranked element is an AI Overview citation —
1702
- // expressed as a serp_item.type filter (item_types is invalid on this Labs endpoint).
1703
1620
  const clauses = [];
1704
1621
  if (folderPath)
1705
1622
  clauses.push(['ranked_serp_element.serp_item.relative_url', 'like', `${folderPath}%`]);
1706
- // Exposure, not citation: Labs ranked_keywords carries no citation fields (verified via
1707
- // available_filters) — this filters to keywords whose SERP CONTAINS an AI Overview.
1708
- // Per-keyword citation checking (are WE a reference?) is the SERP-advanced recipe.
1709
1623
  if (aioOnly)
1710
1624
  clauses.push(['keyword_data.serp_info.serp_item_types', 'has', 'ai_overview']);
1711
1625
  const filters = clauses.length === 0 ? undefined : clauses.length === 1 ? clauses : [clauses[0], 'and', clauses[1]];
@@ -1717,7 +1631,7 @@ export function createServer() {
1717
1631
  const kd = it.keyword_data ?? {};
1718
1632
  const serp = it.ranked_serp_element?.serp_item ?? {};
1719
1633
  if (!kd.keyword)
1720
- return null; // skip malformed rows
1634
+ return null;
1721
1635
  const serpFeatures = Array.isArray(kd.serp_info?.serp_item_types) ? kd.serp_info.serp_item_types : null;
1722
1636
  return {
1723
1637
  keyword: kd.keyword,
@@ -1725,7 +1639,6 @@ export function createServer() {
1725
1639
  searchVolume: kd.keyword_info?.search_volume ?? null,
1726
1640
  etv: serp.etv != null ? Math.round(Number(serp.etv) * 100) / 100 : null,
1727
1641
  url: serp.relative_url ?? serp.url ?? null,
1728
- // Fields we already pay for but previously dropped (plan/data-utilisation.md Part 2 §4):
1729
1642
  keywordDifficulty: kd.keyword_properties?.keyword_difficulty ?? null,
1730
1643
  intent: kd.search_intent_info?.main_intent ?? null,
1731
1644
  serpFeatures,
@@ -1744,7 +1657,6 @@ export function createServer() {
1744
1657
  }
1745
1658
  const totalCount = Number(result.total_count) || kws.length;
1746
1659
  const sumEtv = kws.reduce((s, k) => s + (k.etv ?? 0), 0);
1747
- // Optional columns — only where the response actually carries the field (older cache entries won't).
1748
1660
  const hasKd = kws.some(k => k.keywordDifficulty != null);
1749
1661
  const hasIntent = kws.some(k => k.intent != null);
1750
1662
  const hasFeatures = kws.some(k => (k.serpFeatures && k.serpFeatures.length) || k.isFeaturedSnippet);
@@ -1767,9 +1679,6 @@ export function createServer() {
1767
1679
  structuredContent: { target: dfsTarget, scope: mode, aioOnly: aioOnly ?? false, totalCount, rowsTotal: kws.length, keywords: kws.slice(0, 100), cached: r.cached, cost: r.cost },
1768
1680
  };
1769
1681
  });
1770
- // serp_features — the SERP-feature footprint: "how much of my market do AI Overviews,
1771
- // snippets and other features sit on, and do I already rank page 1 there?" One cached
1772
- // Labs pull, volume-weighted; persisted so the dashboard can chart it.
1773
1682
  server.registerTool('serp_features', {
1774
1683
  title: 'SERP-feature footprint (AI Overviews, snippets, PAA)',
1775
1684
  description: '[Paid: Labs, ONE cached call | Use for: AI-Overview / zero-click exposure with real numbers] How much of your keyword universe carries each SERP feature - AI Overviews, featured snippets, People Also Ask, shopping, video - weighted by search volume, and how much of that volume you already rank page 1 for. ONE DataForSEO Labs ranked_keywords pull (top-volume sample, cached 20 days, never per-keyword SERP loops). Answers "how exposed are we to AI Overviews / zero-click?" with real numbers. Deterministic: feature PRESENCE (from the Labs index). Judgement proxy: page-1 rank stands in for feature ownership - true ownership needs per-keyword SERP calls (see ranked_keywords aioOnly + the cookbook). Persists to the property DB so the dashboard charts it.',
@@ -1814,8 +1723,6 @@ export function createServer() {
1814
1723
  structuredContent: fp,
1815
1724
  };
1816
1725
  });
1817
- // market_sizing — Market Sizing and Prioritisation: the organic market read that opens
1818
- // an engagement. Top-down Labs pulls only (client + <=4 rivals), cached, ~$0.65 worst case.
1819
1726
  server.registerTool('market_sizing', {
1820
1727
  title: 'Market Sizing and Prioritisation (share of voice vs competitors)',
1821
1728
  description: '[Paid: Labs, one cached call per domain (<=5) | Use for: sizing the organic market and who owns it] Build the organic market map: your domain plus up to 4 named competitors, ONE cached Labs ranked_keywords pull each, unioned into a keyword universe. Returns total monthly demand (deduplicated search volume), each domain\'s share of voice (ETV share) overall and per topic cluster, and the leader per cluster - the "here is the market, here is who owns it, here is where to attack" table that opens an engagement. Deterministic: the competitor set and ranked keywords (Labs index). Judgement: ETV is DataForSEO\'s CTR-curve traffic estimate - the SoV percentages inherit that. Persists to the property DB for the dashboard chart. Get the competitor set from competitors_domain first if unsure.',
@@ -1833,7 +1740,7 @@ export function createServer() {
1833
1740
  const inputs = [];
1834
1741
  let cost = 0;
1835
1742
  let cachedAll = true;
1836
- for (const domain of domains) { // sequential - the client serialises anyway
1743
+ for (const domain of domains) {
1837
1744
  const r = await client.rankedKeywords(domain, location, languageCode ?? 'en', limitPerDomain ?? 1000, 'keyword_data.keyword_info.search_volume,desc');
1838
1745
  cost += r.cost;
1839
1746
  cachedAll = cachedAll && r.cached;
@@ -1899,8 +1806,6 @@ export function createServer() {
1899
1806
  const ourHost = dfsHost(siteUrl);
1900
1807
  let totalCost = 0, liveCalls = 0, cachedCalls = 0;
1901
1808
  const tally = (r) => { totalCost += r.cost; r.cached ? cachedCalls++ : liveCalls++; };
1902
- // Competitors: explicit (≤3) or derived top-2 by keyword overlap. Total Labs
1903
- // budget stays ≤4 calls: at most 1 discovery + at most 3 footprint pulls.
1904
1809
  let comps = [...new Set((competitors ?? []).map(dfsHost).filter(c => c && c !== ourHost))].slice(0, 3);
1905
1810
  let derived = false;
1906
1811
  if (!comps.length) {
@@ -1919,7 +1824,6 @@ export function createServer() {
1919
1824
  };
1920
1825
  }
1921
1826
  }
1922
- // One ranked_keywords pull per competitor (top 1000 rows by ETV — their money keywords).
1923
1827
  const rows = [];
1924
1828
  const perCompetitor = {};
1925
1829
  for (const c of comps) {
@@ -1968,8 +1872,6 @@ export function createServer() {
1968
1872
  `${res.competitorKeywords} distinct competitor keywords → ${res.afterSubtraction} survive subtraction of your footprint (${fmtVol(res.ourQueryCount)} GSC queries + page titles/H1s) → top ${res.clusters.length} topic clusters by volume × competitor coverage.\n\n` +
1969
1873
  sections.join('\n\n') +
1970
1874
  `\n\n_${costLine} Competitor rows: ${comps.map(c => `${c} ${perCompetitor[c] ?? 0}`).join(', ')}._`;
1971
- // structuredContent: keyword lists capped at 10 per cluster with explicit totals
1972
- // (host ~60k model-facing ceiling) — the counts always state the true size.
1973
1875
  const scClusters = res.clusters.map(c => ({
1974
1876
  ...c,
1975
1877
  keywords: c.keywords.slice(0, 10),
@@ -1989,16 +1891,12 @@ export function createServer() {
1989
1891
  db.close();
1990
1892
  }
1991
1893
  });
1992
- // ── Dashboard (MCP App UI — houtini design + ECharts) ───────────────────
1993
1894
  registerAppTool(server, 'get_dashboard', {
1994
1895
  title: 'SEO dashboard',
1995
1896
  description: 'Interactive dashboard for a property: summary metrics, rank & clicks over time, and top-keyword performance (click a keyword for related terms). Needs synced GSC data — run refresh_property first.',
1996
1897
  inputSchema: { siteUrl: z.string() },
1997
1898
  _meta: { ui: { resourceUri: DASHBOARD_URI } },
1998
1899
  }, async ({ siteUrl }) => {
1999
- // Return only a TINY model-facing result + the siteUrl; the widget fetches the full
2000
- // (large) dataset itself via the app-only get_dashboard_data tool, which keeps the
2001
- // big payload OUT of the model's context/token limit (per the MCP Apps large-data pattern).
2002
1900
  const data = getDashboardData(dataDir(), siteUrl);
2003
1901
  if (data.empty) {
2004
1902
  return { content: [{ type: 'text', text: `No synced data for ${siteUrl} yet — run refresh_property.` }], structuredContent: { siteUrl, empty: true } };
@@ -2008,9 +1906,6 @@ export function createServer() {
2008
1906
  `${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.` + browserLink(siteUrl);
2009
1907
  return { content: [{ type: 'text', text: summary }], structuredContent: { siteUrl } };
2010
1908
  });
2011
- // App-only data tool: the dashboard widget calls this via app.callServerTool to fetch its
2012
- // full dataset. visibility:['app'] hides it from the model; results route to the iframe,
2013
- // bypassing the model token cap that a large model-facing result would hit.
2014
1909
  registerAppTool(server, 'get_dashboard_data', {
2015
1910
  title: 'Dashboard data (internal)',
2016
1911
  description: 'Full dashboard dataset for the UI widget. App-only — not for direct use.',
@@ -2020,9 +1915,6 @@ export function createServer() {
2020
1915
  const data = { ...getDashboardData(dataDir(), siteUrl), apiKeys: apiKeysStatus() };
2021
1916
  return { content: [{ type: 'text', text: 'ok' }], structuredContent: data };
2022
1917
  });
2023
- // export_report — the dependable deliverable: a self-contained interactive dashboard
2024
- // HTML (data inlined) the user opens in any browser / emails to a client. Works
2025
- // regardless of whether the host renders MCP-App widgets inline.
2026
1918
  server.registerTool('export_report', {
2027
1919
  title: 'Export a shareable dashboard report (HTML)',
2028
1920
  description: 'Write a self-contained, interactive dashboard HTML for a property (all data + charts inlined) to the reports folder, and return the file path. Open it in any browser or send it to a client — no server, no MCP-App host support needed. Run refresh_property (+ run_audit for findings) first.',
@@ -2033,10 +1925,8 @@ export function createServer() {
2033
1925
  return { content: [{ type: 'text', text: `No synced data for ${siteUrl} — run refresh_property first.` }], structuredContent: { error: 'empty', siteUrl } };
2034
1926
  }
2035
1927
  const tpl = readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8');
2036
- const json = JSON.stringify(data).replace(/</g, '\\u003c'); // prevent </script> breakout
1928
+ const json = JSON.stringify(data).replace(/</g, '\\u003c');
2037
1929
  const inject = `<script>window.__DASH_FIXTURE__=${json};window.__DASH_THEME__=${JSON.stringify(theme ?? 'light')};</script>`;
2038
- // Replacement FUNCTION, not string — crawl data containing $& / $' would otherwise
2039
- // be interpreted as String.replace substitution patterns and corrupt the report.
2040
1930
  const html = tpl.replace(/<head([^>]*)>/i, (_m, attrs) => `<head${attrs}>${inject}`);
2041
1931
  const dir = path.join(dataDir(), 'reports');
2042
1932
  mkdirSync(dir, { recursive: true });
@@ -2047,8 +1937,6 @@ export function createServer() {
2047
1937
  structuredContent: { path: file, siteUrl, findings: data.findings?.total ?? 0, bytes: html.length },
2048
1938
  };
2049
1939
  });
2050
- // serve_dashboard — the local webserver delivery surface: the full dashboard in a real
2051
- // browser tab (live data, property switcher, native downloads), no MCP-App host needed.
2052
1940
  server.registerTool('serve_dashboard', {
2053
1941
  title: 'Serve the dashboard on a local webserver',
2054
1942
  description: 'Start a localhost-only webserver and return a URL that opens the full interactive dashboard in your browser — live data straight from the local database (always current, unlike export_report snapshots), a property switcher, working CSV downloads, and no host widget limits. The server stays up while the MCP server runs; call again with stop=true to shut it down. Localhost only — nothing is exposed to the network.',
@@ -2072,8 +1960,6 @@ export function createServer() {
2072
1960
  structuredContent: { url: open, base: url },
2073
1961
  };
2074
1962
  });
2075
- // pull_backlinks — on-demand backlink profile (DataForSEO, paid + 20-day cached). Powers
2076
- // backlinks-to-404 (the big quick win), top-linked pages, and true-orphan detection.
2077
1963
  server.registerTool('pull_backlinks', {
2078
1964
  title: 'Pull backlink profile (DataForSEO)',
2079
1965
  description: '[Paid: Backlinks subscription (separate - 40204 = not activated), cached 20d | Use for: authority + dead-backlink recovery] Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
@@ -2086,9 +1972,6 @@ export function createServer() {
2086
1972
  structuredContent: { jobId, status: 'running', siteUrl },
2087
1973
  };
2088
1974
  });
2089
- // link_intersect — "what links do our competitors have that we don't?". One DataForSEO
2090
- // backlinks/domain_intersection call (paid, ~$0.024, 20-day cached), aggregated + prioritised
2091
- // client-side. Default sort = the link-builder's: followed links first, then domain trust.
2092
1975
  server.registerTool('link_intersect', {
2093
1976
  title: 'Link intersect — links your competitors have that you don’t (DataForSEO + Majestic)',
2094
1977
  description: '[Paid: Backlinks subscription (separate - 40204 = not activated), one cached call ~$0.024 | Use for: prospect list for link outreach] Answers "what links do our competitors have that we don’t?" - and for a SINGLE company, "what links does company X have that we don’t?" (pass competitors:["companyx.com"]). One DataForSEO domain_intersection call over the target set (excluding your domain), aggregated per prospect domain: how many of the targets it links to, its DataForSEO domain trust (rank 0-1000), worst spam score, whether the link is followed, the anchor/link-type mix. Default sort is the link-builder’s view - FOLLOWED links first, then domain trust - with spam filtered out. When MAJESTIC_API_KEY is set, prospects are enriched with Majestic Trust Flow + Topical Trust Flow and RE-SORTED by Trust Flow (the directory-killer: a DataForSEO rank-227 domain is often Trust Flow 0). Results persist to link_prospects. Pass targets explicitly (max 20) or let it derive the top few via competitors_domain. Grain: one row per prospect domain. Join: domain.',
@@ -2107,8 +1990,6 @@ export function createServer() {
2107
1990
  }, async ({ siteUrl, competitors, location, poolLimit, minIntersections, maxSpamScore, dofollowOnly, topN, sort, enrichLimit }) => {
2108
1991
  const li = requireDfs(linkIntersect);
2109
1992
  const ownHost = dfsHost(siteUrl);
2110
- // Prior-run awareness: if this property already has link_prospects, frame the run as an
2111
- // update (the previous set ages as competitors keep earning links). Read-only, cheap.
2112
1993
  let priorNote = '';
2113
1994
  {
2114
1995
  const d = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
@@ -2119,12 +2000,11 @@ export function createServer() {
2119
2000
  priorNote = ` _(updating a prior intersect of ${prior.n} prospects from ${prior.t.slice(0, 10)}, ${ageDays}d ago)_`;
2120
2001
  }
2121
2002
  }
2122
- catch { /* pre-link_intersect DB */ }
2003
+ catch { }
2123
2004
  finally {
2124
2005
  d.close();
2125
2006
  }
2126
2007
  }
2127
- // Competitors: explicit (deduped, minus self) or derived top few by keyword overlap.
2128
2008
  let comps = [...new Set((competitors ?? []).map(dfsHost).filter(c => c && c !== ownHost))];
2129
2009
  let derived = false;
2130
2010
  let deriveCost = 0;
@@ -2161,7 +2041,6 @@ export function createServer() {
2161
2041
  ...(enrichLimit != null ? { enrichLimit } : {}),
2162
2042
  }, majestic);
2163
2043
  const enriched = result.majesticEnriched > 0;
2164
- // With Majestic: show Trust Flow (0-100) + the domain's top topic. Without: DataForSEO domain rank.
2165
2044
  const trustHead = enriched ? 'Trust Flow' : 'Domain trust';
2166
2045
  const rows = result.prospects.map(p => {
2167
2046
  const trust = enriched ? (p.trustFlow ?? '–') : (p.domainTrust ?? '–');
@@ -2193,7 +2072,6 @@ export function createServer() {
2193
2072
  structuredContent: { jobId, status: 'running', siteUrl },
2194
2073
  };
2195
2074
  });
2196
- // ── Static reference resources (same content the tools return — one source of truth) ──
2197
2075
  server.registerResource('checks-reference', 'seo-audit://checks-reference', {
2198
2076
  title: 'Check registry reference',
2199
2077
  description: 'The full audit check catalogue (the list_checks data) rendered as markdown: every check with category, severity, labels, certainty, fix type and its one-line fix.',