@houtini/seo-audit-console 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/LICENSE +92 -0
  2. package/README.md +211 -0
  3. package/dist/audit/checks.d.ts +35 -0
  4. package/dist/audit/checks.d.ts.map +1 -0
  5. package/dist/audit/checks.js +1475 -0
  6. package/dist/audit/checks.js.map +1 -0
  7. package/dist/audit/drift.d.ts +40 -0
  8. package/dist/audit/drift.d.ts.map +1 -0
  9. package/dist/audit/drift.js +148 -0
  10. package/dist/audit/drift.js.map +1 -0
  11. package/dist/audit/engine.d.ts +33 -0
  12. package/dist/audit/engine.d.ts.map +1 -0
  13. package/dist/audit/engine.js +186 -0
  14. package/dist/audit/engine.js.map +1 -0
  15. package/dist/audit/opportunities.d.ts +23 -0
  16. package/dist/audit/opportunities.d.ts.map +1 -0
  17. package/dist/audit/opportunities.js +149 -0
  18. package/dist/audit/opportunities.js.map +1 -0
  19. package/dist/audit/report.d.ts +11 -0
  20. package/dist/audit/report.d.ts.map +1 -0
  21. package/dist/audit/report.js +59 -0
  22. package/dist/audit/report.js.map +1 -0
  23. package/dist/audit/schema-validate.d.ts +35 -0
  24. package/dist/audit/schema-validate.d.ts.map +1 -0
  25. package/dist/audit/schema-validate.js +293 -0
  26. package/dist/audit/schema-validate.js.map +1 -0
  27. package/dist/audit/templates.d.ts +43 -0
  28. package/dist/audit/templates.d.ts.map +1 -0
  29. package/dist/audit/templates.js +129 -0
  30. package/dist/audit/templates.js.map +1 -0
  31. package/dist/audit/topicGaps.d.ts +42 -0
  32. package/dist/audit/topicGaps.d.ts.map +1 -0
  33. package/dist/audit/topicGaps.js +181 -0
  34. package/dist/audit/topicGaps.js.map +1 -0
  35. package/dist/core/AuditDatabase.d.ts +33 -0
  36. package/dist/core/AuditDatabase.d.ts.map +1 -0
  37. package/dist/core/AuditDatabase.js +481 -0
  38. package/dist/core/AuditDatabase.js.map +1 -0
  39. package/dist/core/Backlinks.d.ts +33 -0
  40. package/dist/core/Backlinks.d.ts.map +1 -0
  41. package/dist/core/Backlinks.js +110 -0
  42. package/dist/core/Backlinks.js.map +1 -0
  43. package/dist/core/Crawler.d.ts +23 -0
  44. package/dist/core/Crawler.d.ts.map +1 -0
  45. package/dist/core/Crawler.js +588 -0
  46. package/dist/core/Crawler.js.map +1 -0
  47. package/dist/core/DataForSeoClient.d.ts +86 -0
  48. package/dist/core/DataForSeoClient.d.ts.map +1 -0
  49. package/dist/core/DataForSeoClient.js +232 -0
  50. package/dist/core/DataForSeoClient.js.map +1 -0
  51. package/dist/core/Entities.d.ts +23 -0
  52. package/dist/core/Entities.d.ts.map +1 -0
  53. package/dist/core/Entities.js +62 -0
  54. package/dist/core/Entities.js.map +1 -0
  55. package/dist/core/GscClient.d.ts +22 -0
  56. package/dist/core/GscClient.d.ts.map +1 -0
  57. package/dist/core/GscClient.js +93 -0
  58. package/dist/core/GscClient.js.map +1 -0
  59. package/dist/core/GscSync.d.ts +20 -0
  60. package/dist/core/GscSync.d.ts.map +1 -0
  61. package/dist/core/GscSync.js +133 -0
  62. package/dist/core/GscSync.js.map +1 -0
  63. package/dist/core/JobManager.d.ts +30 -0
  64. package/dist/core/JobManager.d.ts.map +1 -0
  65. package/dist/core/JobManager.js +68 -0
  66. package/dist/core/JobManager.js.map +1 -0
  67. package/dist/core/RankTracker.d.ts +25 -0
  68. package/dist/core/RankTracker.d.ts.map +1 -0
  69. package/dist/core/RankTracker.js +78 -0
  70. package/dist/core/RankTracker.js.map +1 -0
  71. package/dist/core/Refresh.d.ts +32 -0
  72. package/dist/core/Refresh.d.ts.map +1 -0
  73. package/dist/core/Refresh.js +73 -0
  74. package/dist/core/Refresh.js.map +1 -0
  75. package/dist/core/UrlInspector.d.ts +22 -0
  76. package/dist/core/UrlInspector.d.ts.map +1 -0
  77. package/dist/core/UrlInspector.js +92 -0
  78. package/dist/core/UrlInspector.js.map +1 -0
  79. package/dist/core/WikidataClient.d.ts +19 -0
  80. package/dist/core/WikidataClient.d.ts.map +1 -0
  81. package/dist/core/WikidataClient.js +59 -0
  82. package/dist/core/WikidataClient.js.map +1 -0
  83. package/dist/core/agentReadiness.d.ts +34 -0
  84. package/dist/core/agentReadiness.d.ts.map +1 -0
  85. package/dist/core/agentReadiness.js +119 -0
  86. package/dist/core/agentReadiness.js.map +1 -0
  87. package/dist/core/ctrModel.d.ts +2 -0
  88. package/dist/core/ctrModel.d.ts.map +1 -0
  89. package/dist/core/ctrModel.js +6 -0
  90. package/dist/core/ctrModel.js.map +1 -0
  91. package/dist/core/dashboardData.d.ts +312 -0
  92. package/dist/core/dashboardData.d.ts.map +1 -0
  93. package/dist/core/dashboardData.js +550 -0
  94. package/dist/core/dashboardData.js.map +1 -0
  95. package/dist/core/dataStorage.d.ts +42 -0
  96. package/dist/core/dataStorage.d.ts.map +1 -0
  97. package/dist/core/dataStorage.js +193 -0
  98. package/dist/core/dataStorage.js.map +1 -0
  99. package/dist/core/draftBrief.d.ts +27 -0
  100. package/dist/core/draftBrief.d.ts.map +1 -0
  101. package/dist/core/draftBrief.js +69 -0
  102. package/dist/core/draftBrief.js.map +1 -0
  103. package/dist/core/extract.d.ts +67 -0
  104. package/dist/core/extract.d.ts.map +1 -0
  105. package/dist/core/extract.js +262 -0
  106. package/dist/core/extract.js.map +1 -0
  107. package/dist/core/gscFreshness.d.ts +18 -0
  108. package/dist/core/gscFreshness.d.ts.map +1 -0
  109. package/dist/core/gscFreshness.js +32 -0
  110. package/dist/core/gscFreshness.js.map +1 -0
  111. package/dist/core/linkGraph.d.ts +19 -0
  112. package/dist/core/linkGraph.d.ts.map +1 -0
  113. package/dist/core/linkGraph.js +125 -0
  114. package/dist/core/linkGraph.js.map +1 -0
  115. package/dist/core/passageScore.d.ts +19 -0
  116. package/dist/core/passageScore.d.ts.map +1 -0
  117. package/dist/core/passageScore.js +59 -0
  118. package/dist/core/passageScore.js.map +1 -0
  119. package/dist/core/paths.d.ts +5 -0
  120. package/dist/core/paths.d.ts.map +1 -0
  121. package/dist/core/paths.js +15 -0
  122. package/dist/core/paths.js.map +1 -0
  123. package/dist/core/queryData.d.ts +35 -0
  124. package/dist/core/queryData.d.ts.map +1 -0
  125. package/dist/core/queryData.js +200 -0
  126. package/dist/core/queryData.js.map +1 -0
  127. package/dist/core/reranker.d.ts +8 -0
  128. package/dist/core/reranker.d.ts.map +1 -0
  129. package/dist/core/reranker.js +68 -0
  130. package/dist/core/reranker.js.map +1 -0
  131. package/dist/core/robots.d.ts +11 -0
  132. package/dist/core/robots.d.ts.map +1 -0
  133. package/dist/core/robots.js +76 -0
  134. package/dist/core/robots.js.map +1 -0
  135. package/dist/core/sitemap.d.ts +15 -0
  136. package/dist/core/sitemap.d.ts.map +1 -0
  137. package/dist/core/sitemap.js +100 -0
  138. package/dist/core/sitemap.js.map +1 -0
  139. package/dist/core/sql.d.ts +6 -0
  140. package/dist/core/sql.d.ts.map +1 -0
  141. package/dist/core/sql.js +6 -0
  142. package/dist/core/sql.js.map +1 -0
  143. package/dist/core/types.d.ts +22 -0
  144. package/dist/core/types.d.ts.map +1 -0
  145. package/dist/core/types.js +2 -0
  146. package/dist/core/types.js.map +1 -0
  147. package/dist/core/url-key.d.ts +39 -0
  148. package/dist/core/url-key.d.ts.map +1 -0
  149. package/dist/core/url-key.js +106 -0
  150. package/dist/core/url-key.js.map +1 -0
  151. package/dist/generators/index.d.ts +45 -0
  152. package/dist/generators/index.d.ts.map +1 -0
  153. package/dist/generators/index.js +184 -0
  154. package/dist/generators/index.js.map +1 -0
  155. package/dist/index.d.ts +3 -0
  156. package/dist/index.d.ts.map +1 -0
  157. package/dist/index.js +10 -0
  158. package/dist/index.js.map +1 -0
  159. package/dist/server.d.ts +7 -0
  160. package/dist/server.d.ts.map +1 -0
  161. package/dist/server.js +1387 -0
  162. package/dist/server.js.map +1 -0
  163. package/dist/src/ui/dashboard.html +347 -0
  164. package/dist/src/ui/sync-progress.html +104 -0
  165. package/package.json +101 -0
  166. package/server.json +57 -0
package/dist/server.js ADDED
@@ -0,0 +1,1387 @@
1
+ import { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js';
2
+ import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
3
+ import { registerAppTool, registerAppResource, RESOURCE_MIME_TYPE } from '@modelcontextprotocol/ext-apps/server';
4
+ import { readFileSync, writeFileSync, mkdirSync } from 'node:fs';
5
+ import { readFile } from 'node:fs/promises';
6
+ import { homedir } from 'node:os';
7
+ import path from 'node:path';
8
+ import { fileURLToPath } from 'node:url';
9
+ import { z } from 'zod';
10
+ import { getDashboardData } from './core/dashboardData.js';
11
+ import { runAudit, runSingleCheck, listChecks } from './audit/engine.js';
12
+ import { buildAuditMarkdown } from './audit/report.js';
13
+ import { diffLatest, buildDriftMarkdown } from './audit/drift.js';
14
+ import { checkAgentReadiness, buildAgentReadinessMarkdown } from './core/agentReadiness.js';
15
+ import { detectTemplates } from './audit/templates.js';
16
+ import { suggestPages } from './audit/opportunities.js';
17
+ import { computeTopicGaps } from './audit/topicGaps.js';
18
+ import { AuditDatabase } from './core/AuditDatabase.js';
19
+ import { dbPathFor, sanitizeProperty } from './core/paths.js';
20
+ import { runQueryData, QUERYABLE_TABLES, FILTER_OPS, METRIC_FNS, truncateCell } from './core/queryData.js';
21
+ import { storageSummary, pruneProperty, fmtBytes } from './core/dataStorage.js';
22
+ import { scoreSitePassages } from './core/passageScore.js';
23
+ import { buildDraftBrief } from './core/draftBrief.js';
24
+ import { generateJsonLd, suggestRedirect, suggestInternalLinks, collapseChainRules } from './generators/index.js';
25
+ import { urlKey, hostFormForProperty } from './core/url-key.js';
26
+ import { GscClient } from './core/GscClient.js';
27
+ import { GscSync, FULL_DIMENSIONS } from './core/GscSync.js';
28
+ import { UrlInspector } from './core/UrlInspector.js';
29
+ import { Crawler } from './core/Crawler.js';
30
+ import { Refresh } from './core/Refresh.js';
31
+ import { DataForSeoClient } from './core/DataForSeoClient.js';
32
+ import { RankTracker } from './core/RankTracker.js';
33
+ import { Backlinks } from './core/Backlinks.js';
34
+ import { WikidataClient } from './core/WikidataClient.js';
35
+ import { Entities } from './core/Entities.js';
36
+ import { JobManager } from './core/JobManager.js';
37
+ const SERVER_NAME = 'seo-audit-console';
38
+ const SERVER_VERSION = JSON.parse(readFileSync(new URL('../package.json', import.meta.url), 'utf8')).version;
39
+ // Where per-property crawl/audit DBs live. Resolution order (computed once per run):
40
+ // SAC_DATA_DIR env > persisted choice (~/.seo-audit-console.json) > Documents default.
41
+ // The user can set the persisted choice through the `data_location` tool (no JSON editing).
42
+ const CONFIG_PATH = path.join(homedir(), '.seo-audit-console.json');
43
+ const DEFAULT_DATA_DIR = path.join(homedir(), 'Documents', 'seo-audit-console');
44
+ function readConfigDataDir() {
45
+ try {
46
+ return JSON.parse(readFileSync(CONFIG_PATH, 'utf8')).dataDir;
47
+ }
48
+ catch {
49
+ return undefined;
50
+ }
51
+ }
52
+ let RESOLVED_DATA_DIR = null;
53
+ export function dataDir() {
54
+ if (!RESOLVED_DATA_DIR)
55
+ RESOLVED_DATA_DIR = process.env.SAC_DATA_DIR ?? readConfigDataDir() ?? DEFAULT_DATA_DIR;
56
+ return RESOLVED_DATA_DIR;
57
+ }
58
+ const __dirname = path.dirname(fileURLToPath(import.meta.url));
59
+ const DASHBOARD_URI = 'ui://dashboard/main.html';
60
+ const SYNC_PROGRESS_URI = 'ui://sync-progress/main.html';
61
+ // Must match the categories actually used by CHECKS (src/audit/checks.ts) so category
62
+ // filters never silently return empty. (Was listing performance/agentic/integrity/war-stories
63
+ // which no check uses, and omitting content/security which checks do use.)
64
+ const CHECK_CATEGORIES = [
65
+ 'crawlability', 'indexation', 'onpage', 'content', 'schema', 'security', 'performance', 'merged',
66
+ ];
67
+ function isoDaysAgo(days) {
68
+ return new Date(Date.now() - days * 86400000).toISOString().slice(0, 10);
69
+ }
70
+ // Server-level instructions (MCP `initialize` result): teach the assistant how the data
71
+ // sources JOIN so it can COMPOSE bespoke multi-source analyses, not just run presets.
72
+ const SERVER_INSTRUCTIONS = `SEO Audit Console fuses four data sources into one SQLite database per property: Google Search Console history, a first-party site crawl, GSC URL Inspection, and on-demand DataForSEO (SERP/Labs/Backlinks). Its real power is COMPOSITION — joining sources to answer questions no single tool answers.
73
+
74
+ JOIN KEYS (memorise these):
75
+ - url_key — one normalised URL form (https, unified www/apex, sorted params, no tracking params/fragments). Joins the crawl (pages, links) ↔ GSC (search_analytics.page_key) ↔ url_inspection ↔ page_backlinks ↔ page_cwv ↔ page_entity. normalize_url shows the key for any URL.
76
+ - query — the literal search term. Joins GSC search_analytics ↔ DataForSEO keyword tools (keyword_volume, search_intent, ranked_keywords rows) ↔ keyword_intent.
77
+ - domain — a bare host (no scheme, no www). Joins the Labs tools (ranked_keywords, domain_visibility, top_pages, competitors_domain, topic_gaps) ↔ backlinks summary.
78
+
79
+ GRAIN (one line per source):
80
+ - search_analytics: date × query × page (plus device/country when synced with segments). Rows are additive; position must be impression-weighted when aggregated.
81
+ - pages / links: the LATEST crawl only, one row per url_key (history lives in page_snapshots, diffed by detect_changes). Carries title/H1/meta, canonical, schema (json_ld), body_chunks, iPR, click_depth, inlink_count.
82
+ - url_inspection: one row per inspected URL — Google's own view (coverage_state, google_canonical vs user_canonical, last_crawl_time, crawled_as, rich_results).
83
+ - rank_history: month × domain (DataForSEO rank distribution + ETV).
84
+ - page_backlinks: one row per backlinked URL (counts + live HTTP status).
85
+ - Labs tools: keyword × target, or month × target; cached 20 days; each call costs money — never loop them in bulk.
86
+
87
+ ARCHETYPE CHAINS (compose along these lines):
88
+ 1. Demand → reality: a GSC query with impressions but weak rank → the ranking page's crawl fields (title/H1/body_chunks) → does the page actually say what the query asks? → fix on-page or draft content.
89
+ 2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
90
+ 3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
91
+
92
+ Before planning ANY complex multi-source question, call composition_cookbook — it returns the full data-surface map and worked recipes using these exact tool and table names.`;
93
+ const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
94
+
95
+ This server's value is composition: joining Search Console, the crawl, URL Inspection and DataForSEO to answer questions no preset check covers. This page is static (no API calls) — use it to PLAN, then run the tools.
96
+
97
+ ## The data surface
98
+
99
+ | Source (table) | Grain | Key dimensions | Join keys | Freshness | Cost |
100
+ |---|---|---|---|---|---|
101
+ | search_analytics (GSC) | date × query × page (+device/country with segments) | clicks, impressions, ctr, position | page_key (url_key), query, date | sync_gsc / refresh_property — incremental, GSC lags ~2–3 days | free |
102
+ | pages + links (crawl) | one row per url_key, LATEST crawl | status, title, H1, meta, canonical_key, robots, json_ld, hreflang, redirects, body_chunks, word_count, ipr, click_depth, inlink_count, conditional_304 | url_key | start_crawl / refresh_property (on demand) | free |
103
+ | page_snapshots (drift) | url_key × crawl | field-level SEO snapshot per crawl | url_key, captured_at | every crawl, automatically | free |
104
+ | url_inspection | one row per inspected URL (top pages by clicks, quota-limited) | coverage_state, page_fetch_state, google_canonical, user_canonical, last_crawl_time, crawled_as, rich_results | url_key | inspect_urls | free (GSC quota) |
105
+ | sitemap_urls | one row per sitemap URL | lastmod | url_key | captured at crawl time | free |
106
+ | rank_history | month × domain | rank distribution (1–3/4–10/11–20/21–100), ETV | period | track_ranks | paid, 20-day cache |
107
+ | page_backlinks | one row per backlinked URL | backlinks, referring_domains, live status_code | url_key, domain | pull_backlinks (needs the DataForSEO Backlinks subscription) | paid, 20-day cache |
108
+ | keyword_intent | one row per keyword | intent + probability | query | search_intent (pass siteUrl to persist) | paid (cheap), cached |
109
+ | page_cwv | one row per audited URL | performance, LCP, CLS, TBT | url_key | page_lighthouse (pass siteUrl to persist) | paid, cached |
110
+ | page_entity + entity_edge | one row per page / edge per relation | QID, label, subclass-of / part-of | url_key, qid | resolve_entities (free Wikidata) | free |
111
+ | Labs (ranked_keywords, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps) | keyword × target, or month × target | volume, ETV, position, KD, intent, SERP features, AIO citations | domain, query | on demand | paid, 20-day cache |
112
+ | findings (audit_runs) | finding per check × URL | priority, evidence JSON | url_key, check_id | run_audit | free |
113
+
114
+ Raw access: query_audit runs any single check with full evidence; every table above lives in one SQLite file per property (path via data_location) if you need direct SQL.
115
+
116
+ ## Worked recipes
117
+
118
+ 1. **AI Overview citation loss.** ranked_keywords target:<your domain> aioOnly:true → the keywords where you are cited as an AIO source. For each cited keyword's ranking page, pull its per-day clicks from search_analytics (page_key × date). Pages whose clicks fell while the AIO citation appeared = you are feeding the answer without earning the visit.
119
+ 2. **Striking distance without body coverage.** query_audit check:striking-distance (GSC rank 11–20) → for each page, check pages.body_chunks for the query's terms. The automated versions: body-missing-top-query, rag-answer-gap, and score_passages for the dense-answer test. Pages ranking 11–20 that never answer the query in one passage are the highest-yield rewrites.
120
+ 3. **Not indexed + no equity.** url_inspection.coverage_state ~ 'not indexed' joined to pages.ipr + inlink_count. Low-iPR unindexed pages need internal links, not resubmission; high-iPR unindexed pages are the real anomalies. (Checks: coverage-not-indexed, underlinked-high-demand.)
121
+ 4. **Cannibalisation with semantic overlap.** run_audit → keyword-cannibalisation evidence lists the competing URLs per query → compare those pages' body_chunks: heavy chunk overlap = consolidate (301 the loser); light overlap = differentiate the titles/H1s and interlink with distinct anchors.
122
+ 5. **Stable rank, falling CTR → SERP feature shift.** In search_analytics find queries where weekly position is flat but ctr declines → related_terms / ranked_keywords serpFeatures for that keyword shows what now sits above you (AIO, featured snippet, shopping). ctr-below-expected is the deterministic starting list.
123
+ 6. **Schema vs rich-result reality.** pages.json_ld (declared @types) joined to url_inspection.rich_results (what Google actually detected + issues). The rich-result-issues check automates the per-URL diff; the composition question is per-TEMPLATE (list_templates): which template's schema never earns its rich result?
124
+ 7. **Migration signal transfer.** pages.redirects (recorded chains) → url_inspection.google_canonical of the target (has Google accepted the move?) → search_analytics clicks by page_key before/after the migration date. Equity that didn't follow the 301 shows up as a target with no canonical adoption and no click recovery.
125
+ 8. **404s with backlinks.** pages.status_code = 404 joined to page_backlinks.backlinks (run pull_backlinks first) → run_audit surfaces backlinks-to-404; fix_finding generates the 301 that recovers the equity.
126
+ 9. **Competitor topic gap.** topic_gaps (bounded + cached: competitor ranked_keywords minus your GSC queries and page titles/H1s, clustered and scored) — or do it manually with ranked_keywords per competitor when you want the raw rows.
127
+
128
+ ## Novel combinations (nothing else surfaces these)
129
+
130
+ - **Crawl budget vs equity:** url_inspection.last_crawl_time × pages.ipr — your highest-iPR pages should be recrawled often; a high-iPR page Google rarely revisits is a freshness/priority problem (and vice versa: junk crawled daily = wasted budget).
131
+ - **AIO text vs your content:** the ai_overview items in a SERP call (related_terms' underlying serpOrganic) × pages.body_chunks — is the text Google quotes actually on your page, and in one extractable chunk?
132
+ - **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
133
+ - **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
134
+
135
+ Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
136
+ // The check catalogue rendered as markdown — single source of truth is listChecks();
137
+ // shared by the seo-audit://checks-reference resource (and buildable for any category subset).
138
+ function buildChecksMarkdown(checks) {
139
+ const byCategory = new Map();
140
+ for (const c of checks) {
141
+ const cat = String(c.category);
142
+ if (!byCategory.has(cat))
143
+ byCategory.set(cat, []);
144
+ byCategory.get(cat).push(c);
145
+ }
146
+ const esc = (s) => s.replace(/\|/g, '\\|').replace(/\r?\n/g, ' ');
147
+ const sections = [...byCategory.entries()].map(([cat, list]) => `## ${cat} (${list.length})\n\n| Check | Severity | Labels | Certainty | Fix type | What it catches | Fix |\n|---|---|---|---|---|---|---|\n` +
148
+ list.map(c => `| \`${c.id}\` | ${c.severity} | ${c.labels.join(',')} | ${c.certainty} | ${c.fixType} | ${esc(c.title)} | ${esc(c.fix)} |`).join('\n'));
149
+ return `# Check registry — ${checks.length} checks\n\nLabels: D = deterministic (cites bytes), G = evidence from Google's own data (Search Console / URL Inspection), N = judgement (heuristic, gated behind includeJudgement). Certainty < 1 discounts a finding's priority.\n\n${sections.join('\n\n')}\n`;
150
+ }
151
+ const HELP_TEXT = `# SEO Audit Console — what it can do
152
+ A technical-SEO audit that fuses **Search Console + a site crawl + DataForSEO**, joined on a normalised URL, with evidence on every finding. Typical flow: **refresh → audit → fix → report**.
153
+
154
+ ## 1. Sync the data
155
+ - **refresh_property** — sync everything (GSC → crawl → URL inspection → rank history). _"Refresh sc-domain:example.com"_ · add \`segments:true\` for device/country, \`maxPages\`, \`startDate\`.
156
+ - **sync_gsc / start_crawl / inspect_urls / track_ranks** — run just one part. _"Crawl example.com, 500 pages"_
157
+ - **check_sync_status / check_crawl_status** — poll a job. _"Check sync status"_
158
+ - **list_properties** — _"List my Search Console properties"_
159
+
160
+ ## 2. Audit
161
+ - **run_audit** — score all checks; returns a prioritised markdown report. _"Run an SEO audit on sc-domain:example.com"_ · \`scope:full\`, \`categories\`, \`includeJudgement:true\`.
162
+ - **query_audit** — one named check with evidence (\`columns\`/\`offset\` for big sets). _"Show striking-distance for example.com"_
163
+ - **query_data** — read-only queries over the raw tables; aggregates in the database (counts/percentages/sums), answers not rows. _"How do status codes break down on example.com?"_
164
+ - **list_checks** — _"What does the audit check for?"_
165
+
166
+ ## 3. Fix (the moat)
167
+ - **fix_finding** — paste-ready remediation from your own data: JSON-LD for missing/invalid schema, a 301 rule for broken links, iPR-ranked internal-link suggestions. _"Generate the fix for finding 12"_ or _"fix_finding check:missing-required-fields url:https://example.com/x"_
168
+ - **detect_changes** — what changed since the last crawl (status, canonical, noindex, title, schema), severity-ranked — the regression monitor. _"Detect changes on example.com"_
169
+ - **check_agent_readiness** — is your site ready for AI agents? Scores llms.txt / agents.md / AI-bot rules / Content Signals / MCP server card / Agent Skills / API Catalog / OAuth signals (0–100 + level) with copy-paste fixes. _"Check agent readiness for example.com"_
170
+
171
+ ## 4. Backlinks, keywords & competitive (DataForSEO, on-demand, cached 20 days)
172
+ - **pull_backlinks** — backlink profile + per-page counts + live status → unlocks **backlinks-to-404** (recover lost equity), top-linked pages, true orphans. _"Pull backlinks for example.com"_
173
+ - **keyword_volume / related_terms** — volume/CPC, and People-Also-Ask + related searches. _"Search volume for [\\"best widgets\\"]"_
174
+ - **search_intent** — informational/navigational/commercial/transactional per keyword → spot intent mismatch behind low CTR. _"Classify intent for [\\"buy running shoes\\", \\"how to clean shoes\\"]"_
175
+ - **page_lighthouse** — lab Core Web Vitals + opportunities for one URL (~20–120s). _"Run Lighthouse on https://example.com/slow-page"_
176
+ - **competitors_domain** — domains competing for your organic keywords. _"Find competitors for example.com in the UK"_
177
+ - **page_intersection** — keywords competitor pages rank for but yours doesn’t (content gap). _"Content gap: competitorUrls [\\"https://rival.com/guide\\"], excludePages [\\"https://example.com/guide\\"]"_
178
+ - **domain_visibility** — monthly ranking-keyword distribution + ETV trend for ANY domain/subdomain (Semrush-style organic overview). _"Show visibility over time for competitor.com"_
179
+ - **top_pages** — a domain's top organic pages by estimated traffic. _"Top pages on competitor.com"_
180
+ - **ranked_keywords** — keywords a domain / subdomain / URL / subfolder ranks for (+ difficulty, intent, SERP features). _"What does competitor.com/blog/ rank for?"_ · \`scope:url|folder\` · \`aioOnly:true\` = keywords where the target is cited in AI Overviews
181
+ - **topic_gaps** — what related topics should this site cover to be expert in its space: competitor keyword footprints minus everything you already rank or have a page for, clustered into ranked topics with volumes, the owning competitor and your nearest existing page. _"What topics should example.com cover? Compare against rival.com"_
182
+
183
+ ## 5. Templates & opportunities
184
+ - **list_templates** — cluster pages into templates (one fix → N pages) with a representative exemplar. _"List page templates for example.com"_
185
+ - **suggest_pages** — new-page ideas grounded in real GSC demand, minus what you already cover. _"Suggest new pages for example.com"_
186
+ - **resolve_entities** — map pages to Wikidata entities → unlocks entity-internal-link-gap (judgement). _"Resolve entities for example.com"_
187
+
188
+ ## 6. Reports & dashboard
189
+ - **get_dashboard** — interactive dashboard (renders in chat). _"Show the dashboard for example.com"_
190
+ - **export_report** — self-contained shareable HTML to send a client. _"Export the report for example.com"_
191
+
192
+ ## 7. Utilities
193
+ - **data_location** — where DBs are stored (set with a path). · **normalize_url** — the join key for a URL.
194
+ - **data_storage** — per-property disk usage + row counts; prune (vacuum / clear-crawl-history / delete-property, destructive ones need \`confirm:true\`). _"How much disk is my audit data using?"_
195
+
196
+ _Tip: first time on a property → \`refresh_property\` then \`run_audit\`._
197
+ _Composing your own analysis? Call \`composition_cookbook\` first — the data-surface map (grain + join keys per source) and worked multi-source recipes._`;
198
+ export function createServer() {
199
+ const server = new McpServer({ name: SERVER_NAME, version: SERVER_VERSION }, { instructions: SERVER_INSTRUCTIONS });
200
+ const jobs = new JobManager();
201
+ const credPath = process.env.GOOGLE_APPLICATION_CREDENTIALS;
202
+ const gsc = credPath ? new GscClient(credPath) : null;
203
+ const sync = gsc ? new GscSync(gsc, dataDir()) : null;
204
+ const inspector = gsc ? new UrlInspector(gsc, dataDir()) : null;
205
+ const crawler = new Crawler(dataDir()); // no GSC credentials required
206
+ const dfsUser = process.env.DATAFORSEO_USERNAME;
207
+ const dfsPass = process.env.DATAFORSEO_PASSWORD;
208
+ const dfsCacheDays = Number(process.env.DATAFORSEO_CACHE_DAYS) || 20;
209
+ const dfs = dfsUser && dfsPass
210
+ ? new DataForSeoClient(dfsUser, dfsPass, path.join(dataDir(), 'dataforseo-cache.db'), dfsCacheDays)
211
+ : null;
212
+ const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
213
+ const backlinks = dfs ? new Backlinks(dfs, dataDir()) : null;
214
+ const entities = new Entities(new WikidataClient(path.join(dataDir(), 'wikidata-cache.db')), dataDir());
215
+ const refresh = new Refresh(sync, crawler, inspector, rankTracker);
216
+ const requireGsc = (v) => {
217
+ if (!v)
218
+ throw new Error('GOOGLE_APPLICATION_CREDENTIALS is not set — required for Search Console access.');
219
+ return v;
220
+ };
221
+ const requireDfs = (v) => {
222
+ if (!v)
223
+ throw new Error('DATAFORSEO_USERNAME / DATAFORSEO_PASSWORD not set — required for DataForSEO.');
224
+ return v;
225
+ };
226
+ // ── Introspection (no data / creds required) ────────────────────────────
227
+ server.registerTool('seo_audit_help', {
228
+ title: 'Help — what this audit can do',
229
+ description: 'Overview of every tool/feature with an example prompt for each. Start here.',
230
+ inputSchema: {},
231
+ }, async () => ({ content: [{ type: 'text', text: HELP_TEXT }], structuredContent: { version: SERVER_VERSION } }));
232
+ server.registerTool('composition_cookbook', {
233
+ title: 'Composition cookbook (data-surface map + recipes)',
234
+ description: 'The full data-surface map (every source: grain, dimensions, join keys, freshness, cost) plus worked multi-source recipes and novel combinations, in this server\'s actual tool and table names. Static — no API calls, no cost. Call this BEFORE planning any complex question that spans more than one data source.',
235
+ inputSchema: {},
236
+ }, async () => ({ content: [{ type: 'text', text: COOKBOOK_TEXT }], structuredContent: { version: SERVER_VERSION } }));
237
+ server.registerTool('list_checks', {
238
+ title: 'List audit checks',
239
+ description: 'List the technical-SEO check categories this server can evaluate.',
240
+ inputSchema: { category: z.enum(CHECK_CATEGORIES).optional() },
241
+ }, async ({ category }) => {
242
+ const all = listChecks();
243
+ const checks = category ? all.filter(c => c.category === category) : all;
244
+ return { content: [{ type: 'text', text: `${checks.length} checks${category ? ` in ${category}` : ''}` }], structuredContent: { checks } };
245
+ });
246
+ server.registerTool('list_templates', {
247
+ title: 'Detect page templates',
248
+ description: 'Cluster the crawled pages into templates (by URL morphology + JSON-LD @type) and return each template with its page count and a representative exemplar URL. A single template-level fix corrects every page in the cluster — this is the map for per-template analysis. Needs a crawl (run refresh_property / start_crawl first).',
249
+ inputSchema: { siteUrl: z.string(), minMembers: z.number().int().min(2).max(100).optional() },
250
+ }, async ({ siteUrl, minMembers }) => {
251
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
252
+ try {
253
+ const clusters = detectTemplates(db.db, minMembers != null ? { minMembers } : {});
254
+ const summary = clusters.slice(0, 20).map(c => `• ${c.morphology} [${c.schemaType}] — ${c.count} pages (e.g. ${c.exemplarUrl})`).join('\n');
255
+ return {
256
+ content: [{ type: 'text', text: clusters.length ? `${clusters.length} templates detected:\n${summary}` : 'No templates with ≥ the minimum members — crawl first, or lower minMembers.' }],
257
+ structuredContent: { templates: clusters.map(({ memberKeys: _m, ...c }) => c), count: clusters.length },
258
+ };
259
+ }
260
+ finally {
261
+ db.close();
262
+ }
263
+ });
264
+ server.registerTool('data_location', {
265
+ title: 'Get or set where audit data is stored',
266
+ description: 'No args: report the folder where per-property crawl/audit databases are saved. With `path`: set it (persisted to ~/.seo-audit-console.json) — restart Claude Desktop to apply. Default is your Documents folder; the SAC_DATA_DIR env var overrides everything.',
267
+ inputSchema: { path: z.string().optional() },
268
+ }, async ({ path: newPath }) => {
269
+ if (newPath) {
270
+ // Persist an ABSOLUTE path — a relative one resolves against the host process's
271
+ // cwd, which differs across Claude Desktop launches, so DBs would "disappear".
272
+ newPath = path.resolve(newPath);
273
+ mkdirSync(newPath, { recursive: true });
274
+ writeFileSync(CONFIG_PATH, JSON.stringify({ dataDir: newPath }, null, 2));
275
+ return {
276
+ content: [{ type: 'text', text: `Data location set to ${newPath}. Restart Claude Desktop to apply (this session still uses ${dataDir()}). Move existing .db files there to keep your synced data.` }],
277
+ structuredContent: { requested: newPath, active: dataDir(), restartRequired: true, configFile: CONFIG_PATH },
278
+ };
279
+ }
280
+ return {
281
+ content: [{ type: 'text', text: `Audit data is stored in: ${dataDir()}` }],
282
+ structuredContent: { active: dataDir(), default: DEFAULT_DATA_DIR, env: process.env.SAC_DATA_DIR ?? null, configFile: CONFIG_PATH },
283
+ };
284
+ });
285
+ server.registerTool('data_storage', {
286
+ title: 'Data storage summary + pruning',
287
+ description: 'Data hygiene. No args: list every per-property database in the data dir — size (incl. WAL), key row counts (search_analytics, pages, links, page_snapshots, findings), last sync and last crawl — plus the DataForSEO cache and reports folder sizes. Optional siteUrl narrows to one property. Pruning via `prune:{siteUrl, action}`: "vacuum" (compact the DB, reports bytes reclaimed), "clear-crawl-history" (delete page_snapshots + audit runs/findings older than the 5 most recent, then vacuum), "delete-property" (remove the DB files entirely). Destructive actions (clear-crawl-history, delete-property) REQUIRE confirm:true and refuse loudly without it, naming what would be deleted. Refuses to prune while any job is running.',
288
+ inputSchema: {
289
+ siteUrl: z.string().optional(),
290
+ prune: z.object({
291
+ siteUrl: z.string(),
292
+ action: z.enum(['vacuum', 'clear-crawl-history', 'delete-property']),
293
+ }).optional(),
294
+ confirm: z.boolean().optional(),
295
+ },
296
+ }, async ({ siteUrl, prune, confirm }) => {
297
+ if (prune) {
298
+ const running = jobs.list().filter(j => j.state === 'running').map(j => ({ id: j.id, type: j.type }));
299
+ const r = pruneProperty(dataDir(), prune.siteUrl, prune.action, confirm === true, running);
300
+ if (!r.ok) {
301
+ return { content: [{ type: 'text', text: r.refused ?? `Could not ${prune.action}.` }], structuredContent: r };
302
+ }
303
+ const d = r.detail;
304
+ const line = prune.action === 'delete-property'
305
+ ? `Deleted ${prune.siteUrl}'s database (${fmtBytes(Number(d.bytesFreed) || 0)} freed).`
306
+ : prune.action === 'vacuum'
307
+ ? `Vacuumed ${prune.siteUrl}: ${fmtBytes(Number(d.bytesBefore) || 0)} → ${fmtBytes(Number(d.bytesAfter) || 0)} (${fmtBytes(Number(d.bytesReclaimed) || 0)} reclaimed).`
308
+ : `Cleared crawl/audit history for ${prune.siteUrl} beyond the 5 most recent runs: ${fmtBytes(Number(d.bytesBefore) || 0)} → ${fmtBytes(Number(d.bytesAfter) || 0)} (${fmtBytes(Number(d.bytesReclaimed) || 0)} reclaimed).`;
309
+ return { content: [{ type: 'text', text: line }], structuredContent: r };
310
+ }
311
+ const s = storageSummary(dataDir(), siteUrl);
312
+ if (!s.properties.length && !s.caches.length) {
313
+ return { content: [{ type: 'text', text: `No databases in ${s.dataDir}${siteUrl ? ` matching ${siteUrl}` : ''} — run refresh_property first.` }], structuredContent: s };
314
+ }
315
+ const rows = s.properties.map(p => `| ${p.siteUrl ?? p.file} | ${fmtBytes(p.bytes)} | ${p.searchAnalytics.toLocaleString('en-US')} | ${p.pages.toLocaleString('en-US')} | ${p.links.toLocaleString('en-US')} | ${p.pageSnapshots.toLocaleString('en-US')} | ${p.findings.toLocaleString('en-US')} | ${p.lastSynced?.slice(0, 10) ?? '—'} | ${p.lastCrawl?.slice(0, 10) ?? '—'} |`).join('\n');
316
+ const cacheLines = s.caches.map(c => `- ${c.file}: ${fmtBytes(c.bytes)}`).join('\n');
317
+ const md = `**Data dir:** ${s.dataDir} — total ${fmtBytes(s.totalBytes)}\n\n` +
318
+ `| Property | Size | GSC rows | Pages | Links | Snapshots | Findings | Last sync | Last crawl |\n|---|---|---|---|---|---|---|---|---|\n${rows}\n\n` +
319
+ `${cacheLines ? `Caches:\n${cacheLines}\n` : ''}Reports folder: ${fmtBytes(s.reportsBytes)}` +
320
+ `${s.other.length ? `\nOther .db files: ${s.other.map(o => `${o.file} (${fmtBytes(o.bytes)})`).join(', ')}` : ''}\n\n` +
321
+ `Prune with data_storage prune:{siteUrl, action:"vacuum" | "clear-crawl-history" | "delete-property"} (destructive actions need confirm:true).`;
322
+ return { content: [{ type: 'text', text: md }], structuredContent: s };
323
+ });
324
+ // ── Audit engine ────────────────────────────────────────────────────────
325
+ server.registerTool('run_audit', {
326
+ title: 'Run SEO audit',
327
+ description: 'Run the technical-SEO checks against synced data and return scored findings, ranked by expected clicks per dev-hour — Priority = (T × Y × C) / E (T = clicks at stake from real GSC data, Y = expected yield, C = certainty, E = effort hours). Crawl + GSC + URL-inspection checks. Set includeJudgement=true to include heuristic (N) checks.',
328
+ inputSchema: {
329
+ siteUrl: z.string(),
330
+ scope: z.enum(['core', 'full']).optional(),
331
+ categories: z.array(z.enum(CHECK_CATEGORIES)).optional(),
332
+ includeJudgement: z.boolean().optional(),
333
+ },
334
+ }, async ({ siteUrl, scope, categories, includeJudgement }) => {
335
+ const result = runAudit(dataDir(), siteUrl, { scope, categories, includeJudgement });
336
+ return {
337
+ content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) }],
338
+ structuredContent: result,
339
+ };
340
+ });
341
+ server.registerTool('query_audit', {
342
+ title: 'Run one audit check',
343
+ description: 'Run a single named check (see list_checks) and return its affected URLs + evidence. Token discipline: `columns` narrows evidence to the named keys, string evidence values are truncated at 120 chars (trailing …), `offset` + `limit` page through large result sets (the footer states showing X–Y of TOTAL). Grain: finding per URL with evidence JSON. Joins: url_key across all sources.',
344
+ inputSchema: {
345
+ siteUrl: z.string(),
346
+ check: z.string(),
347
+ limit: z.number().int().min(1).max(1000).optional(),
348
+ offset: z.number().int().min(0).optional(),
349
+ columns: z.array(z.string()).max(20).optional().describe('Evidence keys to keep (default: all)'),
350
+ },
351
+ }, async ({ siteUrl, check, limit, offset, columns }) => {
352
+ const r = runSingleCheck(dataDir(), siteUrl, check, limit, offset ?? 0);
353
+ // Token discipline (mirrors query_data): optional evidence-column selection + loud
354
+ // 120-char cell truncation, and an explicit offset/total footer.
355
+ const shape = (f) => {
356
+ const src = (f.evidence ?? {});
357
+ const keys = columns?.length ? columns.filter(k => k in src) : Object.keys(src);
358
+ const evidence = {};
359
+ for (const k of keys)
360
+ evidence[k] = truncateCell(src[k]);
361
+ return { ...f, evidence };
362
+ };
363
+ let sc = { ...r, findingsTotal: r.total };
364
+ let findings = r.findings.map(shape);
365
+ // Hosts cap the model-facing result (~60k chars). On evidence-heavy checks a large
366
+ // `limit` can blow that ceiling and error the whole call — trim findings (keeping the
367
+ // true total) instead of failing. Same guard pattern as detect_changes.
368
+ while (findings.length > 25 && JSON.stringify({ ...sc, findings }).length > 45000) {
369
+ findings = findings.slice(0, Math.floor(findings.length / 2));
370
+ }
371
+ const start = (offset ?? 0) + (findings.length ? 1 : 0);
372
+ const end = (offset ?? 0) + findings.length;
373
+ const nextOffset = end < r.total ? end : null;
374
+ sc = { ...sc, findings, findingsReturned: findings.length, offset: offset ?? 0, nextOffset };
375
+ return {
376
+ content: [{ type: 'text', text: `${check}: showing ${start}–${end} of ${r.total} findings${nextOffset != null ? `; next offset ${nextOffset}` : ''}${findings.length < r.findings.length ? ` (trimmed to fit the token cap — use a smaller limit)` : ''}` }],
377
+ structuredContent: sc,
378
+ };
379
+ });
380
+ server.registerTool('query_data', {
381
+ title: 'Query the raw data (aggregate first)',
382
+ description: 'Direct read-only access to the property database — the composition layer\'s escape hatch. AGGREGATE IN THE DATABASE and return answers, not rows: default mode groups by the columns you name and returns counts, percentages and sum/avg/min/max metrics (Screaming Frog-style distributions in one call — "status codes by folder", "clicks by page_key"). mode:rows returns raw rows with strict token discipline (curated default columns, 120-char cells, loud paging footer). Filters are exact/range/like/in and always parameterised; column names are validated against the live table schema. Grain: per table — search_analytics = date × query × page; pages/links = latest crawl per url_key; url_inspection / sitemap_urls / page_backlinks = one row per URL; findings = check × URL per audit run. Joins: url_key across pages/links/search_analytics.page_key/url_inspection/sitemap_urls/page_backlinks; query → keyword tools; run_id → audit runs.',
383
+ inputSchema: {
384
+ siteUrl: z.string(),
385
+ table: z.enum(QUERYABLE_TABLES),
386
+ mode: z.enum(['aggregate', 'rows']).optional().describe('Default aggregate — return grouped answers, not rows'),
387
+ groupBy: z.array(z.string()).max(4).optional().describe('Columns to group by (aggregate mode)'),
388
+ metrics: z.array(z.object({ fn: z.enum(METRIC_FNS), column: z.string() })).max(6).optional().describe('count is always included; add sum/avg/min/max over numeric columns'),
389
+ filters: z.array(z.object({
390
+ column: z.string(),
391
+ op: z.enum(FILTER_OPS),
392
+ value: z.union([z.string(), z.number(), z.null(), z.array(z.union([z.string(), z.number()])).max(100)]),
393
+ })).max(8).optional(),
394
+ orderBy: z.string().optional().describe('Aggregate: a groupBy column, "count", or a metric alias like sum_clicks. Rows: any column'),
395
+ orderDir: z.enum(['asc', 'desc']).optional(),
396
+ limit: z.number().int().min(1).max(500).optional().describe('Default 25 (aggregate) / 50 (rows)'),
397
+ offset: z.number().int().min(0).optional().describe('Rows mode paging'),
398
+ columns: z.array(z.string()).max(20).optional().describe('Rows mode: columns to return (default: a curated per-table set)'),
399
+ },
400
+ }, async ({ siteUrl, table, mode, groupBy, metrics, filters, orderBy, orderDir, limit, offset, columns }) => {
401
+ const r = runQueryData(dbPathFor(dataDir(), siteUrl), {
402
+ table, mode, groupBy, metrics, filters, orderBy, orderDir, limit, offset, columns,
403
+ });
404
+ return { content: [{ type: 'text', text: r.markdown }], structuredContent: r.structured };
405
+ });
406
+ server.registerTool('score_passages', {
407
+ title: 'Score passage relevance (AI-search readiness)',
408
+ description: 'Run a local cross-encoder reranker (ms-marco-MiniLM-L-6-v2, downloaded once to a cache, no Python) over each ranking page\'s heading chunks against its top Search Console query, and persist the single best-passage relevance score. It SCORES relevance (a classifier — it cannot hallucinate). Powers the `weak-passage-answer` check — the "RAG snippetability" test: does any passage confidently answer the query, the way AI/passage search re-ranks? On-demand + ML-heavy (~0.6s/page), bounded to pages that already rank. `limit` caps pages (highest-impression first); `minImpressions` sets the floor (default 50). First run downloads ~25MB. Re-run after a re-crawl.',
409
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(2000).optional(), minImpressions: z.number().int().min(0).optional() },
410
+ }, async ({ siteUrl, limit, minImpressions }) => {
411
+ const r = await scoreSitePassages(dataDir(), siteUrl, { limit, minImpressions });
412
+ const lines = r.weakest.slice(0, 15).map(w => ` ${w.score} ${w.url.replace(/^https?:\/\/[^/]+/, '')} · "${w.query}"`).join('\n');
413
+ return {
414
+ content: [{ type: 'text', text: `Scored ${r.scored} ranking pages; ${r.flagged} have no strongly-relevant passage (score < 3) — run_audit surfaces them as weak-passage-answer.${lines ? `\n\nWeakest:\n${lines}` : ''}` }],
415
+ structuredContent: r,
416
+ };
417
+ });
418
+ server.registerTool('draft_content', {
419
+ title: 'Draft a grounded fix in the author\'s own voice',
420
+ description: 'Assemble a GROUNDED writing brief for a page so YOU (the assistant) can draft the fix — no bundled LLM. Returns the ranking-query gap, the page\'s OWN passages as voice exemplars (match tone/pace/lexicon), and the most query-relevant existing passages (selected by the reranker) as facts to ground in. Use it on a weak-passage-answer / not-front-loaded page to write a self-contained answer passage in the site\'s voice, inventing nothing. Omit urlKey to target the weakest-scoring page. Needs body_chunks (+ score_passages for the gap).',
421
+ inputSchema: { siteUrl: z.string(), urlKey: z.string().optional() },
422
+ }, async ({ siteUrl, urlKey }) => {
423
+ const r = await buildDraftBrief(dataDir(), siteUrl, urlKey);
424
+ if ('error' in r)
425
+ return { content: [{ type: 'text', text: r.error }], structuredContent: r };
426
+ const voice = r.voice.map((v, i) => ` [V${i + 1}${v.heading ? ` · ${v.heading}` : ''}] ${v.text}`).join('\n');
427
+ const facts = r.facts.map((f, i) => ` [F${i + 1} · rel ${f.relevance}${f.heading ? ` · ${f.heading}` : ''}] ${f.text}`).join('\n');
428
+ const text = `WRITING BRIEF — ${r.url}\nTarget query: “${r.topQuery ?? '(unknown)'}”\nGap: ${r.gap}\n\nTASK\n${r.brief}\n\nVOICE (match this — the page's own writing):\n${voice || ' (no substantial passages)'}\n\nFACTS (ground in these — the page's most query-relevant content; invent nothing beyond them):\n${facts || ' (none)'}`;
429
+ return { content: [{ type: 'text', text }], structuredContent: r };
430
+ });
431
+ // detect_changes — change-detection / drift: what changed on the site since the last crawl.
432
+ server.registerTool('detect_changes', {
433
+ title: 'Detect changes since the last crawl',
434
+ description: 'Compare the two most recent crawls and report what changed per URL — status code, indexability, canonical, robots/noindex, title, meta, H1, schema, large content swings — each classified by severity (critical → info). This is the monitor: run refresh_property on different days to build history, then this surfaces regressions (a page that went noindex, a canonical that flipped, a 200 that became a 404). Needs at least two crawls.',
435
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(500).optional() },
436
+ }, async ({ siteUrl, limit }) => {
437
+ const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
438
+ try {
439
+ const d = diffLatest(adb.db);
440
+ // Cap the changes array in structuredContent to `limit` — the full diff can be tens of
441
+ // thousands of rows (e.g. a crawl-methodology change) and blow the MCP token ceiling. The
442
+ // summary keeps the TRUE totals; markdown is already limited.
443
+ const lim = limit ?? 50;
444
+ const sc = { ...d, changes: d.changes.slice(0, lim), changesReturned: Math.min(lim, d.changes.length), changesTotal: d.changes.length };
445
+ return {
446
+ content: [{ type: 'text', text: buildDriftMarkdown(d, siteUrl, lim) }],
447
+ structuredContent: sc,
448
+ };
449
+ }
450
+ finally {
451
+ adb.close();
452
+ }
453
+ });
454
+ // check_agent_readiness — is the site ready for AI agents? (the agentic-SEO / GEO frontier)
455
+ server.registerTool('check_agent_readiness', {
456
+ title: 'Check agent readiness (AI-agent / GEO signals)',
457
+ description: 'Probe a site for AI-agent readiness — the signals agents use to discover and use it: robots.txt AI-bot rules + Content Signals, sitemap, Link headers, llms.txt, agents.md, Markdown content negotiation, Web Bot Auth, MCP server card, Agent Skills, API Catalog, OAuth discovery. Returns a 0–100 score, a level (Basic web presence → Agent-native), and a per-check checklist with copy-paste fixes. Live HTTP probes of the property origin (no crawl/GSC needed). Modelled on Cloudflare\'s isitagentready.com.',
458
+ inputSchema: { siteUrl: z.string() },
459
+ }, async ({ siteUrl }) => {
460
+ const r = await checkAgentReadiness(siteUrl);
461
+ // Persist so the dashboard/export can show it (live probe, not crawl-derived).
462
+ try {
463
+ const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
464
+ try {
465
+ adb.db.prepare(`INSERT INTO agent_readiness (origin,score,level,checks,by_category,checked_at) VALUES (?,?,?,?,?,datetime('now'))
466
+ ON CONFLICT(origin) DO UPDATE SET score=excluded.score, level=excluded.level, checks=excluded.checks, by_category=excluded.by_category, checked_at=excluded.checked_at`)
467
+ .run(r.origin, r.score, r.level, JSON.stringify(r.checks), JSON.stringify(r.byCategory));
468
+ }
469
+ finally {
470
+ adb.close();
471
+ }
472
+ }
473
+ catch { /* persistence is best-effort */ }
474
+ return {
475
+ content: [{ type: 'text', text: buildAgentReadinessMarkdown(r) }],
476
+ structuredContent: r,
477
+ };
478
+ });
479
+ // fix_finding — the moat: turn a finding into a concrete, paste-ready remediation.
480
+ server.registerTool('fix_finding', {
481
+ title: 'Generate a fix for a finding',
482
+ description: 'Turn an audit finding into a concrete, paste-ready remediation: JSON-LD for missing structured data, a 301 rule for broken / redirecting internal links, or internal-link suggestions for orphan / striking-distance pages. Other checks return their deterministic fix guidance. Dry-run — returns artifacts, never writes to your site. Identify the finding by findingId (from run_audit) or by check + url.',
483
+ inputSchema: {
484
+ siteUrl: z.string(),
485
+ findingId: z.number().int().optional(),
486
+ check: z.string().optional(),
487
+ url: z.string().optional(),
488
+ redirectFormat: z.enum(['htaccess', 'nginx', 'nextjs']).optional(),
489
+ },
490
+ }, async ({ siteUrl, findingId, check, url, redirectFormat }) => {
491
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
492
+ try {
493
+ // Resolve the finding → checkId, affected url_key, evidence.
494
+ let checkId = check;
495
+ let affectedKey = null;
496
+ let evidence = {};
497
+ if (findingId != null) {
498
+ const row = db.db.prepare('SELECT check_id, url_key, evidence FROM findings WHERE id = ?').get(findingId);
499
+ if (!row) {
500
+ return { content: [{ type: 'text', text: `No finding #${findingId} for ${siteUrl}. Run run_audit first.` }], structuredContent: { error: 'not_found', findingId } };
501
+ }
502
+ checkId = row.check_id;
503
+ affectedKey = row.url_key;
504
+ evidence = row.evidence ? JSON.parse(row.evidence) : {};
505
+ }
506
+ else if (check && url) {
507
+ const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
508
+ affectedKey = urlKey(url, { hostForm });
509
+ const row = db.db.prepare('SELECT evidence FROM findings WHERE check_id = ? AND url_key = ? ORDER BY id DESC LIMIT 1').get(check, affectedKey);
510
+ evidence = row?.evidence ? JSON.parse(row.evidence) : {};
511
+ }
512
+ else {
513
+ throw new Error('Provide either findingId, or check + url.');
514
+ }
515
+ let kind;
516
+ let fix;
517
+ switch (checkId) {
518
+ case 'missing-structured-data':
519
+ case 'invalid-schema':
520
+ case 'missing-required-fields': {
521
+ const page = db.db
522
+ .prepare('SELECT url, url_key, title, h1, meta_description, og_tags FROM pages WHERE url_key = ?')
523
+ .get(affectedKey);
524
+ if (!page) {
525
+ return { content: [{ type: 'text', text: `No crawled page for ${affectedKey} — crawl the property first.` }], structuredContent: { error: 'no_page', urlKey: affectedKey } };
526
+ }
527
+ kind = 'json-ld';
528
+ fix = generateJsonLd(page);
529
+ break;
530
+ }
531
+ case 'redirect-chain': {
532
+ // The finding's url_key IS the chain's final page — its row holds the recorded
533
+ // hops. Collapse each hop straight to the final URL; never fuzzy-match here.
534
+ const page = db.db.prepare('SELECT url, redirects FROM pages WHERE url_key = ?').get(affectedKey);
535
+ let hops = [];
536
+ try {
537
+ hops = page?.redirects ? JSON.parse(page.redirects) : [];
538
+ }
539
+ catch { /* malformed chain */ }
540
+ kind = 'redirect';
541
+ fix = page && hops.length
542
+ ? { ...collapseChainRules(page.url, hops, redirectFormat ?? 'htaccess'), chain: hops }
543
+ : { explanation: 'Redirect chain no longer present in the latest crawl — re-crawl and re-audit.' };
544
+ break;
545
+ }
546
+ case 'internal-links-to-redirects': {
547
+ // The flagged key is a redirect SOURCE; the crawler recorded where it actually
548
+ // goes — find the chain containing it and use that real destination.
549
+ const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
550
+ let knownTarget = null;
551
+ for (const row of db.db.prepare(`SELECT url, redirects FROM pages WHERE redirects IS NOT NULL AND redirects <> ''`).iterate()) {
552
+ try {
553
+ const hops = JSON.parse(row.redirects);
554
+ if (hops.some(h => h.from && urlKey(String(h.from), { hostForm }) === affectedKey)) {
555
+ knownTarget = row.url;
556
+ break;
557
+ }
558
+ }
559
+ catch { /* skip malformed */ }
560
+ }
561
+ kind = 'redirect';
562
+ fix = suggestRedirect(affectedKey ?? '', [], redirectFormat ?? 'htaccess', knownTarget);
563
+ break;
564
+ }
565
+ case 'broken-internal-links': {
566
+ // Genuinely dead target (4xx/5xx) — no recorded destination exists, so fall back
567
+ // to token-overlap fuzzy matching against live pages.
568
+ const livePages = db.db
569
+ .prepare('SELECT url FROM pages WHERE status_code = 200 AND indexable = 1')
570
+ .all();
571
+ kind = 'redirect';
572
+ fix = suggestRedirect(affectedKey ?? '', livePages, redirectFormat ?? 'htaccess');
573
+ break;
574
+ }
575
+ case 'orphan-with-impressions':
576
+ case 'striking-distance': {
577
+ const page = db.db.prepare('SELECT h1, title FROM pages WHERE url_key = ?').get(affectedKey);
578
+ // Prefer the GSC query as anchor, but fall back to H1/title when it's a
579
+ // boolean/over-long search string (common on job boards) — not usable anchor text.
580
+ const q = evidence.query;
581
+ const cleanQuery = q && q.length <= 60 && !/["()]|\bor\b|\bnot\b|\s-\w/i.test(q) ? q : undefined;
582
+ // Never fall back to the raw q — it was just rejected as unusable anchor text.
583
+ const anchor = cleanQuery ?? page?.h1 ?? page?.title ?? '';
584
+ kind = 'internal-links';
585
+ fix = suggestInternalLinks(db.db, affectedKey ?? '', anchor);
586
+ break;
587
+ }
588
+ default: {
589
+ const def = listChecks().find(c => c.id === checkId);
590
+ kind = 'explanation';
591
+ fix = { explanation: def?.fix ?? 'No automated generator for this check — apply the recommendation manually.', fixType: def?.fixType, title: def?.title };
592
+ }
593
+ }
594
+ return {
595
+ content: [{ type: 'text', text: `Fix for ${checkId} (${kind})${affectedKey ? ` on ${affectedKey}` : ''}` }],
596
+ structuredContent: { checkId, urlKey: affectedKey, kind, fix },
597
+ };
598
+ }
599
+ finally {
600
+ db.close();
601
+ }
602
+ });
603
+ server.registerTool('normalize_url', {
604
+ title: 'Normalise a URL to its join key',
605
+ description: 'Return the canonical url_key used to join GSC data to crawl data.',
606
+ inputSchema: { url: z.string(), siteUrl: z.string().optional() },
607
+ }, async ({ url, siteUrl }) => {
608
+ const hostForm = siteUrl ? hostFormForProperty(siteUrl) : 'asis';
609
+ const key = urlKey(url, { hostForm });
610
+ return { content: [{ type: 'text', text: key }], structuredContent: { url, key, hostForm } };
611
+ });
612
+ // ── GSC ─────────────────────────────────────────────────────────────────
613
+ server.registerTool('list_properties', {
614
+ title: 'List GSC properties',
615
+ description: 'List Google Search Console properties accessible to the service account.',
616
+ inputSchema: {},
617
+ }, async () => {
618
+ const properties = await requireGsc(gsc).listProperties();
619
+ return {
620
+ content: [{ type: 'text', text: `${properties.length} properties: ${properties.map(p => p.siteUrl).join(', ')}` }],
621
+ structuredContent: { properties },
622
+ };
623
+ });
624
+ // refresh_property — the "sync everything" verb (GSC + crawl + inspection in one job).
625
+ // Skip flags let the same tool do "just update X".
626
+ registerAppTool(server, 'refresh_property', {
627
+ title: 'Refresh a property (sync + crawl + inspect)',
628
+ description: 'Full refresh for a property in one async job: GSC sync → site crawl → URL inspection → DataForSEO rank history. Opens a live progress widget (phases + counts). Set gsc/crawl/inspect/ranks=false to run just part. GSC sync is "lite" (date×query×page) by default — set segments=true to also pull device/country breakdowns (much heavier on large sites). GSC sync is INCREMENTAL: the first sync pulls the window (default last 90 days; pass startDate to go deeper), later syncs only fetch new days — set full=true to force a full re-pull. Use this for "sync everything"; use the single-purpose tools to update just one thing.',
629
+ inputSchema: {
630
+ siteUrl: z.string(),
631
+ gsc: z.boolean().optional(),
632
+ crawl: z.boolean().optional(),
633
+ inspect: z.boolean().optional(),
634
+ ranks: z.boolean().optional(),
635
+ segments: z.boolean().optional(),
636
+ full: z.boolean().optional(),
637
+ location: z.union([z.string(), z.number()]).optional(),
638
+ startDate: z.string().optional(),
639
+ endDate: z.string().optional(),
640
+ maxPages: z.number().int().min(1).max(50000).optional(),
641
+ inspectLimit: z.number().int().min(1).max(500).optional(),
642
+ },
643
+ _meta: { ui: { resourceUri: SYNC_PROGRESS_URI } },
644
+ }, async ({ siteUrl, gsc: doGsc, crawl, inspect, ranks, segments, full, location, startDate, endDate, maxPages, inspectLimit }) => {
645
+ const jobId = jobs.start('refresh', (update, signal) => refresh.run(siteUrl, { gsc: doGsc, crawl, inspect, ranks, segments, full, location, startDate, endDate, maxPages, inspectLimit }, update, signal));
646
+ return {
647
+ content: [{ type: 'text', text: `Refresh started for ${siteUrl} (job ${jobId}). Poll check_sync_status.` }],
648
+ structuredContent: { jobId, status: 'running', siteUrl },
649
+ };
650
+ });
651
+ server.registerTool('sync_gsc', {
652
+ title: 'Sync Search Console data (just GSC)',
653
+ description: 'Update just the GSC search-analytics history (async job — poll with check_sync_status). Incremental: the first sync pulls the range (default last 90 days; pass startDate to go deeper), later syncs only fetch new days — set full=true to force a full re-pull. Lite by default (date×query×page); set segments=true (or pass dimensions) to also pull device/country. For a full refresh use refresh_property. Grain: date × query × page. Joins: page_key (url_key) → crawl/inspection/backlinks; query → DataForSEO keyword tools.',
654
+ inputSchema: {
655
+ siteUrl: z.string(),
656
+ startDate: z.string().optional(),
657
+ endDate: z.string().optional(),
658
+ dimensions: z.array(z.string()).optional(),
659
+ segments: z.boolean().optional(),
660
+ full: z.boolean().optional(),
661
+ searchType: z.enum(['web', 'discover', 'googleNews', 'image', 'video']).optional(),
662
+ },
663
+ }, async ({ siteUrl, startDate, endDate, dimensions, segments, full, searchType }) => {
664
+ const syncer = requireGsc(sync);
665
+ const options = {
666
+ startDate: startDate ?? isoDaysAgo(90),
667
+ endDate: endDate ?? isoDaysAgo(0),
668
+ dimensions: dimensions ?? (segments ? FULL_DIMENSIONS : undefined),
669
+ fullResync: full,
670
+ searchType,
671
+ };
672
+ const jobId = jobs.start('sync', (update, signal) => syncer.run(siteUrl, options, update, signal));
673
+ return {
674
+ content: [{ type: 'text', text: `Sync started for ${siteUrl} (job ${jobId}). Poll check_sync_status.` }],
675
+ structuredContent: { jobId, status: 'running', siteUrl, ...options },
676
+ };
677
+ });
678
+ server.registerTool('check_sync_status', {
679
+ title: 'Check sync status',
680
+ description: 'Poll a sync job by id, or omit to list recent jobs.',
681
+ inputSchema: { jobId: z.string().optional() },
682
+ }, async ({ jobId }) => {
683
+ if (jobId) {
684
+ const job = jobs.get(jobId);
685
+ if (!job)
686
+ return { content: [{ type: 'text', text: `No job ${jobId}` }], structuredContent: { error: 'not_found', jobId } };
687
+ return { content: [{ type: 'text', text: `Job ${jobId}: ${job.state}` }], structuredContent: job };
688
+ }
689
+ const all = jobs.list();
690
+ return { content: [{ type: 'text', text: `${all.length} jobs` }], structuredContent: { jobs: all } };
691
+ });
692
+ server.registerTool('inspect_urls', {
693
+ title: 'Inspect URLs (just URL inspection)',
694
+ description: 'Update just the GSC URL Inspection data (coverage, indexing, Google-vs-declared canonical, last crawl) for top URLs into url_inspection. Quota-limited — samples top pages by clicks. Async job. For a full refresh use refresh_property. Grain: one row per inspected URL. Joins: url_key → pages/GSC.',
695
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(500).optional() },
696
+ }, async ({ siteUrl, limit }) => {
697
+ const insp = requireGsc(inspector);
698
+ const jobId = jobs.start('inspect', (update, signal) => insp.run(siteUrl, { limit }, update, signal));
699
+ return {
700
+ content: [{ type: 'text', text: `Inspection started for ${siteUrl} (job ${jobId}). Poll check_sync_status.` }],
701
+ structuredContent: { jobId, status: 'running', siteUrl },
702
+ };
703
+ });
704
+ // ── Crawl ─────────────────────────────────────────────────────────────────
705
+ server.registerTool('start_crawl', {
706
+ title: 'Crawl a site (just the crawl)',
707
+ description: 'Update just the crawl: fetch the site into the local database (async job — poll with check_crawl_status). HTTP crawl; respects robots.txt; asset file-types are HEAD-only (no body download); internal-search / cart / wp-json / builder junk URLs are skipped by default. For a full refresh use refresh_property. excludePatterns adds extra URL regexes to skip (e.g. ["/author/","/tag/","/page/"]) to keep big crawls light. Grain: one row per url_key, latest crawl. Joins: url_key → GSC/inspection/backlinks.',
708
+ inputSchema: {
709
+ siteUrl: z.string(),
710
+ maxPages: z.number().int().min(1).max(50000).optional(),
711
+ maxDepth: z.number().int().min(1).max(20).optional(),
712
+ maxConcurrency: z.number().int().min(1).max(16).optional(),
713
+ delayMs: z.number().int().min(0).max(10000).optional(),
714
+ userAgent: z.string().optional(),
715
+ excludePatterns: z.array(z.string()).optional(),
716
+ },
717
+ }, async ({ siteUrl, maxPages, maxDepth, maxConcurrency, delayMs, userAgent, excludePatterns }) => {
718
+ const jobId = jobs.start('crawl', (update, signal) => crawler.run(siteUrl, { maxPages, maxDepth, maxConcurrency, delayMs, userAgent, excludePatterns }, update, signal));
719
+ return {
720
+ content: [{ type: 'text', text: `Crawl started for ${siteUrl} (job ${jobId}). Poll check_crawl_status.` }],
721
+ structuredContent: { jobId, status: 'running', siteUrl },
722
+ };
723
+ });
724
+ server.registerTool('check_crawl_status', {
725
+ title: 'Check crawl status',
726
+ description: 'Poll a crawl job by id, or omit to list recent jobs.',
727
+ inputSchema: { jobId: z.string().optional() },
728
+ }, async ({ jobId }) => {
729
+ if (jobId) {
730
+ const job = jobs.get(jobId);
731
+ if (!job)
732
+ return { content: [{ type: 'text', text: `No job ${jobId}` }], structuredContent: { error: 'not_found', jobId } };
733
+ return { content: [{ type: 'text', text: `Job ${jobId}: ${job.state}` }], structuredContent: job };
734
+ }
735
+ const all = jobs.list();
736
+ return { content: [{ type: 'text', text: `${all.length} jobs` }], structuredContent: { jobs: all } };
737
+ });
738
+ // ── DataForSEO (cached 20 days, single-worker) ──────────────────────────
739
+ server.registerTool('keyword_volume', {
740
+ title: 'Keyword search volume (DataForSEO)',
741
+ description: 'True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
742
+ inputSchema: {
743
+ keywords: z.array(z.string()).min(1).max(700),
744
+ location: z.union([z.string(), z.number()]).optional(),
745
+ languageCode: z.string().optional(),
746
+ },
747
+ }, async ({ keywords, location, languageCode }) => {
748
+ const client = requireDfs(dfs);
749
+ const r = await client.searchVolume(keywords, location, languageCode);
750
+ const items = (r.tasks[0]?.result ?? []).map((k) => ({
751
+ keyword: k.keyword, searchVolume: k.search_volume, cpc: k.cpc, competition: k.competition,
752
+ }));
753
+ return {
754
+ content: [{ type: 'text', text: `${items.length} keywords${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
755
+ structuredContent: { keywords: items, cached: r.cached, cost: r.cost },
756
+ };
757
+ });
758
+ server.registerTool('related_terms', {
759
+ title: 'Related terms (People Also Ask + related searches)',
760
+ description: 'People Also Ask questions and related searches for a keyword (DataForSEO SERP). Powers click-through "related terms" on the keyword charts. SERP call — cached 20 days.',
761
+ inputSchema: { keyword: z.string(), location: z.union([z.string(), z.number()]).optional(), languageCode: z.string().optional() },
762
+ }, async ({ keyword, location, languageCode }) => {
763
+ const client = requireDfs(dfs);
764
+ const r = await client.relatedTerms(keyword, location, languageCode);
765
+ return {
766
+ content: [{ type: 'text', text: `PAA: ${r.peopleAlsoAsk.length}, related: ${r.relatedSearches.length}${r.cached ? ' (cached)' : ` ($${r.cost.toFixed(4)})`}` }],
767
+ structuredContent: r,
768
+ };
769
+ });
770
+ server.registerTool('suggest_pages', {
771
+ title: 'Suggest new pages from real demand (gaps)',
772
+ description: 'Propose NEW pages grounded in real Search Console demand: queries you already get impressions for but rank 11+ (no winning page), after subtracting demand you already satisfy (you rank ≤10, or a page already covers the query, or one URL owns it). Survivors are clustered into one proposed page per intent and scored by impressions × intent. Returns evidence + the nearest existing page to link the new page from. Computable from synced GSC + crawl — no paid calls (run search_intent siteUrl:<property> first to weight by intent).',
773
+ inputSchema: { siteUrl: z.string(), minImpressions: z.number().int().min(1).optional(), maxProposals: z.number().int().min(1).max(200).optional() },
774
+ }, async ({ siteUrl, minImpressions, maxProposals }) => {
775
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
776
+ try {
777
+ const r = suggestPages(db.db, { ...(minImpressions != null ? { minImpressions } : {}), ...(maxProposals != null ? { maxProposals } : {}) });
778
+ const top = r.proposals.slice(0, 15).map(p => `• "${p.headTerm}" — ${p.totalImpressions} impr, currently pos ${p.bestPosition}${p.intent ? `, ${p.intent}` : ''} (${p.queries.length} queries)`).join('\n');
779
+ const intentTip = r.proposals.length && r.proposals.every(p => !p.intent) ? `\n\nTip: run \`search_intent\` for this property first to rank these by search intent (currently unweighted).` : '';
780
+ return {
781
+ content: [{ type: 'text', text: r.proposals.length ? `${r.proposals.length} new-page opportunities (from ${r.consideredQueries} queries, ${r.afterDedup} after dedup):\n${top}${intentTip}` : `No new-page gaps found (${r.consideredQueries} queries considered). Needs synced GSC data.` }],
782
+ structuredContent: r,
783
+ };
784
+ }
785
+ finally {
786
+ db.close();
787
+ }
788
+ });
789
+ server.registerTool('search_intent', {
790
+ title: 'Search intent classification (DataForSEO Labs)',
791
+ description: 'Classify the search intent (informational / navigational / commercial / transactional) of keywords — primary label + probability + secondary intents. Use it to spot intent mismatch: e.g. a transactional product page ranking for an informational query (a common cause of high impressions / low CTR). Pass siteUrl to persist the intents so run_audit can surface intent-vs-pagetype-mismatch. Labs call, cached 20 days, up to 1000 keywords. Language-only (no location).',
792
+ inputSchema: { keywords: z.array(z.string()).min(1).max(1000), languageCode: z.string().optional(), siteUrl: z.string().optional() },
793
+ }, async ({ keywords, languageCode, siteUrl }) => {
794
+ const client = requireDfs(dfs);
795
+ const r = await client.searchIntent(keywords, languageCode);
796
+ const items = (r.tasks[0]?.result?.[0]?.items ?? []).map((it) => ({
797
+ keyword: it.keyword,
798
+ intent: it.keyword_intent?.label ?? null,
799
+ probability: it.keyword_intent?.probability ?? null,
800
+ secondary: (it.secondary_keyword_intents ?? []).map((s) => ({ intent: s.label, probability: s.probability })),
801
+ }));
802
+ let persisted = 0;
803
+ if (siteUrl) {
804
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
805
+ try {
806
+ const up = db.db.prepare(`INSERT INTO keyword_intent (keyword,intent,probability,fetched_at) VALUES (?,?,?,datetime('now'))
807
+ ON CONFLICT(keyword) DO UPDATE SET intent=excluded.intent, probability=excluded.probability, fetched_at=datetime('now')`);
808
+ db.db.transaction(() => { for (const it of items)
809
+ if (it.keyword && it.intent) {
810
+ up.run(it.keyword.toLowerCase(), it.intent, it.probability);
811
+ persisted++;
812
+ } })();
813
+ }
814
+ finally {
815
+ db.close();
816
+ }
817
+ }
818
+ return {
819
+ content: [{ type: 'text', text: `${items.length} keywords classified${persisted ? `, ${persisted} saved for ${siteUrl}` : ''}${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
820
+ structuredContent: { intents: items, persisted, cached: r.cached, cost: r.cost },
821
+ };
822
+ });
823
+ server.registerTool('page_lighthouse', {
824
+ title: 'Page Lighthouse — lab Core Web Vitals (DataForSEO On-Page)',
825
+ description: 'Run a live Lighthouse audit for ONE url: lab Core Web Vitals (LCP, CLS, TBT, FCP, Speed Index), category scores (performance/SEO/best-practices/accessibility), and the top time-saving opportunities. Complements GSC/CrUX field data (aggregate + delayed) with on-demand lab data. Pass siteUrl to persist the CWV so run_audit can surface high-yield-cwv-fail. Paid On-Page call (~2000 credits), slow (~20–120s), cached 20 days.',
826
+ inputSchema: { url: z.string().url(), forMobile: z.boolean().optional(), siteUrl: z.string().optional() },
827
+ }, async ({ url, forMobile, siteUrl }) => {
828
+ const client = requireDfs(dfs);
829
+ const mobile = forMobile ?? true;
830
+ const r = await client.lighthouse(url, mobile);
831
+ const res = r.tasks[0]?.result?.[0] ?? {};
832
+ const audits = res.audits ?? {};
833
+ const cwv = (id) => ({ score: audits[id]?.score ?? null, value: audits[id]?.displayValue ?? null });
834
+ const numeric = (id) => audits[id]?.numericValue ?? null;
835
+ const cats = res.categories ?? {};
836
+ const opportunities = Object.values(audits)
837
+ .filter((a) => a?.details?.type === 'opportunity' && (a.details.overallSavingsMs ?? 0) > 0)
838
+ .sort((a, b) => (b.details.overallSavingsMs ?? 0) - (a.details.overallSavingsMs ?? 0))
839
+ .slice(0, 8)
840
+ .map((a) => ({ id: a.id, title: a.title, savingsMs: Math.round(a.details.overallSavingsMs) }));
841
+ const out = {
842
+ url, forMobile: mobile,
843
+ scores: {
844
+ performance: cats.performance?.score ?? null,
845
+ seo: cats.seo?.score ?? null,
846
+ bestPractices: cats['best-practices']?.score ?? null,
847
+ accessibility: cats.accessibility?.score ?? null,
848
+ },
849
+ coreWebVitals: {
850
+ lcp: cwv('largest-contentful-paint'), cls: cwv('cumulative-layout-shift'),
851
+ tbt: cwv('total-blocking-time'), fcp: cwv('first-contentful-paint'), speedIndex: cwv('speed-index'),
852
+ },
853
+ opportunities,
854
+ cached: r.cached, cost: r.cost,
855
+ };
856
+ let persisted = false;
857
+ if (siteUrl && cats.performance) {
858
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
859
+ try {
860
+ db.db.prepare(`INSERT INTO page_cwv (url_key,url,for_mobile,performance,lcp_ms,cls,tbt_ms,fetched_at)
861
+ VALUES (?,?,?,?,?,?,?,datetime('now'))
862
+ ON CONFLICT(url_key) DO UPDATE SET url=excluded.url, for_mobile=excluded.for_mobile, performance=excluded.performance,
863
+ lcp_ms=excluded.lcp_ms, cls=excluded.cls, tbt_ms=excluded.tbt_ms, fetched_at=datetime('now')`)
864
+ .run(urlKey(url, { hostForm: hostFormForProperty(siteUrl) }), url, mobile ? 1 : 0, out.scores.performance, numeric('largest-contentful-paint'), numeric('cumulative-layout-shift'), numeric('total-blocking-time'));
865
+ persisted = true;
866
+ }
867
+ finally {
868
+ db.close();
869
+ }
870
+ }
871
+ const perf = out.scores.performance != null ? Math.round(out.scores.performance * 100) : '?';
872
+ return {
873
+ content: [{ type: 'text', text: `Lighthouse ${url}: perf ${perf}/100, LCP ${out.coreWebVitals.lcp.value ?? '?'}, CLS ${out.coreWebVitals.cls.value ?? '?'}${persisted ? ` (saved for ${siteUrl})` : ''}${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
874
+ structuredContent: { ...out, persisted },
875
+ };
876
+ });
877
+ server.registerTool('competitors_domain', {
878
+ title: 'Competitor domains (DataForSEO Labs)',
879
+ description: 'Discover the domains competing with a target for the same organic keywords (ranked by keyword overlap), with intersection counts and organic traffic estimates. The seed list for content-gap analysis (feed these into page_intersection). Labs call, cached 20 days. Pass location + language for the right market.',
880
+ inputSchema: { target: z.string(), location: z.union([z.string(), z.number()]).optional(), languageCode: z.string().optional(), limit: z.number().int().min(1).max(100).optional() },
881
+ }, async ({ target, location, languageCode, limit }) => {
882
+ const client = requireDfs(dfs);
883
+ const cleaned = target.replace(/^sc-domain:/, '').replace(/^https?:\/\//, '').replace(/^www\./, '').replace(/\/+$/, '');
884
+ const r = await client.competitorsDomain(cleaned, location, languageCode, limit ?? 20);
885
+ const items = (r.tasks[0]?.result?.[0]?.items ?? [])
886
+ .filter((it) => it.domain && it.domain.replace(/^www\./, '') !== cleaned) // drop the target itself
887
+ .map((it) => ({
888
+ domain: it.domain,
889
+ intersections: it.intersections ?? null,
890
+ avgPosition: it.avg_position ?? null,
891
+ organicKeywords: it.full_domain_metrics?.organic?.count ?? null,
892
+ organicEtv: it.full_domain_metrics?.organic?.etv ?? null,
893
+ }));
894
+ return {
895
+ content: [{ type: 'text', text: `${items.length} competitor domains for ${cleaned}${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
896
+ structuredContent: { target: cleaned, competitors: items, cached: r.cached, cost: r.cost },
897
+ };
898
+ });
899
+ server.registerTool('page_intersection', {
900
+ title: 'Content gap — page intersection (DataForSEO Labs)',
901
+ description: 'Find keywords that competitor pages rank for but your page does NOT (the content gap). Pass competitor URLs as `competitorUrls` (max 20; wildcards like https://site.com/blog/* allowed) and your own URL(s) in `excludePages` (max 10) to subtract. Returns gap keywords sorted by search volume, with each competitor’s rank. Labs call, cached 20 days. Pass location + language.',
902
+ inputSchema: {
903
+ competitorUrls: z.array(z.string()).min(1).max(20),
904
+ excludePages: z.array(z.string()).max(10).optional(),
905
+ location: z.union([z.string(), z.number()]).optional(),
906
+ languageCode: z.string().optional(),
907
+ limit: z.number().int().min(1).max(1000).optional(),
908
+ },
909
+ }, async ({ competitorUrls, excludePages, location, languageCode, limit }) => {
910
+ const client = requireDfs(dfs);
911
+ const r = await client.pageIntersection(competitorUrls, excludePages ?? [], location, languageCode, limit ?? 100);
912
+ const items = (r.tasks[0]?.result?.[0]?.items ?? [])
913
+ .map((it) => {
914
+ const kd = it.keyword_data ?? {}; // actual shape: item.keyword_data.{keyword,keyword_info}
915
+ return {
916
+ keyword: kd.keyword ?? null,
917
+ searchVolume: kd.keyword_info?.search_volume ?? null,
918
+ cpc: kd.keyword_info?.cpc ?? null,
919
+ competition: kd.keyword_info?.competition ?? null,
920
+ ranks: Object.entries(it.intersection_result ?? {}).map(([idx, v]) => ({
921
+ page: Number(idx), rank: v?.rank_absolute ?? null, url: v?.url ?? null,
922
+ })),
923
+ };
924
+ })
925
+ .sort((a, b) => (b.searchVolume ?? 0) - (a.searchVolume ?? 0)); // order_by unsupported on this endpoint
926
+ return {
927
+ content: [{ type: 'text', text: `${items.length} gap keywords${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
928
+ structuredContent: { gaps: items, cached: r.cached, cost: r.cost },
929
+ };
930
+ });
931
+ // Strip any scheme/sc-domain:/path down to a bare host for Labs domain targets.
932
+ // Always strips leading www. — Labs treats www.example.com as a subdomain, not the domain.
933
+ const dfsHost = (t) => t.replace(/^sc-domain:/, '').replace(/^https?:\/\//, '').replace(/\/.*$/, '').replace(/^www\./, '').trim();
934
+ // Cap a markdown table well under the ~40k model-facing ceiling: keep whole rows, note the rest.
935
+ const capMdRows = (header, rows, footer = '', maxChars = 30000) => {
936
+ let out = header;
937
+ let used = 0;
938
+ for (; used < rows.length; used++) {
939
+ if (out.length + rows[used].length + 1 + footer.length > maxChars)
940
+ break;
941
+ out += '\n' + rows[used];
942
+ }
943
+ if (used < rows.length)
944
+ out += `\n\n_+${rows.length - used} more rows (truncated for length — lower the limit or use structuredContent)._`;
945
+ return out + footer;
946
+ };
947
+ const fmtNum = (v) => {
948
+ const x = Number(v);
949
+ return Number.isFinite(x) ? Math.round(x).toLocaleString('en-US') : '—';
950
+ };
951
+ server.registerTool('domain_visibility', {
952
+ title: 'Domain visibility over time (DataForSEO Labs)',
953
+ description: 'Monthly organic visibility for ANY domain or subdomain — no Search Console access needed: ranking-keyword totals, position distribution (1–3 / 4–10 / 11–20 / 21–100) and estimated traffic value (ETV) per month, plus a trend verdict. The Semrush-style "organic overview / visibility over time" for you or a competitor. Labs call, cached 20 days. Pass location (name or code) for the right market. Grain: month × domain. Joins: domain → Labs/backlinks tools; period → rank_history.',
954
+ inputSchema: {
955
+ target: z.string().describe('Domain or subdomain, no scheme (e.g. example.com or blog.example.com)'),
956
+ location: z.union([z.string(), z.number()]).optional(),
957
+ languageName: z.string().optional(),
958
+ months: z.number().int().min(1).max(24).optional(),
959
+ },
960
+ }, async ({ target, location, languageName, months }) => {
961
+ const client = requireDfs(dfs);
962
+ const cleaned = dfsHost(target);
963
+ const r = await client.historicalRankOverview(cleaned, location, 'en', languageName);
964
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
965
+ const n = (v) => Number(v) || 0;
966
+ // Guard year/month — a malformed item would emit an "undefined-NaN" period row.
967
+ const series = items
968
+ .filter((it) => Number.isInteger(it?.year) && Number.isInteger(it?.month))
969
+ .map((it) => {
970
+ const o = it.metrics?.organic ?? {};
971
+ const pos_1_3 = n(o.pos_1) + n(o.pos_2_3);
972
+ const pos_4_10 = n(o.pos_4_10);
973
+ const pos_11_20 = n(o.pos_11_20);
974
+ const pos_21_100 = n(o.pos_21_30) + n(o.pos_31_40) + n(o.pos_41_50) + n(o.pos_51_60) +
975
+ n(o.pos_61_70) + n(o.pos_71_80) + n(o.pos_81_90) + n(o.pos_91_100);
976
+ return {
977
+ period: `${it.year}-${String(it.month).padStart(2, '0')}`,
978
+ keywords: n(o.count) || (pos_1_3 + pos_4_10 + pos_11_20 + pos_21_100),
979
+ pos_1_3, pos_4_10, pos_11_20, pos_21_100,
980
+ etv: Math.round(n(o.etv) * 100) / 100,
981
+ isNew: n(o.is_new), isLost: n(o.is_lost), isUp: n(o.is_up), isDown: n(o.is_down),
982
+ };
983
+ })
984
+ .sort((a, b) => (a.period < b.period ? -1 : 1))
985
+ .slice(-(months ?? 12));
986
+ if (!series.length) {
987
+ return {
988
+ content: [{ type: 'text', text: `No historical rank data for ${cleaned} — check the target (bare domain/subdomain) and location.` }],
989
+ structuredContent: { target: cleaned, series: [], cached: r.cached, cost: r.cost },
990
+ };
991
+ }
992
+ const first = series[0];
993
+ const last = series[series.length - 1];
994
+ const delta = first.etv > 0 ? ((last.etv - first.etv) / first.etv) * 100 : (last.etv > 0 ? 100 : 0);
995
+ const verdict = Math.abs(delta) < 10
996
+ ? `flat (ETV ${fmtNum(first.etv)} → ${fmtNum(last.etv)}, ${delta >= 0 ? '+' : ''}${delta.toFixed(0)}% ${first.period} → ${last.period})`
997
+ : `${delta > 0 ? 'rising' : 'declining'} (ETV ${fmtNum(first.etv)} → ${fmtNum(last.etv)}, ${delta > 0 ? '+' : ''}${delta.toFixed(0)}% ${first.period} → ${last.period})`;
998
+ const header = `**${cleaned}** — organic visibility, last ${series.length} months. Trend: ${verdict}\n\n` +
999
+ `| Period | Keywords | Pos 1–3 | Pos 4–10 | Pos 11–20 | Pos 21–100 | New | Lost | ETV |\n|---|---|---|---|---|---|---|---|---|`;
1000
+ const rows = series.map(s => `| ${s.period} | ${fmtNum(s.keywords)} | ${fmtNum(s.pos_1_3)} | ${fmtNum(s.pos_4_10)} | ${fmtNum(s.pos_11_20)} | ${fmtNum(s.pos_21_100)} | ${fmtNum(s.isNew)} | ${fmtNum(s.isLost)} | ${fmtNum(s.etv)} |`);
1001
+ const md = capMdRows(header, rows, `\n\n${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}).`}`);
1002
+ return {
1003
+ content: [{ type: 'text', text: md }],
1004
+ structuredContent: { target: cleaned, months: series.length, verdict, series, cached: r.cached, cost: r.cost },
1005
+ };
1006
+ });
1007
+ server.registerTool('top_pages', {
1008
+ title: 'Top ranking pages on a domain (DataForSEO Labs)',
1009
+ description: 'The top organic pages of ANY domain or subdomain, ranked by estimated traffic value (ETV): page, ranking-keyword count, ETV and top-3 / top-10 keyword counts. The Semrush-style "top pages" view — works on competitors, no crawl or GSC needed. Labs call, cached 20 days. Pass location for the right market. Grain: page × domain. Joins: domain → Labs tools; URL → pages.url_key on your own property.',
1010
+ inputSchema: {
1011
+ target: z.string().describe('Domain or subdomain, no scheme'),
1012
+ location: z.union([z.string(), z.number()]).optional(),
1013
+ languageCode: z.string().optional(),
1014
+ limit: z.number().int().min(1).max(100).optional(),
1015
+ },
1016
+ }, async ({ target, location, languageCode, limit }) => {
1017
+ const client = requireDfs(dfs);
1018
+ const cleaned = dfsHost(target);
1019
+ const r = await client.relevantPages(cleaned, location, languageCode ?? 'en', limit ?? 25);
1020
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1021
+ const n = (v) => Number(v) || 0;
1022
+ const pages = items
1023
+ .map((it) => {
1024
+ const page = it.page_address ?? it.relative_url ?? null;
1025
+ if (!page)
1026
+ return null; // skip malformed rows
1027
+ const o = it.metrics?.organic ?? {};
1028
+ const top3 = n(o.pos_1) + n(o.pos_2_3);
1029
+ return {
1030
+ page,
1031
+ keywords: n(o.count),
1032
+ etv: Math.round(n(o.etv) * 100) / 100,
1033
+ top3,
1034
+ top10: top3 + n(o.pos_4_10),
1035
+ };
1036
+ })
1037
+ .filter((p) => p != null)
1038
+ .sort((a, b) => b.etv - a.etv);
1039
+ if (!pages.length) {
1040
+ return {
1041
+ content: [{ type: 'text', text: `No ranking pages found for ${cleaned} — check the target and location.` }],
1042
+ structuredContent: { target: cleaned, pages: [], cached: r.cached, cost: r.cost },
1043
+ };
1044
+ }
1045
+ const header = `**${cleaned}** — top ${pages.length} pages by estimated organic traffic\n\n` +
1046
+ `| # | Page | Keywords | ETV | Top 3 | Top 10 |\n|---|---|---|---|---|---|`;
1047
+ const rows = pages.map((p, i) => `| ${i + 1} | ${String(p.page).replace(/\|/g, '\\|')} | ${fmtNum(p.keywords)} | ${fmtNum(p.etv)} | ${fmtNum(p.top3)} | ${fmtNum(p.top10)} |`);
1048
+ const md = capMdRows(header, rows, `\n\n${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}).`}`);
1049
+ return {
1050
+ content: [{ type: 'text', text: md }],
1051
+ structuredContent: { target: cleaned, rowsTotal: pages.length, pages: pages.slice(0, 100), cached: r.cached, cost: r.cost },
1052
+ };
1053
+ });
1054
+ server.registerTool('ranked_keywords', {
1055
+ title: 'Ranked keywords for a domain / URL / folder (DataForSEO Labs)',
1056
+ description: 'Every keyword a target ranks for in Google organic — scope it to a whole domain, a subdomain, ONE page (scope:url with the full URL), or a subfolder (scope:folder + folder:"/blog/"). Returns keyword, position, search volume, ETV, the ranking URL, plus keyword difficulty, search intent and SERP features where the response carries them. Set aioOnly:true to list only keywords where the target is CITED as a source in Google AI Overviews (the "which keywords cite us in AIO" view). The Semrush-style "keywords a page or site ranks for" view — works on any site. Labs call, cached 20 days. Pass location for the right market. Grain: keyword × target. Joins: keyword → GSC search_analytics.query; URL → pages.url_key.',
1057
+ inputSchema: {
1058
+ target: z.string().describe('Domain, subdomain, or full URL (full URL required for scope:url)'),
1059
+ scope: z.enum(['domain', 'subdomain', 'url', 'folder']).optional(),
1060
+ folder: z.string().optional().describe('Required when scope is folder, e.g. "/blog/"'),
1061
+ location: z.union([z.string(), z.number()]).optional(),
1062
+ languageCode: z.string().optional(),
1063
+ limit: z.number().int().min(1).max(200).optional(),
1064
+ orderBy: z.enum(['etv', 'volume', 'position']).optional(),
1065
+ aioOnly: z.boolean().optional().describe('Only keywords where the target is cited as an AI Overview source (Labs item_types:["ai_overview"])'),
1066
+ },
1067
+ }, async ({ target, scope, folder, location, languageCode, limit, orderBy, aioOnly }) => {
1068
+ const client = requireDfs(dfs);
1069
+ const mode = scope ?? 'domain';
1070
+ if (mode === 'folder' && !folder?.trim())
1071
+ throw new Error('scope:"folder" requires the `folder` input (e.g. "/blog/").');
1072
+ const dfsTarget = mode === 'url'
1073
+ ? (/^https?:\/\//i.test(target) ? target : `https://${target}`)
1074
+ : dfsHost(target);
1075
+ const order = orderBy === 'volume'
1076
+ ? 'keyword_data.keyword_info.search_volume,desc'
1077
+ : orderBy === 'position'
1078
+ ? 'ranked_serp_element.serp_item.rank_absolute,asc'
1079
+ : 'ranked_serp_element.serp_item.etv,desc';
1080
+ const folderPath = mode === 'folder'
1081
+ ? (folder.trim().startsWith('/') ? folder.trim() : '/' + folder.trim())
1082
+ : null;
1083
+ const filters = folderPath
1084
+ ? [['ranked_serp_element.serp_item.relative_url', 'like', `${folderPath}%`]]
1085
+ : undefined;
1086
+ // aioOnly: Labs returns only rows where the target appears as an ai_overview_reference —
1087
+ // i.e. keywords where the domain is CITED as a source in the AI Overview (plan/data-utilisation.md 1a route 2).
1088
+ const itemTypes = aioOnly ? ['ai_overview'] : undefined;
1089
+ const r = await client.rankedKeywords(dfsTarget, location, languageCode ?? 'en', limit ?? 50, order, filters, itemTypes);
1090
+ const result = r.tasks[0]?.result?.[0] ?? {};
1091
+ const items = result.items ?? [];
1092
+ const kws = items
1093
+ .map((it) => {
1094
+ const kd = it.keyword_data ?? {};
1095
+ const serp = it.ranked_serp_element?.serp_item ?? {};
1096
+ if (!kd.keyword)
1097
+ return null; // skip malformed rows
1098
+ const serpFeatures = Array.isArray(kd.serp_info?.serp_item_types) ? kd.serp_info.serp_item_types : null;
1099
+ return {
1100
+ keyword: kd.keyword,
1101
+ position: Number(serp.rank_absolute) || null,
1102
+ searchVolume: kd.keyword_info?.search_volume ?? null,
1103
+ etv: serp.etv != null ? Math.round(Number(serp.etv) * 100) / 100 : null,
1104
+ url: serp.relative_url ?? serp.url ?? null,
1105
+ // Fields we already pay for but previously dropped (plan/data-utilisation.md Part 2 §4):
1106
+ keywordDifficulty: kd.keyword_properties?.keyword_difficulty ?? null,
1107
+ intent: kd.search_intent_info?.main_intent ?? null,
1108
+ serpFeatures,
1109
+ isFeaturedSnippet: serp.is_featured_snippet === true ? true : null,
1110
+ };
1111
+ })
1112
+ .filter((k) => k != null);
1113
+ const scopeLabel = folderPath ? `${dfsTarget}${folderPath}*` : dfsTarget;
1114
+ if (!kws.length) {
1115
+ return {
1116
+ content: [{ type: 'text', text: aioOnly
1117
+ ? `No AI Overview citations found for ${scopeLabel} (scope: ${mode}) — the target isn't cited as an AIO source for any tracked keyword in this market.`
1118
+ : `No ranked keywords found for ${scopeLabel} (scope: ${mode}) — check the target, scope and location.` }],
1119
+ structuredContent: { target: dfsTarget, scope: mode, aioOnly: aioOnly ?? false, keywords: [], cached: r.cached, cost: r.cost },
1120
+ };
1121
+ }
1122
+ const totalCount = Number(result.total_count) || kws.length;
1123
+ const sumEtv = kws.reduce((s, k) => s + (k.etv ?? 0), 0);
1124
+ // Optional columns — only where the response actually carries the field (older cache entries won't).
1125
+ const hasKd = kws.some(k => k.keywordDifficulty != null);
1126
+ const hasIntent = kws.some(k => k.intent != null);
1127
+ const hasFeatures = kws.some(k => (k.serpFeatures && k.serpFeatures.length) || k.isFeaturedSnippet);
1128
+ const optHead = `${hasKd ? ' KD |' : ''}${hasIntent ? ' Intent |' : ''}${hasFeatures ? ' SERP features |' : ''}`;
1129
+ const optSep = `${hasKd ? '---|' : ''}${hasIntent ? '---|' : ''}${hasFeatures ? '---|' : ''}`;
1130
+ const header = `**${scopeLabel}** (scope: ${mode})${aioOnly ? ' — AI Overview citations only' : ''} — ${kws.length} of ${fmtNum(totalCount)} ${aioOnly ? 'citing keywords' : 'ranked keywords'}, ordered by ${orderBy ?? 'etv'}\n\n` +
1131
+ `| Keyword | Pos | Volume | ETV |${optHead} URL |\n|---|---|---|---|${optSep}---|`;
1132
+ const featCell = (k) => {
1133
+ const f = (k.serpFeatures ?? []).map(t => (t === 'featured_snippet' && k.isFeaturedSnippet ? 'featured_snippet (owned)' : t));
1134
+ if (k.isFeaturedSnippet && !f.some(t => t.startsWith('featured_snippet')))
1135
+ f.unshift('featured_snippet (owned)');
1136
+ return f.length ? f.join(', ') : '—';
1137
+ };
1138
+ const rows = kws.map(k => `| ${k.keyword.replace(/\|/g, '\\|')} | ${k.position ?? '—'} | ${fmtNum(k.searchVolume)} | ${fmtNum(k.etv)} |` +
1139
+ `${hasKd ? ` ${k.keywordDifficulty ?? '—'} |` : ''}${hasIntent ? ` ${k.intent ?? '—'} |` : ''}${hasFeatures ? ` ${featCell(k).replace(/\|/g, '\\|')} |` : ''}` +
1140
+ ` ${k.url ? String(k.url).replace(/\|/g, '\\|') : '—'} |`);
1141
+ const md = capMdRows(header, rows, `\n\nTotals: ${fmtNum(totalCount)} keywords ${aioOnly ? 'citing the target in AI Overviews' : 'ranked'}, listed rows sum ${fmtNum(sumEtv)} ETV. ${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}).`}`);
1142
+ return {
1143
+ content: [{ type: 'text', text: md }],
1144
+ structuredContent: { target: dfsTarget, scope: mode, aioOnly: aioOnly ?? false, totalCount, rowsTotal: kws.length, keywords: kws.slice(0, 100), cached: r.cached, cost: r.cost },
1145
+ };
1146
+ });
1147
+ server.registerTool('topic_gaps', {
1148
+ title: 'Topic gaps — what to cover to be expert in your space (Labs + GSC)',
1149
+ description: 'What related topics should this site cover to be seen as expert in its space? Pulls competitor keyword footprints (DataForSEO Labs ranked_keywords, ≤4 bounded + 20-day-cached calls), subtracts everything YOU already surface for (every GSC query with impressions) or already have a page about (title/H1/slug near-match), clusters the surviving gap keywords lexically, and scores each topic by summed search volume × how many competitors rank there. Each topic names the owning competitor with an example URL, your nearest existing page to build from, and (when resolve_entities has run) whether it sits beside entities you already cover. Pass competitors explicitly (max 3) or let it derive the top 2 from ranking overlap. Needs synced GSC data + DataForSEO credentials.',
1150
+ inputSchema: {
1151
+ siteUrl: z.string(),
1152
+ competitors: z.array(z.string()).max(3).optional().describe('Competitor domains (max 3, e.g. ["rival.com"]). Omitted → derived via competitors_domain'),
1153
+ location: z.union([z.string(), z.number()]).optional(),
1154
+ limit: z.number().int().min(1).max(50).optional().describe('Topics to return (default 15)'),
1155
+ },
1156
+ }, async ({ siteUrl, competitors, location, limit }) => {
1157
+ if (!dfs) {
1158
+ return {
1159
+ content: [{ type: 'text', text: 'topic_gaps needs DataForSEO credentials (set DATAFORSEO_USERNAME / DATAFORSEO_PASSWORD in your MCP config). The rest of the audit works without them — this tool compares competitor keyword footprints, which requires DataForSEO Labs.' }],
1160
+ structuredContent: { error: 'no_dataforseo_credentials' },
1161
+ };
1162
+ }
1163
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1164
+ try {
1165
+ const gscRows = db.db.prepare('SELECT COUNT(*) n FROM search_analytics').get().n;
1166
+ if (!gscRows) {
1167
+ return {
1168
+ content: [{ type: 'text', text: `No synced Search Console data for ${siteUrl} — run refresh_property (or sync_gsc) first. topic_gaps subtracts your real GSC footprint; without it every competitor keyword would look like a gap.` }],
1169
+ structuredContent: { error: 'no_gsc_data', siteUrl },
1170
+ };
1171
+ }
1172
+ const loc = location ?? db.getDfsLocation(siteUrl) ?? undefined;
1173
+ const ourHost = dfsHost(siteUrl);
1174
+ let totalCost = 0, liveCalls = 0, cachedCalls = 0;
1175
+ const tally = (r) => { totalCost += r.cost; r.cached ? cachedCalls++ : liveCalls++; };
1176
+ // Competitors: explicit (≤3) or derived top-2 by keyword overlap. Total Labs
1177
+ // budget stays ≤4 calls: at most 1 discovery + at most 3 footprint pulls.
1178
+ let comps = [...new Set((competitors ?? []).map(dfsHost).filter(c => c && c !== ourHost))].slice(0, 3);
1179
+ let derived = false;
1180
+ if (!comps.length) {
1181
+ const r = await dfs.competitorsDomain(ourHost, loc, 'en', 10);
1182
+ tally(r);
1183
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1184
+ comps = items
1185
+ .map((it) => String(it.domain ?? '').replace(/^www\./, ''))
1186
+ .filter(d => d && d !== ourHost)
1187
+ .slice(0, 2);
1188
+ derived = true;
1189
+ if (!comps.length) {
1190
+ return {
1191
+ content: [{ type: 'text', text: `No competitor domains derivable for ${ourHost} in this market — pass competitors explicitly (e.g. competitors:["rival.com"]).` }],
1192
+ structuredContent: { error: 'no_competitors', target: ourHost, cost: totalCost },
1193
+ };
1194
+ }
1195
+ }
1196
+ // One ranked_keywords pull per competitor (top 1000 rows by ETV — their money keywords).
1197
+ const rows = [];
1198
+ const perCompetitor = {};
1199
+ for (const c of comps) {
1200
+ const r = await dfs.rankedKeywords(c, loc, 'en', 1000, 'ranked_serp_element.serp_item.etv,desc');
1201
+ tally(r);
1202
+ const items = r.tasks[0]?.result?.[0]?.items ?? [];
1203
+ for (const it of items) {
1204
+ const kd = it.keyword_data ?? {};
1205
+ const serp = it.ranked_serp_element?.serp_item ?? {};
1206
+ if (!kd.keyword)
1207
+ continue;
1208
+ rows.push({
1209
+ keyword: String(kd.keyword),
1210
+ searchVolume: kd.keyword_info?.search_volume ?? null,
1211
+ etv: serp.etv != null ? Number(serp.etv) : null,
1212
+ position: Number(serp.rank_absolute) || null,
1213
+ url: serp.url ?? (serp.relative_url ? `https://${c}${serp.relative_url}` : null),
1214
+ competitor: c,
1215
+ });
1216
+ }
1217
+ perCompetitor[c] = items.length;
1218
+ }
1219
+ const res = computeTopicGaps(db.db, rows, { limit: limit ?? 15 });
1220
+ const costLine = `Labs calls: ${liveCalls} live ($${totalCost.toFixed(4)}) + ${cachedCalls} cached${derived ? ' (competitors derived via competitors_domain)' : ''}.`;
1221
+ if (!res.clusters.length) {
1222
+ return {
1223
+ content: [{ type: 'text', text: `No topic gaps found: ${res.competitorKeywords} competitor keywords (${comps.join(', ')}) all fell inside your existing footprint (${res.ourQueryCount} GSC queries + page titles/H1s). Either you genuinely cover the space, or try different competitors. ${costLine}` }],
1224
+ structuredContent: { target: ourHost, competitors: comps, clustersTotal: 0, clusters: [], competitorKeywords: res.competitorKeywords, ourQueryCount: res.ourQueryCount, afterSubtraction: res.afterSubtraction, cost: totalCost, liveCalls, cachedCalls },
1225
+ };
1226
+ }
1227
+ const fmtVol = (v) => v.toLocaleString('en-US');
1228
+ const sections = res.clusters.map((c, i) => {
1229
+ const shown = c.keywords.slice(0, 6);
1230
+ const more = c.keywords.length - shown.length;
1231
+ const kwLine = shown.map(k => `\`${k.keyword}\`${k.searchVolume ? ` (${fmtVol(k.searchVolume)})` : ''}`).join(', ') + (more > 0 ? ` — +${more} more keywords in this cluster` : '');
1232
+ return [
1233
+ `### ${i + 1}. ${c.headTerm} — ~${fmtVol(c.totalVolume)}/mo · ${c.competitorCoverage} competitor${c.competitorCoverage === 1 ? '' : 's'} · owned by ${c.ownedBy}`,
1234
+ `- Why: ${c.why}`,
1235
+ `- Keywords: ${kwLine}`,
1236
+ c.exampleUrl ? `- Competitor example: ${c.exampleUrl}` : null,
1237
+ c.nearestExistingPage ? `- Your nearest page: ${c.nearestExistingPage.title ?? '(untitled)'} — ${c.nearestExistingPage.url}` : `- Your nearest page: none found — this is greenfield for you`,
1238
+ c.entityNote ? `- Entity graph: ${c.entityNote}` : null,
1239
+ ].filter(Boolean).join('\n');
1240
+ });
1241
+ const md = `## Topic gaps for ${ourHost} vs ${comps.join(' + ')}\n\n` +
1242
+ `${res.competitorKeywords} distinct competitor keywords → ${res.afterSubtraction} survive subtraction of your footprint (${fmtVol(res.ourQueryCount)} GSC queries + page titles/H1s) → top ${res.clusters.length} topic clusters by volume × competitor coverage.\n\n` +
1243
+ sections.join('\n\n') +
1244
+ `\n\n_${costLine} Competitor rows: ${comps.map(c => `${c} ${perCompetitor[c] ?? 0}`).join(', ')}._`;
1245
+ // structuredContent: keyword lists capped at 10 per cluster with explicit totals
1246
+ // (host ~60k model-facing ceiling) — the counts always state the true size.
1247
+ const scClusters = res.clusters.map(c => ({
1248
+ ...c,
1249
+ keywords: c.keywords.slice(0, 10),
1250
+ keywordsTotal: c.keywords.length,
1251
+ }));
1252
+ return {
1253
+ content: [{ type: 'text', text: md }],
1254
+ structuredContent: {
1255
+ target: ourHost, competitors: comps, derivedCompetitors: derived,
1256
+ clustersTotal: res.clusters.length, clusters: scClusters,
1257
+ competitorKeywords: res.competitorKeywords, ourQueryCount: res.ourQueryCount, afterSubtraction: res.afterSubtraction,
1258
+ cost: totalCost, liveCalls, cachedCalls,
1259
+ },
1260
+ };
1261
+ }
1262
+ finally {
1263
+ db.close();
1264
+ }
1265
+ });
1266
+ // ── Dashboard (MCP App UI — houtini design + ECharts) ───────────────────
1267
+ registerAppTool(server, 'get_dashboard', {
1268
+ title: 'SEO dashboard',
1269
+ description: 'Interactive dashboard for a property: summary metrics, rank & clicks over time, and top-keyword performance (click a keyword for related terms). Needs synced GSC data — run refresh_property first.',
1270
+ inputSchema: { siteUrl: z.string() },
1271
+ _meta: { ui: { resourceUri: DASHBOARD_URI } },
1272
+ }, async ({ siteUrl }) => {
1273
+ // Return only a TINY model-facing result + the siteUrl; the widget fetches the full
1274
+ // (large) dataset itself via the app-only get_dashboard_data tool, which keeps the
1275
+ // big payload OUT of the model's context/token limit (per the MCP Apps large-data pattern).
1276
+ const data = getDashboardData(dataDir(), siteUrl);
1277
+ if (data.empty) {
1278
+ return { content: [{ type: 'text', text: `No synced data for ${siteUrl} yet — run refresh_property.` }], structuredContent: { siteUrl, empty: true } };
1279
+ }
1280
+ const c = data.summary?.current;
1281
+ const summary = `Dashboard opened for ${siteUrl} — ${c?.clicks ?? 0} clicks / ${c?.impressions ?? 0} impressions (last 28d)` +
1282
+ `${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.`;
1283
+ return { content: [{ type: 'text', text: summary }], structuredContent: { siteUrl } };
1284
+ });
1285
+ // App-only data tool: the dashboard widget calls this via app.callServerTool to fetch its
1286
+ // full dataset. visibility:['app'] hides it from the model; results route to the iframe,
1287
+ // bypassing the model token cap that a large model-facing result would hit.
1288
+ registerAppTool(server, 'get_dashboard_data', {
1289
+ title: 'Dashboard data (internal)',
1290
+ description: 'Full dashboard dataset for the UI widget. App-only — not for direct use.',
1291
+ inputSchema: { siteUrl: z.string() },
1292
+ _meta: { ui: { visibility: ['app'] } },
1293
+ }, async ({ siteUrl }) => {
1294
+ const data = getDashboardData(dataDir(), siteUrl);
1295
+ return { content: [{ type: 'text', text: 'ok' }], structuredContent: data };
1296
+ });
1297
+ // export_report — the dependable deliverable: a self-contained interactive dashboard
1298
+ // HTML (data inlined) the user opens in any browser / emails to a client. Works
1299
+ // regardless of whether the host renders MCP-App widgets inline.
1300
+ server.registerTool('export_report', {
1301
+ title: 'Export a shareable dashboard report (HTML)',
1302
+ description: 'Write a self-contained, interactive dashboard HTML for a property (all data + charts inlined) to the reports folder, and return the file path. Open it in any browser or send it to a client — no server, no MCP-App host support needed. Run refresh_property (+ run_audit for findings) first.',
1303
+ inputSchema: { siteUrl: z.string(), theme: z.enum(['light', 'dark']).optional() },
1304
+ }, async ({ siteUrl, theme }) => {
1305
+ const data = getDashboardData(dataDir(), siteUrl);
1306
+ if (data.empty) {
1307
+ return { content: [{ type: 'text', text: `No synced data for ${siteUrl} — run refresh_property first.` }], structuredContent: { error: 'empty', siteUrl } };
1308
+ }
1309
+ const tpl = readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8');
1310
+ const json = JSON.stringify(data).replace(/</g, '\\u003c'); // prevent </script> breakout
1311
+ const inject = `<script>window.__DASH_FIXTURE__=${json};window.__DASH_THEME__=${JSON.stringify(theme ?? 'light')};</script>`;
1312
+ const html = tpl.replace(/<head([^>]*)>/i, `<head$1>${inject}`);
1313
+ const dir = path.join(dataDir(), 'reports');
1314
+ mkdirSync(dir, { recursive: true });
1315
+ const file = path.join(dir, `${sanitizeProperty(siteUrl)}-dashboard.html`);
1316
+ writeFileSync(file, html);
1317
+ return {
1318
+ content: [{ type: 'text', text: `Report saved: ${file}\nOpen it in any browser for the full interactive dashboard (${data.findings?.total ?? 0} findings). Shareable — send it to a client as-is.` }],
1319
+ structuredContent: { path: file, siteUrl, findings: data.findings?.total ?? 0, bytes: html.length },
1320
+ };
1321
+ });
1322
+ // pull_backlinks — on-demand backlink profile (DataForSEO, paid + 20-day cached). Powers
1323
+ // backlinks-to-404 (the big quick win), top-linked pages, and true-orphan detection.
1324
+ server.registerTool('pull_backlinks', {
1325
+ title: 'Pull backlink profile (DataForSEO)',
1326
+ description: 'Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
1327
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(1000).optional(), statusLimit: z.number().int().min(0).max(1000).optional() },
1328
+ }, async ({ siteUrl, limit, statusLimit }) => {
1329
+ const bl = requireDfs(backlinks);
1330
+ const jobId = jobs.start('backlinks', (update, signal) => bl.run(siteUrl, { limit, statusLimit }, update, signal));
1331
+ return {
1332
+ content: [{ type: 'text', text: `Backlink pull started for ${siteUrl} (job ${jobId}). Poll check_sync_status, then run_audit for backlinks-to-404.` }],
1333
+ structuredContent: { jobId, status: 'running', siteUrl },
1334
+ };
1335
+ });
1336
+ server.registerTool('resolve_entities', {
1337
+ title: 'Resolve page entities (Wikidata)',
1338
+ description: 'Heuristically resolve each top page’s primary entity (by H1, fallback title) to a Wikidata QID, and store subclass-of / part-of relationships between them. Unlocks the entity-internal-link-gap finding (suggests internal links a topical mesh implies). Heuristic — those findings are judgement (N), shown only with includeJudgement. Free (public Wikidata API), cached ~45 days, on-demand. Async job — poll check_sync_status.',
1339
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(500).optional(), language: z.string().optional() },
1340
+ }, async ({ siteUrl, limit, language }) => {
1341
+ const jobId = jobs.start('entities', (update, signal) => entities.run(siteUrl, { ...(limit != null ? { limit } : {}), ...(language ? { language } : {}) }, update, signal));
1342
+ return {
1343
+ content: [{ type: 'text', text: `Entity resolution started for ${siteUrl} (job ${jobId}). Poll check_sync_status, then run_audit includeJudgement:true for entity-internal-link-gap.` }],
1344
+ structuredContent: { jobId, status: 'running', siteUrl },
1345
+ };
1346
+ });
1347
+ // ── Static reference resources (same content the tools return — one source of truth) ──
1348
+ server.registerResource('checks-reference', 'seo-audit://checks-reference', {
1349
+ title: 'Check registry reference',
1350
+ description: 'The full audit check catalogue (the list_checks data) rendered as markdown: every check with category, severity, labels, certainty, fix type and its one-line fix.',
1351
+ mimeType: 'text/markdown',
1352
+ }, async () => ({
1353
+ contents: [{ uri: 'seo-audit://checks-reference', mimeType: 'text/markdown', text: buildChecksMarkdown(listChecks()) }],
1354
+ }));
1355
+ server.registerResource('cookbook', 'seo-audit://cookbook', {
1356
+ title: 'Composition cookbook',
1357
+ description: 'The data-surface map (grain, join keys, freshness, cost per source) and worked multi-source recipes — identical to the composition_cookbook tool output.',
1358
+ mimeType: 'text/markdown',
1359
+ }, async () => ({
1360
+ contents: [{ uri: 'seo-audit://cookbook', mimeType: 'text/markdown', text: COOKBOOK_TEXT }],
1361
+ }));
1362
+ registerAppResource(server, 'SEO Dashboard', DASHBOARD_URI, {}, async () => ({
1363
+ contents: [{ uri: DASHBOARD_URI, mimeType: RESOURCE_MIME_TYPE, text: await readFile(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf-8') }],
1364
+ }));
1365
+ registerAppResource(server, 'Sync Progress', SYNC_PROGRESS_URI, {}, async () => ({
1366
+ contents: [{ uri: SYNC_PROGRESS_URI, mimeType: RESOURCE_MIME_TYPE, text: await readFile(path.join(__dirname, 'src', 'ui', 'sync-progress.html'), 'utf-8') }],
1367
+ }));
1368
+ server.registerTool('track_ranks', {
1369
+ title: 'Track ranks over time (DataForSEO)',
1370
+ description: 'Ingest the DataForSEO over-time sequence (monthly rank distribution + ETV) into rank_history, so rank charts have a real time axis reconciled with GSC. DataForSEO Labs call — cached 20 days. Pass location once (e.g. "Australia", "United Kingdom") — it is saved per property. Async job.',
1371
+ inputSchema: { siteUrl: z.string(), location: z.union([z.string(), z.number()]).optional() },
1372
+ }, async ({ siteUrl, location }) => {
1373
+ const tracker = requireDfs(rankTracker);
1374
+ const jobId = jobs.start('rank_history', (update, signal) => tracker.run(siteUrl, { location }, update, signal));
1375
+ return {
1376
+ content: [{ type: 'text', text: `Rank-history ingest started for ${siteUrl} (job ${jobId}). Poll check_sync_status.` }],
1377
+ structuredContent: { jobId, status: 'running', siteUrl },
1378
+ };
1379
+ });
1380
+ const run = async () => {
1381
+ const transport = new StdioServerTransport();
1382
+ await server.connect(transport);
1383
+ console.error(`${SERVER_NAME} v${SERVER_VERSION} running on stdio (data: ${dataDir()}${credPath ? '' : ', GSC disabled — no credentials'})`);
1384
+ };
1385
+ return { server, run };
1386
+ }
1387
+ //# sourceMappingURL=server.js.map