@houtini/seo-audit-console 0.9.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -13
- package/dist/audit/checks.d.ts +0 -7
- package/dist/audit/checks.d.ts.map +1 -1
- package/dist/audit/checks.js +22 -282
- package/dist/audit/checks.js.map +1 -1
- package/dist/audit/drift.d.ts +0 -12
- package/dist/audit/drift.d.ts.map +1 -1
- package/dist/audit/drift.js +0 -15
- package/dist/audit/drift.js.map +1 -1
- package/dist/audit/engine.d.ts +0 -3
- package/dist/audit/engine.d.ts.map +1 -1
- package/dist/audit/engine.js +7 -42
- package/dist/audit/engine.js.map +1 -1
- package/dist/audit/keywordList.d.ts +0 -14
- package/dist/audit/keywordList.d.ts.map +1 -1
- package/dist/audit/keywordList.js +0 -5
- package/dist/audit/keywordList.js.map +1 -1
- package/dist/audit/opportunities.js +3 -26
- package/dist/audit/opportunities.js.map +1 -1
- package/dist/audit/recon.d.ts +0 -31
- package/dist/audit/recon.d.ts.map +1 -1
- package/dist/audit/recon.js +0 -27
- package/dist/audit/recon.js.map +1 -1
- package/dist/audit/report.d.ts +0 -4
- package/dist/audit/report.d.ts.map +1 -1
- package/dist/audit/report.js +0 -7
- package/dist/audit/report.js.map +1 -1
- package/dist/audit/schema-validate.d.ts +0 -26
- package/dist/audit/schema-validate.d.ts.map +1 -1
- package/dist/audit/schema-validate.js +8 -64
- package/dist/audit/schema-validate.js.map +1 -1
- package/dist/audit/templates.d.ts +0 -24
- package/dist/audit/templates.d.ts.map +1 -1
- package/dist/audit/templates.js +1 -21
- package/dist/audit/templates.js.map +1 -1
- package/dist/audit/topicGaps.d.ts +0 -1
- package/dist/audit/topicGaps.d.ts.map +1 -1
- package/dist/audit/topicGaps.js +3 -24
- package/dist/audit/topicGaps.js.map +1 -1
- package/dist/core/AuditDatabase.d.ts +0 -14
- package/dist/core/AuditDatabase.d.ts.map +1 -1
- package/dist/core/AuditDatabase.js +8 -61
- package/dist/core/AuditDatabase.js.map +1 -1
- package/dist/core/Backlinks.d.ts +0 -7
- package/dist/core/Backlinks.d.ts.map +1 -1
- package/dist/core/Backlinks.js +1 -23
- package/dist/core/Backlinks.js.map +1 -1
- package/dist/core/Crawler.d.ts +0 -1
- package/dist/core/Crawler.d.ts.map +1 -1
- package/dist/core/Crawler.js +24 -119
- package/dist/core/Crawler.js.map +1 -1
- package/dist/core/DataForSeoClient.d.ts +0 -74
- package/dist/core/DataForSeoClient.d.ts.map +1 -1
- package/dist/core/DataForSeoClient.js +1 -72
- package/dist/core/DataForSeoClient.js.map +1 -1
- package/dist/core/Entities.d.ts +0 -6
- package/dist/core/Entities.d.ts.map +1 -1
- package/dist/core/Entities.js +0 -8
- package/dist/core/Entities.js.map +1 -1
- package/dist/core/FirecrawlClient.d.ts +0 -18
- package/dist/core/FirecrawlClient.d.ts.map +1 -1
- package/dist/core/FirecrawlClient.js +1 -13
- package/dist/core/FirecrawlClient.js.map +1 -1
- package/dist/core/GscClient.d.ts +0 -7
- package/dist/core/GscClient.d.ts.map +1 -1
- package/dist/core/GscClient.js +1 -10
- package/dist/core/GscClient.js.map +1 -1
- package/dist/core/GscSync.d.ts +0 -4
- package/dist/core/GscSync.d.ts.map +1 -1
- package/dist/core/GscSync.js +2 -33
- package/dist/core/GscSync.js.map +1 -1
- package/dist/core/JobManager.d.ts +0 -5
- package/dist/core/JobManager.d.ts.map +1 -1
- package/dist/core/JobManager.js +0 -8
- package/dist/core/JobManager.js.map +1 -1
- package/dist/core/LinkIntersect.d.ts +0 -1
- package/dist/core/LinkIntersect.d.ts.map +1 -1
- package/dist/core/LinkIntersect.js +1 -37
- package/dist/core/LinkIntersect.js.map +1 -1
- package/dist/core/MajesticClient.d.ts +0 -26
- package/dist/core/MajesticClient.d.ts.map +1 -1
- package/dist/core/MajesticClient.js +0 -15
- package/dist/core/MajesticClient.js.map +1 -1
- package/dist/core/RankTracker.d.ts +0 -4
- package/dist/core/RankTracker.d.ts.map +1 -1
- package/dist/core/RankTracker.js +0 -9
- package/dist/core/RankTracker.js.map +1 -1
- package/dist/core/Refresh.d.ts +0 -6
- package/dist/core/Refresh.d.ts.map +1 -1
- package/dist/core/Refresh.js +0 -6
- package/dist/core/Refresh.js.map +1 -1
- package/dist/core/SupadataClient.d.ts +0 -11
- package/dist/core/SupadataClient.d.ts.map +1 -1
- package/dist/core/SupadataClient.js +0 -5
- package/dist/core/SupadataClient.js.map +1 -1
- package/dist/core/UrlInspector.d.ts +0 -5
- package/dist/core/UrlInspector.d.ts.map +1 -1
- package/dist/core/UrlInspector.js +0 -6
- package/dist/core/UrlInspector.js.map +1 -1
- package/dist/core/WikidataClient.d.ts +0 -2
- package/dist/core/WikidataClient.d.ts.map +1 -1
- package/dist/core/WikidataClient.js +0 -7
- package/dist/core/WikidataClient.js.map +1 -1
- package/dist/core/agentReadiness.d.ts +0 -9
- package/dist/core/agentReadiness.d.ts.map +1 -1
- package/dist/core/agentReadiness.js +0 -13
- package/dist/core/agentReadiness.js.map +1 -1
- package/dist/core/ctrModel.js +0 -3
- package/dist/core/ctrModel.js.map +1 -1
- package/dist/core/dashboardData.d.ts +0 -1
- package/dist/core/dashboardData.d.ts.map +1 -1
- package/dist/core/dashboardData.js +13 -95
- package/dist/core/dashboardData.js.map +1 -1
- package/dist/core/dataStorage.d.ts +0 -2
- package/dist/core/dataStorage.d.ts.map +1 -1
- package/dist/core/dataStorage.js +6 -19
- package/dist/core/dataStorage.js.map +1 -1
- package/dist/core/draftBrief.d.ts +0 -7
- package/dist/core/draftBrief.d.ts.map +1 -1
- package/dist/core/draftBrief.js +0 -10
- package/dist/core/draftBrief.js.map +1 -1
- package/dist/core/extract.d.ts +0 -11
- package/dist/core/extract.d.ts.map +1 -1
- package/dist/core/extract.js +1 -38
- package/dist/core/extract.js.map +1 -1
- package/dist/core/googleNews.d.ts +0 -11
- package/dist/core/googleNews.d.ts.map +1 -1
- package/dist/core/googleNews.js +0 -7
- package/dist/core/googleNews.js.map +1 -1
- package/dist/core/gscFreshness.d.ts +0 -10
- package/dist/core/gscFreshness.d.ts.map +1 -1
- package/dist/core/gscFreshness.js +0 -11
- package/dist/core/gscFreshness.js.map +1 -1
- package/dist/core/linkGraph.d.ts +0 -15
- package/dist/core/linkGraph.d.ts.map +1 -1
- package/dist/core/linkGraph.js +1 -23
- package/dist/core/linkGraph.js.map +1 -1
- package/dist/core/marketSizing.d.ts +0 -13
- package/dist/core/marketSizing.d.ts.map +1 -1
- package/dist/core/marketSizing.js +1 -1
- package/dist/core/marketSizing.js.map +1 -1
- package/dist/core/passageScore.d.ts +0 -13
- package/dist/core/passageScore.d.ts.map +1 -1
- package/dist/core/passageScore.js +1 -16
- package/dist/core/passageScore.js.map +1 -1
- package/dist/core/paths.d.ts +0 -3
- package/dist/core/paths.d.ts.map +1 -1
- package/dist/core/paths.js +0 -0
- package/dist/core/paths.js.map +1 -1
- package/dist/core/queryData.d.ts +0 -1
- package/dist/core/queryData.d.ts.map +1 -1
- package/dist/core/queryData.js +0 -18
- package/dist/core/queryData.js.map +1 -1
- package/dist/core/reconFetch.d.ts +0 -6
- package/dist/core/reconFetch.d.ts.map +1 -1
- package/dist/core/reconFetch.js +1 -4
- package/dist/core/reconFetch.js.map +1 -1
- package/dist/core/reconResearch.d.ts +0 -20
- package/dist/core/reconResearch.d.ts.map +1 -1
- package/dist/core/reconResearch.js +1 -16
- package/dist/core/reconResearch.js.map +1 -1
- package/dist/core/reranker.d.ts +0 -2
- package/dist/core/reranker.d.ts.map +1 -1
- package/dist/core/reranker.js +1 -11
- package/dist/core/reranker.js.map +1 -1
- package/dist/core/robots.d.ts +0 -6
- package/dist/core/robots.d.ts.map +1 -1
- package/dist/core/robots.js +1 -7
- package/dist/core/robots.js.map +1 -1
- package/dist/core/serpFootprint.d.ts +0 -12
- package/dist/core/serpFootprint.d.ts.map +1 -1
- package/dist/core/serpFootprint.js +0 -1
- package/dist/core/serpFootprint.js.map +1 -1
- package/dist/core/serpRecon.d.ts +0 -14
- package/dist/core/serpRecon.d.ts.map +1 -1
- package/dist/core/serpRecon.js +1 -13
- package/dist/core/serpRecon.js.map +1 -1
- package/dist/core/sitemap.js +3 -16
- package/dist/core/sitemap.js.map +1 -1
- package/dist/core/sql.d.ts +0 -4
- package/dist/core/sql.d.ts.map +1 -1
- package/dist/core/sql.js +0 -4
- package/dist/core/sql.js.map +1 -1
- package/dist/core/url-key.d.ts +0 -31
- package/dist/core/url-key.d.ts.map +1 -1
- package/dist/core/url-key.js +1 -38
- package/dist/core/url-key.js.map +1 -1
- package/dist/core/webHandlers.d.ts +0 -6
- package/dist/core/webHandlers.d.ts.map +1 -1
- package/dist/core/webHandlers.js +2 -5
- package/dist/core/webHandlers.js.map +1 -1
- package/dist/core/webServer.d.ts +0 -16
- package/dist/core/webServer.d.ts.map +1 -1
- package/dist/core/webServer.js +3 -17
- package/dist/core/webServer.js.map +1 -1
- package/dist/dashboard.js +0 -23
- package/dist/dashboard.js.map +1 -1
- package/dist/generators/index.d.ts +0 -7
- package/dist/generators/index.d.ts.map +1 -1
- package/dist/generators/index.js +1 -26
- package/dist/generators/index.js.map +1 -1
- package/dist/index.js +0 -2
- package/dist/index.js.map +1 -1
- package/dist/server.js +16 -166
- package/dist/server.js.map +1 -1
- package/package.json +1 -1
- package/server.json +2 -2
package/dist/server.js
CHANGED
|
@@ -18,15 +18,10 @@ import { fetchCompetitorContent, routeFor } from './core/reconResearch.js';
|
|
|
18
18
|
import { fetchOwnPage } from './core/reconFetch.js';
|
|
19
19
|
import { parseSerpForRecon, reconVerdict } from './core/serpRecon.js';
|
|
20
20
|
import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos, pageState, crawlRealityOverride, clusterCannibalisation, opportunityBasis } from './audit/recon.js';
|
|
21
|
-
/** A clickable browser-dashboard link appended to tool outputs — the user should always
|
|
22
|
-
* know the full interactive report is one click away (or one serve_dashboard call away). */
|
|
23
21
|
function browserLink(siteUrl) {
|
|
24
22
|
const base = dashboardServerUrl();
|
|
25
23
|
if (base)
|
|
26
24
|
return `\n\n📊 Browser dashboard: ${base}/dashboard${siteUrl ? `?siteUrl=${encodeURIComponent(siteUrl)}` : ''}`;
|
|
27
|
-
// No live server: point at the dashboard surfaces that always work. get_dashboard renders the
|
|
28
|
-
// interactive dashboard in chat (works through the Docker gateway where a served port would not);
|
|
29
|
-
// serve_dashboard opens a browser tab locally; export_report writes a shareable HTML file.
|
|
30
25
|
return `\n\n📊 See it in the dashboard — run get_dashboard${siteUrl ? ` for ${siteUrl}` : ''} (interactive, in chat), serve_dashboard (browser tab), or export_report (shareable HTML).`;
|
|
31
26
|
}
|
|
32
27
|
import { runAudit, runSingleCheck, listChecks } from './audit/engine.js';
|
|
@@ -62,9 +57,6 @@ import { gscFreshness } from './core/gscFreshness.js';
|
|
|
62
57
|
import { clusterKeywordList } from './audit/keywordList.js';
|
|
63
58
|
const SERVER_NAME = 'seo-audit-console';
|
|
64
59
|
const SERVER_VERSION = JSON.parse(readFileSync(new URL('../package.json', import.meta.url), 'utf8')).version;
|
|
65
|
-
// Where per-property crawl/audit DBs live. Resolution order (computed once per run):
|
|
66
|
-
// SAC_DATA_DIR env > persisted choice (~/.seo-audit-console.json) > Documents default.
|
|
67
|
-
// The user can set the persisted choice through the `data_location` tool (no JSON editing).
|
|
68
60
|
const CONFIG_PATH = path.join(homedir(), '.seo-audit-console.json');
|
|
69
61
|
const DEFAULT_DATA_DIR = path.join(homedir(), 'Documents', 'seo-audit-console');
|
|
70
62
|
function readConfigDataDir() {
|
|
@@ -82,21 +74,14 @@ export function dataDir() {
|
|
|
82
74
|
return RESOLVED_DATA_DIR;
|
|
83
75
|
}
|
|
84
76
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
85
|
-
// Version the widget URIs: hosts may cache ui:// resources by URI indefinitely, so an
|
|
86
|
-
// unversioned URI can pin users to a stale (or broken) cached bundle across releases.
|
|
87
77
|
const DASHBOARD_URI = `ui://dashboard/main-${SERVER_VERSION}.html`;
|
|
88
78
|
const SYNC_PROGRESS_URI = `ui://sync-progress/main-${SERVER_VERSION}.html`;
|
|
89
|
-
// Must match the categories actually used by CHECKS (src/audit/checks.ts) so category
|
|
90
|
-
// filters never silently return empty. (Was listing performance/agentic/integrity/war-stories
|
|
91
|
-
// which no check uses, and omitting content/security which checks do use.)
|
|
92
79
|
const CHECK_CATEGORIES = [
|
|
93
80
|
'crawlability', 'indexation', 'onpage', 'content', 'schema', 'security', 'performance', 'merged',
|
|
94
81
|
];
|
|
95
82
|
function isoDaysAgo(days) {
|
|
96
83
|
return new Date(Date.now() - days * 86400000).toISOString().slice(0, 10);
|
|
97
84
|
}
|
|
98
|
-
// Server-level instructions (MCP `initialize` result): teach the assistant how the data
|
|
99
|
-
// sources JOIN so it can COMPOSE bespoke multi-source analyses, not just run presets.
|
|
100
85
|
const SERVER_INSTRUCTIONS = `SEO Audit Console fuses four data sources into one SQLite database per property: Google Search Console history, a first-party site crawl, GSC URL Inspection, and on-demand DataForSEO (SERP/Labs/Backlinks). Its real power is COMPOSITION — joining sources to answer questions no single tool answers.
|
|
101
86
|
|
|
102
87
|
JOIN KEYS (memorise these):
|
|
@@ -185,8 +170,6 @@ Raw access: query_audit runs any single check with full evidence; every table ab
|
|
|
185
170
|
- **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
|
|
186
171
|
|
|
187
172
|
Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
|
|
188
|
-
// The check catalogue rendered as markdown — single source of truth is listChecks();
|
|
189
|
-
// shared by the seo-audit://checks-reference resource (and buildable for any category subset).
|
|
190
173
|
function buildChecksMarkdown(checks) {
|
|
191
174
|
const byCategory = new Map();
|
|
192
175
|
for (const c of checks) {
|
|
@@ -255,7 +238,7 @@ export function createServer() {
|
|
|
255
238
|
const gsc = credPath ? new GscClient(credPath) : null;
|
|
256
239
|
const sync = gsc ? new GscSync(gsc, dataDir()) : null;
|
|
257
240
|
const inspector = gsc ? new UrlInspector(gsc, dataDir()) : null;
|
|
258
|
-
const crawler = new Crawler(dataDir());
|
|
241
|
+
const crawler = new Crawler(dataDir());
|
|
259
242
|
const dfsUser = process.env.DATAFORSEO_USERNAME;
|
|
260
243
|
const dfsPass = process.env.DATAFORSEO_PASSWORD;
|
|
261
244
|
const dfsCacheDays = Number(process.env.DATAFORSEO_CACHE_DAYS) || 7;
|
|
@@ -264,16 +247,12 @@ export function createServer() {
|
|
|
264
247
|
: null;
|
|
265
248
|
const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
|
|
266
249
|
const linkIntersect = dfs ? new LinkIntersect(dfs, dataDir()) : null;
|
|
267
|
-
// Majestic (Trust Flow / Topical Trust Flow) — optional link_intersect + trapped-authority tier.
|
|
268
250
|
const majesticKey = process.env.MAJESTIC_API_KEY;
|
|
269
251
|
const majesticCacheDays = Number(process.env.MAJESTIC_CACHE_DAYS) || 30;
|
|
270
252
|
const majestic = majesticKey ? new MajesticClient(majesticKey, path.join(dataDir(), 'majestic-cache.db'), majesticCacheDays) : null;
|
|
271
|
-
// Backlinks after Majestic, so pull_backlinks can enrich per-URL pages with Trust Flow.
|
|
272
253
|
const backlinks = dfs ? new Backlinks(dfs, dataDir(), majestic) : null;
|
|
273
|
-
// Firecrawl (competitor-page scraping for content recon) — optional; degrades gracefully.
|
|
274
254
|
const firecrawlKey = process.env.FIRECRAWL_API_KEY;
|
|
275
255
|
const firecrawl = firecrawlKey ? new FirecrawlClient(firecrawlKey, path.join(dataDir(), 'firecrawl-cache.db')) : null;
|
|
276
|
-
// Supadata (transcribes the ranking videos for content recon) — optional; degrades gracefully.
|
|
277
256
|
const supadataKey = process.env.SUPADATA_API_KEY;
|
|
278
257
|
const supadata = supadataKey ? new SupadataClient(supadataKey, path.join(dataDir(), 'supadata-cache.db')) : null;
|
|
279
258
|
const entities = new Entities(new WikidataClient(path.join(dataDir(), 'wikidata-cache.db')), dataDir());
|
|
@@ -288,33 +267,21 @@ export function createServer() {
|
|
|
288
267
|
throw new Error('DATAFORSEO_USERNAME / DATAFORSEO_PASSWORD not set — required for DataForSEO.');
|
|
289
268
|
return v;
|
|
290
269
|
};
|
|
291
|
-
// Which optional integrations have a key set — drives the dashboard's key-gated tabs
|
|
292
|
-
// (Links, Content research) and their affiliate-linked upsell states. The data layer has
|
|
293
|
-
// no env access, so it's injected here onto every dashboard payload surface.
|
|
294
270
|
const apiKeysStatus = () => ({ dataforseo: !!dfs, majestic: !!majestic, firecrawl: !!firecrawl, supadata: !!supadata });
|
|
295
|
-
// The dashboard webserver's options — one definition shared by serve_dashboard AND the
|
|
296
|
-
// auto-start at the end of refresh_property / run_audit, so a populated property always has
|
|
297
|
-
// a live browser link (browserLink() surfaces dashboardServerUrl() once this is running).
|
|
298
271
|
const buildWebOpts = (port) => ({
|
|
299
272
|
dataDir,
|
|
300
273
|
uiHtml: () => readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8'),
|
|
301
274
|
call: buildWebCallHandlers({ dataDir, dfs, apiKeys: apiKeysStatus }),
|
|
302
275
|
...(port != null ? { port } : {}),
|
|
303
276
|
});
|
|
304
|
-
// Idempotent auto-start used by refresh_property / run_audit. Best-effort: a served port that
|
|
305
|
-
// can't bind (e.g. Docker without a published port) must never fail the populate/audit — the
|
|
306
|
-
// in-chat get_dashboard and export_report still work, and browserLink() falls back to those.
|
|
307
|
-
// Opt-out with SAC_AUTOSERVE=0 for headless/CI/sandboxed runs where binding a localhost port
|
|
308
|
-
// is unwanted (default: on, since the whole point is to hand the user a link).
|
|
309
277
|
const autoServeDashboard = async () => {
|
|
310
278
|
if (/^(0|false|no|off)$/i.test(process.env.SAC_AUTOSERVE ?? ''))
|
|
311
279
|
return;
|
|
312
280
|
try {
|
|
313
281
|
await startDashboardServer(buildWebOpts());
|
|
314
282
|
}
|
|
315
|
-
catch {
|
|
283
|
+
catch { }
|
|
316
284
|
};
|
|
317
|
-
// ── Introspection (no data / creds required) ────────────────────────────
|
|
318
285
|
server.registerTool('seo_audit_help', {
|
|
319
286
|
title: 'Help — what this audit can do',
|
|
320
287
|
description: 'Overview of every tool/feature with an example prompt for each. Start here.',
|
|
@@ -358,8 +325,6 @@ export function createServer() {
|
|
|
358
325
|
inputSchema: { path: z.string().optional() },
|
|
359
326
|
}, async ({ path: newPath }) => {
|
|
360
327
|
if (newPath) {
|
|
361
|
-
// Persist an ABSOLUTE path — a relative one resolves against the host process's
|
|
362
|
-
// cwd, which differs across Claude Desktop launches, so DBs would "disappear".
|
|
363
328
|
newPath = path.resolve(newPath);
|
|
364
329
|
mkdirSync(newPath, { recursive: true });
|
|
365
330
|
writeFileSync(CONFIG_PATH, JSON.stringify({ dataDir: newPath }, null, 2));
|
|
@@ -405,10 +370,6 @@ export function createServer() {
|
|
|
405
370
|
}
|
|
406
371
|
const rows = s.properties.map(p => `| ${p.siteUrl ?? p.file} | ${fmtBytes(p.bytes)} | ${p.searchAnalytics.toLocaleString('en-US')} | ${p.pages.toLocaleString('en-US')} | ${p.links.toLocaleString('en-US')} | ${p.pageSnapshots.toLocaleString('en-US')} | ${p.findings.toLocaleString('en-US')} | ${p.lastSynced?.slice(0, 10) ?? '—'} | ${p.lastCrawl?.slice(0, 10) ?? '—'} |`).join('\n');
|
|
407
372
|
const cacheLines = s.caches.map(c => `- ${c.file}: ${fmtBytes(c.bytes)}`).join('\n');
|
|
408
|
-
// Link-intersect freshness: prospect data ages as competitors keep earning links, so
|
|
409
|
-
// flag properties that HAVE link_prospects and roughly how stale it is — a re-run of
|
|
410
|
-
// link_intersect updates it. (The DataForSEO call is 20-day cached; older than that a
|
|
411
|
-
// re-run genuinely refetches.)
|
|
412
373
|
const nowMs = Date.now();
|
|
413
374
|
const liProps = s.properties.filter(p => (p.linkProspects ?? 0) > 0);
|
|
414
375
|
const liLines = liProps.map(p => {
|
|
@@ -425,7 +386,6 @@ export function createServer() {
|
|
|
425
386
|
`Prune with data_storage prune:{siteUrl, action:"vacuum" | "clear-crawl-history" | "delete-property"} (destructive actions need confirm:true).`;
|
|
426
387
|
return { content: [{ type: 'text', text: md }], structuredContent: s };
|
|
427
388
|
});
|
|
428
|
-
// ── Audit engine ────────────────────────────────────────────────────────
|
|
429
389
|
server.registerTool('run_audit', {
|
|
430
390
|
title: 'Run SEO audit',
|
|
431
391
|
description: 'Run the technical-SEO checks against synced data and return scored findings, ranked by expected clicks per dev-hour — Priority = (T × Y × C) / E (T = clicks at stake from real GSC data, Y = expected yield, C = certainty, E = effort hours). Crawl + GSC + URL-inspection checks. Set includeJudgement=true to include heuristic (N) checks.',
|
|
@@ -437,7 +397,7 @@ export function createServer() {
|
|
|
437
397
|
},
|
|
438
398
|
}, async ({ siteUrl, scope, categories, includeJudgement }) => {
|
|
439
399
|
const result = runAudit(dataDir(), siteUrl, { scope, categories, includeJudgement });
|
|
440
|
-
await autoServeDashboard();
|
|
400
|
+
await autoServeDashboard();
|
|
441
401
|
return {
|
|
442
402
|
content: [{ type: 'text', text: buildAuditMarkdown(result, siteUrl) + browserLink(siteUrl) }],
|
|
443
403
|
structuredContent: { ...result, dashboardUrl: dashboardServerUrl() },
|
|
@@ -455,8 +415,6 @@ export function createServer() {
|
|
|
455
415
|
},
|
|
456
416
|
}, async ({ siteUrl, check, limit, offset, columns }) => {
|
|
457
417
|
const r = runSingleCheck(dataDir(), siteUrl, check, limit, offset ?? 0);
|
|
458
|
-
// Token discipline (mirrors query_data): optional evidence-column selection + loud
|
|
459
|
-
// 120-char cell truncation, and an explicit offset/total footer.
|
|
460
418
|
const shape = (f) => {
|
|
461
419
|
const src = (f.evidence ?? {});
|
|
462
420
|
const keys = columns?.length ? columns.filter(k => k in src) : Object.keys(src);
|
|
@@ -467,9 +425,6 @@ export function createServer() {
|
|
|
467
425
|
};
|
|
468
426
|
let sc = { ...r, findingsTotal: r.total };
|
|
469
427
|
let findings = r.findings.map(shape);
|
|
470
|
-
// Hosts cap the model-facing result (~60k chars). On evidence-heavy checks a large
|
|
471
|
-
// `limit` can blow that ceiling and error the whole call — trim findings (keeping the
|
|
472
|
-
// true total) instead of failing. Same guard pattern as detect_changes.
|
|
473
428
|
while (findings.length > 25 && JSON.stringify({ ...sc, findings }).length > 45000) {
|
|
474
429
|
findings = findings.slice(0, Math.floor(findings.length / 2));
|
|
475
430
|
}
|
|
@@ -540,7 +495,6 @@ export function createServer() {
|
|
|
540
495
|
const text = `WRITING BRIEF — ${r.url}\nTarget query: “${r.topQuery ?? '(unknown)'}”\nGap: ${r.gap}\n\nTASK\n${r.brief}\n\nVOICE (match this — the page's own writing):\n${voice || ' (no substantial passages)'}\n\nFACTS (ground in these — the page's most query-relevant content; invent nothing beyond them):\n${facts || ' (none)'}`;
|
|
541
496
|
return { content: [{ type: 'text', text }], structuredContent: r };
|
|
542
497
|
});
|
|
543
|
-
// detect_changes — change-detection / drift: what changed on the site since the last crawl.
|
|
544
498
|
server.registerTool('detect_changes', {
|
|
545
499
|
title: 'Detect changes since the last crawl',
|
|
546
500
|
description: 'Compare the two most recent crawls and report what changed per URL — status code, indexability, canonical, robots/noindex, title, meta, H1, schema, large content swings — each classified by severity (critical → info). This is the monitor: run refresh_property on different days to build history, then this surfaces regressions (a page that went noindex, a canonical that flipped, a 200 that became a 404). Needs at least two crawls.',
|
|
@@ -549,9 +503,6 @@ export function createServer() {
|
|
|
549
503
|
const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
550
504
|
try {
|
|
551
505
|
const d = diffLatest(adb.db);
|
|
552
|
-
// Cap the changes array in structuredContent to `limit` — the full diff can be tens of
|
|
553
|
-
// thousands of rows (e.g. a crawl-methodology change) and blow the MCP token ceiling. The
|
|
554
|
-
// summary keeps the TRUE totals; markdown is already limited.
|
|
555
506
|
const lim = limit ?? 50;
|
|
556
507
|
const sc = { ...d, changes: d.changes.slice(0, lim), changesReturned: Math.min(lim, d.changes.length), changesTotal: d.changes.length };
|
|
557
508
|
return {
|
|
@@ -563,14 +514,12 @@ export function createServer() {
|
|
|
563
514
|
adb.close();
|
|
564
515
|
}
|
|
565
516
|
});
|
|
566
|
-
// check_agent_readiness — is the site ready for AI agents? (the agentic-SEO / GEO frontier)
|
|
567
517
|
server.registerTool('check_agent_readiness', {
|
|
568
518
|
title: 'Check agent readiness (AI-agent / GEO signals)',
|
|
569
519
|
description: 'Probe a site for AI-agent readiness — the signals agents use to discover and use it: robots.txt AI-bot rules + Content Signals, sitemap, Link headers, llms.txt, agents.md, Markdown content negotiation, Web Bot Auth, MCP server card, Agent Skills, API Catalog, OAuth discovery. Returns a 0–100 score, a level (Basic web presence → Agent-native), and a per-check checklist with copy-paste fixes. Live HTTP probes of the property origin (no crawl/GSC needed). Modelled on Cloudflare\'s isitagentready.com.',
|
|
570
520
|
inputSchema: { siteUrl: z.string() },
|
|
571
521
|
}, async ({ siteUrl }) => {
|
|
572
522
|
const r = await checkAgentReadiness(siteUrl);
|
|
573
|
-
// Persist so the dashboard/export can show it (live probe, not crawl-derived).
|
|
574
523
|
try {
|
|
575
524
|
const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
576
525
|
try {
|
|
@@ -582,13 +531,12 @@ export function createServer() {
|
|
|
582
531
|
adb.close();
|
|
583
532
|
}
|
|
584
533
|
}
|
|
585
|
-
catch {
|
|
534
|
+
catch { }
|
|
586
535
|
return {
|
|
587
536
|
content: [{ type: 'text', text: buildAgentReadinessMarkdown(r) }],
|
|
588
537
|
structuredContent: r,
|
|
589
538
|
};
|
|
590
539
|
});
|
|
591
|
-
// fix_finding — the moat: turn a finding into a concrete, paste-ready remediation.
|
|
592
540
|
server.registerTool('fix_finding', {
|
|
593
541
|
title: 'Generate a fix for a finding',
|
|
594
542
|
description: 'Turn an audit finding into a concrete, paste-ready remediation: JSON-LD for missing structured data, a 301 rule for broken / redirecting internal links, or internal-link suggestions for orphan / striking-distance pages. Other checks return their deterministic fix guidance. Dry-run — returns artifacts, never writes to your site. Identify the finding by findingId (from run_audit) or by check + url.',
|
|
@@ -602,7 +550,6 @@ export function createServer() {
|
|
|
602
550
|
}, async ({ siteUrl, findingId, check, url, redirectFormat }) => {
|
|
603
551
|
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
604
552
|
try {
|
|
605
|
-
// Resolve the finding → checkId, affected url_key, evidence.
|
|
606
553
|
let checkId = check;
|
|
607
554
|
let affectedKey = null;
|
|
608
555
|
let evidence = {};
|
|
@@ -641,14 +588,12 @@ export function createServer() {
|
|
|
641
588
|
break;
|
|
642
589
|
}
|
|
643
590
|
case 'redirect-chain': {
|
|
644
|
-
// The finding's url_key IS the chain's final page — its row holds the recorded
|
|
645
|
-
// hops. Collapse each hop straight to the final URL; never fuzzy-match here.
|
|
646
591
|
const page = db.db.prepare('SELECT url, redirects FROM pages WHERE url_key = ?').get(affectedKey);
|
|
647
592
|
let hops = [];
|
|
648
593
|
try {
|
|
649
594
|
hops = page?.redirects ? JSON.parse(page.redirects) : [];
|
|
650
595
|
}
|
|
651
|
-
catch {
|
|
596
|
+
catch { }
|
|
652
597
|
kind = 'redirect';
|
|
653
598
|
fix = page && hops.length
|
|
654
599
|
? { ...collapseChainRules(page.url, hops, redirectFormat ?? 'htaccess'), chain: hops }
|
|
@@ -656,8 +601,6 @@ export function createServer() {
|
|
|
656
601
|
break;
|
|
657
602
|
}
|
|
658
603
|
case 'internal-links-to-redirects': {
|
|
659
|
-
// The flagged key is a redirect SOURCE; the crawler recorded where it actually
|
|
660
|
-
// goes — find the chain containing it and use that real destination.
|
|
661
604
|
const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
|
|
662
605
|
let knownTarget = null;
|
|
663
606
|
for (const row of db.db.prepare(`SELECT url, redirects FROM pages WHERE redirects IS NOT NULL AND redirects <> ''`).iterate()) {
|
|
@@ -668,15 +611,13 @@ export function createServer() {
|
|
|
668
611
|
break;
|
|
669
612
|
}
|
|
670
613
|
}
|
|
671
|
-
catch {
|
|
614
|
+
catch { }
|
|
672
615
|
}
|
|
673
616
|
kind = 'redirect';
|
|
674
617
|
fix = suggestRedirect(affectedKey ?? '', [], redirectFormat ?? 'htaccess', knownTarget);
|
|
675
618
|
break;
|
|
676
619
|
}
|
|
677
620
|
case 'broken-internal-links': {
|
|
678
|
-
// Genuinely dead target (4xx/5xx) — no recorded destination exists, so fall back
|
|
679
|
-
// to token-overlap fuzzy matching against live pages.
|
|
680
621
|
const livePages = db.db
|
|
681
622
|
.prepare('SELECT url FROM pages WHERE status_code = 200 AND indexable = 1')
|
|
682
623
|
.all();
|
|
@@ -687,11 +628,8 @@ export function createServer() {
|
|
|
687
628
|
case 'orphan-with-impressions':
|
|
688
629
|
case 'striking-distance': {
|
|
689
630
|
const page = db.db.prepare('SELECT h1, title FROM pages WHERE url_key = ?').get(affectedKey);
|
|
690
|
-
// Prefer the GSC query as anchor, but fall back to H1/title when it's a
|
|
691
|
-
// boolean/over-long search string (common on job boards) — not usable anchor text.
|
|
692
631
|
const q = evidence.query;
|
|
693
632
|
const cleanQuery = q && q.length <= 60 && !/["()]|\bor\b|\bnot\b|\s-\w/i.test(q) ? q : undefined;
|
|
694
|
-
// Never fall back to the raw q — it was just rejected as unusable anchor text.
|
|
695
633
|
const anchor = cleanQuery ?? page?.h1 ?? page?.title ?? '';
|
|
696
634
|
kind = 'internal-links';
|
|
697
635
|
fix = suggestInternalLinks(db.db, affectedKey ?? '', anchor);
|
|
@@ -721,7 +659,6 @@ export function createServer() {
|
|
|
721
659
|
const key = urlKey(url, { hostForm });
|
|
722
660
|
return { content: [{ type: 'text', text: key }], structuredContent: { url, key, hostForm } };
|
|
723
661
|
});
|
|
724
|
-
// ── GSC ─────────────────────────────────────────────────────────────────
|
|
725
662
|
server.registerTool('list_properties', {
|
|
726
663
|
title: 'List GSC properties',
|
|
727
664
|
description: 'List Google Search Console properties accessible to the service account.',
|
|
@@ -733,8 +670,6 @@ export function createServer() {
|
|
|
733
670
|
structuredContent: { properties },
|
|
734
671
|
};
|
|
735
672
|
});
|
|
736
|
-
// refresh_property — the "sync everything" verb (GSC + crawl + inspection in one job).
|
|
737
|
-
// Skip flags let the same tool do "just update X".
|
|
738
673
|
registerAppTool(server, 'refresh_property', {
|
|
739
674
|
title: 'Refresh a property (sync + crawl + inspect)',
|
|
740
675
|
description: 'Full refresh for a property in one async job: GSC sync → site crawl → URL inspection → DataForSEO rank history. Opens a live progress widget (phases + counts). Set gsc/crawl/inspect/ranks=false to run just part. GSC sync is "lite" (date×query×page) by default — set segments=true to also pull device/country breakdowns (much heavier on large sites). GSC sync is INCREMENTAL: the first sync pulls the window (default last 90 days; pass startDate to go deeper), later syncs only fetch new days — set full=true to force a full re-pull. Use this for "sync everything"; use the single-purpose tools to update just one thing.',
|
|
@@ -756,8 +691,6 @@ export function createServer() {
|
|
|
756
691
|
}, async ({ siteUrl, gsc: doGsc, crawl, inspect, ranks, segments, full, location, startDate, endDate, maxPages, inspectLimit }) => {
|
|
757
692
|
const jobId = jobs.start('refresh', async (update, signal) => {
|
|
758
693
|
const r = await refresh.run(siteUrl, { gsc: doGsc, crawl, inspect, ranks, segments, full, location, startDate, endDate, maxPages, inspectLimit }, update, signal);
|
|
759
|
-
// Property is now populated — spin up the browser dashboard and hand back its URL so the
|
|
760
|
-
// finished job (polled via check_sync_status) carries a one-click link, no extra tool call.
|
|
761
694
|
await autoServeDashboard();
|
|
762
695
|
const dashboardUrl = dashboardServerUrl();
|
|
763
696
|
return dashboardUrl ? { ...r, dashboardUrl, dashboard: `${dashboardUrl}/dashboard?siteUrl=${encodeURIComponent(siteUrl)}` } : r;
|
|
@@ -820,7 +753,6 @@ export function createServer() {
|
|
|
820
753
|
structuredContent: { jobId, status: 'running', siteUrl },
|
|
821
754
|
};
|
|
822
755
|
});
|
|
823
|
-
// ── Crawl ─────────────────────────────────────────────────────────────────
|
|
824
756
|
server.registerTool('start_crawl', {
|
|
825
757
|
title: 'Crawl a site (just the crawl)',
|
|
826
758
|
description: 'Update just the crawl: fetch the site into the local database (async job — poll with check_crawl_status). HTTP crawl; respects robots.txt; asset file-types are HEAD-only (no body download); internal-search / cart / wp-json / builder junk URLs are skipped by default. For a full refresh use refresh_property. excludePatterns adds extra URL regexes to skip (e.g. ["/author/","/tag/","/page/"]) to keep big crawls light. Grain: one row per url_key, latest crawl. Joins: url_key → GSC/inspection/backlinks.',
|
|
@@ -854,7 +786,6 @@ export function createServer() {
|
|
|
854
786
|
const all = jobs.list();
|
|
855
787
|
return { content: [{ type: 'text', text: `${all.length} jobs` }], structuredContent: { jobs: all } };
|
|
856
788
|
});
|
|
857
|
-
// ── DataForSEO (cached 20 days, single-worker) ──────────────────────────
|
|
858
789
|
server.registerTool('keyword_volume', {
|
|
859
790
|
title: 'Keyword search volume (DataForSEO)',
|
|
860
791
|
description: '[Paid: Keywords API, cheap, cached 20d | Use for: demand sizing] True monthly search volume + CPC + competition for keywords (DataForSEO KEYWORDS_DATA). Served from a 20-day cache; live calls are serialised. Default location: United States (2840).',
|
|
@@ -924,7 +855,6 @@ export function createServer() {
|
|
|
924
855
|
articles.push(a);
|
|
925
856
|
} };
|
|
926
857
|
let cost = 0, cached = true, googleCount = 0, dfsCount = 0, googleError = null;
|
|
927
|
-
// Free Google News RSS.
|
|
928
858
|
if (src === 'google' || src === 'both') {
|
|
929
859
|
try {
|
|
930
860
|
const g = await fetchGoogleNews(keyword, { limit: n });
|
|
@@ -936,7 +866,6 @@ export function createServer() {
|
|
|
936
866
|
googleError = e instanceof Error ? e.message : 'failed';
|
|
937
867
|
}
|
|
938
868
|
}
|
|
939
|
-
// Paid DataForSEO Google News SERP (only when explicitly asked, or 'both' AND a key is set).
|
|
940
869
|
if (src === 'dataforseo' || (src === 'both' && !!dfs)) {
|
|
941
870
|
const r = await requireDfs(dfs).serpNews(keyword, location, languageCode, n);
|
|
942
871
|
cost += r.cost;
|
|
@@ -945,8 +874,6 @@ export function createServer() {
|
|
|
945
874
|
for (const a of r.articles)
|
|
946
875
|
add({ ...a, via: 'dataforseo' });
|
|
947
876
|
}
|
|
948
|
-
// Top sources: which publishers are covering this topic, ranked by article count — the
|
|
949
|
-
// "who is talking about X" read. Powers "trending news + sources for X" in one call.
|
|
950
877
|
const sourceCount = new Map();
|
|
951
878
|
for (const a of articles) {
|
|
952
879
|
const s = String(a.source ?? '').trim();
|
|
@@ -1010,8 +937,6 @@ export function createServer() {
|
|
|
1010
937
|
db.close();
|
|
1011
938
|
}
|
|
1012
939
|
});
|
|
1013
|
-
// content_opportunities — the content marketer's report: everything the stored data
|
|
1014
|
-
// says about what to WRITE, REFRESH and REWRITE, in one free composition.
|
|
1015
940
|
server.registerTool('content_opportunities', {
|
|
1016
941
|
title: 'Content opportunity report (write / refresh / rewrite)',
|
|
1017
942
|
description: 'The content marketer\'s report, composed entirely from stored data - NO paid calls. Four sections: WRITE NEXT (new pages proposed from queries you already earn impressions for but have no winning page - suggest_pages), REFRESH NOW (pages that lost 20%+ of their clicks vs the prior period - content decay), REWRITE SNIPPETS (page-1 rankings earning far below expected CTR - title/meta rewrites, the fastest wins), and STRENGTHEN (keyword clusters where you rank 4-20 - one push from the money positions). Every line traces to real Search Console data. Chain into draft_content for a brief, or keyword_volume to size a cluster against the market.',
|
|
@@ -1063,7 +988,6 @@ export function createServer() {
|
|
|
1063
988
|
},
|
|
1064
989
|
};
|
|
1065
990
|
});
|
|
1066
|
-
// ── Content recon (recon_targets) — the data-intensive "why are we losing, what to do" mission ──
|
|
1067
991
|
server.registerTool('recon_targets', {
|
|
1068
992
|
title: 'Content recon: why a page is losing, and what to do about it',
|
|
1069
993
|
description: '[Paid: DataForSEO SERP per page (~$0.004 each, plus a small refundable surcharge for loading async AI Overviews), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced, depth 20 organic), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, competitive-gap, or page-cannot-rank (the crawl says the URL is noindex/canonicalised away, so its GSC history is legacy and the SERP read belongs to another page). Honesty rules baked in: every GSC figure carries its 28-day window; organicRank:null means "absent from the top 20 ORGANIC results", never a position; and an AI Overview whose citations could not be resolved reports aioCitesUs:null (UNKNOWN) rather than "not cited". To-dos are prioritised by OPPORTUNITY (impressions x the CTR gap between where you rank and a realistic target, damped by the verdict), not by raw impressions, and cannibalisation is counted across the whole query cluster. Set scrapeCompetitors:true to also pull the top competitors as content, routed by host: YouTube/video → Supadata transcript (SUPADATA_API_KEY), Reddit → its .json, other pages → Firecrawl (FIRECRAWL_API_KEY) with a free HTTP fallback; competitorLimit (default 5) caps how many are fetched and everything above the cap is listed as skipped. Cloudflare-challenge sites (e.g. PCMag) still can\'t be fetched from a server and come back as a per-URL error — the SERP still tells you they rank; if one matters, ask the user to paste its copy or supply a text file and diff that in. Transcribe the videos (usually what wins these SERPs) and use the reachable pages. Then write findings back with save_recon_todo; track with recon_todos. **Async job:** returns a jobId immediately - poll check_sync_status; the finished job carries the per-page verdicts, to-dos and a summary. (Each page is persisted to the ledger as it completes, so recon_todos shows results even mid-run.)',
|
|
@@ -1078,10 +1002,8 @@ export function createServer() {
|
|
|
1078
1002
|
crawlAs: z.enum(['browser', 'googlebot']).optional().describe('UA for the free HTTP fetch: browser (default, mimics a visit from Google — gets Reddit + mid-tier) or googlebot'),
|
|
1079
1003
|
},
|
|
1080
1004
|
}, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, competitorLimit, crawlAs }) => {
|
|
1081
|
-
const client = requireDfs(dfs);
|
|
1005
|
+
const client = requireDfs(dfs);
|
|
1082
1006
|
const count = urls?.length ?? limit ?? 5;
|
|
1083
|
-
// Async job: N live page-fetches + N serialised SERP calls exceed the ~60s MCP ceiling
|
|
1084
|
-
// past a handful of pages, so return a jobId and poll (like refresh_property).
|
|
1085
1007
|
const jobId = jobs.start('recon', async (update, signal) => {
|
|
1086
1008
|
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
1087
1009
|
try {
|
|
@@ -1092,8 +1014,6 @@ export function createServer() {
|
|
|
1092
1014
|
const ownDomain = dfsHost(siteUrl);
|
|
1093
1015
|
const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
|
|
1094
1016
|
const SERP_DEPTH = 20;
|
|
1095
|
-
// Every GSC number below is this window and only this window. Undeclared, a 28-day figure
|
|
1096
|
-
// read against an all-time baseline looks exactly like a decline.
|
|
1097
1017
|
const windowStart = db.db.prepare(`SELECT date(?, '-27 days') d`).get(fresh.effectiveMax);
|
|
1098
1018
|
const gscWindow = { start: windowStart.d, end: fresh.effectiveMax, days: 28, metric: 'Search Console, last 28 days of synced data (not all-time)' };
|
|
1099
1019
|
let targets;
|
|
@@ -1124,12 +1044,10 @@ export function createServer() {
|
|
|
1124
1044
|
try {
|
|
1125
1045
|
own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
|
|
1126
1046
|
}
|
|
1127
|
-
catch {
|
|
1047
|
+
catch { }
|
|
1128
1048
|
const serpResp = await client.serpOrganic(t.query, location, 'en', SERP_DEPTH, { loadAsyncAiOverview: true });
|
|
1129
1049
|
cost += serpResp.cost;
|
|
1130
1050
|
const serp = parseSerpForRecon(serpResp, ownDomain, SERP_DEPTH);
|
|
1131
|
-
// The crawl row overrides the SERP/GSC read: GSC keeps reporting impressions for URLs
|
|
1132
|
-
// that have since been canonicalised away or set noindex.
|
|
1133
1051
|
const state = pageState(db.db, t.urlKey);
|
|
1134
1052
|
const serpVerdict = reconVerdict(serp);
|
|
1135
1053
|
const verdict = crawlRealityOverride(serpVerdict, state) ?? serpVerdict;
|
|
@@ -1148,10 +1066,6 @@ export function createServer() {
|
|
|
1148
1066
|
let competitorContent = undefined;
|
|
1149
1067
|
let competitorFetch = undefined;
|
|
1150
1068
|
if (scrapeCompetitors && (firecrawl || supadata)) {
|
|
1151
|
-
// Build a routed candidate set: organic-above + AI-Overview references + one ranking
|
|
1152
|
-
// video, deduped, our own domain removed. Each URL routes by HOST (YouTube→supadata,
|
|
1153
|
-
// Reddit→.json, else→firecrawl). What we don't fetch is REPORTED, not silently dropped
|
|
1154
|
-
// — an undeclared cap of 3 is what made this look like it returned nothing.
|
|
1155
1069
|
const seen = new Set();
|
|
1156
1070
|
const candidates = [];
|
|
1157
1071
|
const add = (url, from) => {
|
|
@@ -1164,8 +1078,6 @@ export function createServer() {
|
|
|
1164
1078
|
serp.aioReferences.slice(0, 6).forEach(r => add(r.url, 'aio-reference'));
|
|
1165
1079
|
serp.videoItems.slice(0, 2).forEach(v => add(v.url, 'video-pack'));
|
|
1166
1080
|
const lim = competitorLimit ?? 5;
|
|
1167
|
-
// Organic-above first (the editorial pages actually worth diffing), but keep one slot
|
|
1168
|
-
// for a ranking video — video is usually what wins these SERPs.
|
|
1169
1081
|
const video = candidates.find(c => c.route === 'video');
|
|
1170
1082
|
const rest = candidates.filter(c => c !== video);
|
|
1171
1083
|
const reserve = video && lim >= 3 ? 1 : 0;
|
|
@@ -1235,12 +1147,6 @@ export function createServer() {
|
|
|
1235
1147
|
return note;
|
|
1236
1148
|
})() +
|
|
1237
1149
|
`\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
|
|
1238
|
-
// The poll (check_sync_status) injects this result into the model's context, which the
|
|
1239
|
-
// host caps (~25k tokens). The full per-page detail (heading lists, competitor URL arrays,
|
|
1240
|
-
// full to-do objects, cannibalisation clusters) is already persisted to the ledger and read
|
|
1241
|
-
// back via recon_todos, so the job result returns the readable `summary` plus a SLIM per-page
|
|
1242
|
-
// view — verdict, ranks, opportunity — and keeps only competitorContent (the scraped research
|
|
1243
|
-
// payload, which lives nowhere else). This keeps a 10-page poll well under the cap.
|
|
1244
1150
|
const slimTargets = results.map(r => ({
|
|
1245
1151
|
urlKey: r.urlKey, query: r.query, window: r.window,
|
|
1246
1152
|
impressions: r.impressions, gscPosition: r.gscPosition, priorPosition: r.priorPosition, slipped: r.slipped,
|
|
@@ -1262,8 +1168,6 @@ export function createServer() {
|
|
|
1262
1168
|
structuredContent: { jobId, status: 'running', siteUrl },
|
|
1263
1169
|
};
|
|
1264
1170
|
});
|
|
1265
|
-
// save_recon_todo — the research session writes its content-gap / originality findings back
|
|
1266
|
-
// into the ledger against a page (source: 'research').
|
|
1267
1171
|
server.registerTool('save_recon_todo', {
|
|
1268
1172
|
title: 'Save content-recon to-dos (research writeback)',
|
|
1269
1173
|
description: 'Write content-recon findings back into the trackable ledger for a page — the gaps and originality the research session found by diffing competitors (firecrawl) and videos (supadata) against our content. Each to-do is an action with a type (content-gap / originality / schema / freshness / format), rationale and evidence. Snapshots the page baseline so the fix\'s effect on rank/AIO-citation is measurable later. Run recon_targets first (it classifies the page and seeds the deterministic to-dos); this adds the judgement ones. Track everything with recon_todos.',
|
|
@@ -1283,11 +1187,7 @@ export function createServer() {
|
|
|
1283
1187
|
try {
|
|
1284
1188
|
const key = urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' });
|
|
1285
1189
|
const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position, gsc_impressions, verdict FROM recon_page WHERE url_key=?`).get(key);
|
|
1286
|
-
// aio_cites_us is NULL when the AI Overview never resolved — keep that as unknown rather
|
|
1287
|
-
// than coercing it to false, or the outcome diff reads "no → yes" off a missing fetch.
|
|
1288
1190
|
const baseline = page ? { organicRank: page.organic_rank, aioCitesUs: page.aio_cites_us == null ? null : !!page.aio_cites_us, gscPosition: page.gsc_position, at: new Date().toISOString().slice(0, 10) } : {};
|
|
1289
|
-
// Default priority to the PAGE's opportunity so research to-dos sort on the same scale as
|
|
1290
|
-
// the deterministic ones (a default of 0 buried every judgement finding at the bottom).
|
|
1291
1191
|
const basis = page ? opportunityBasis(page.gsc_impressions ?? 0, page.organic_rank, page.gsc_position ?? 20, page.verdict) : null;
|
|
1292
1192
|
const fallbackPriority = basis ? Math.max(1, Math.round(basis.opportunityClicks * basis.verdictFactor)) : 0;
|
|
1293
1193
|
const drafts = todos.map(t => ({ action: t.action, type: t.type ?? 'content-gap', rationale: t.rationale ?? '', evidence: t.evidence ?? {}, priority: t.priority ?? fallbackPriority }));
|
|
@@ -1298,8 +1198,6 @@ export function createServer() {
|
|
|
1298
1198
|
db.close();
|
|
1299
1199
|
}
|
|
1300
1200
|
});
|
|
1301
|
-
// recon_todos — list, track and annotate the ledger. No id → list (optionally filtered);
|
|
1302
|
-
// id → update status and/or append a dated annotation, and optionally re-measure the outcome.
|
|
1303
1201
|
server.registerTool('recon_todos', {
|
|
1304
1202
|
title: 'List, track and annotate content-recon to-dos',
|
|
1305
1203
|
description: 'The content-recon to-do board. With no id: list to-dos (optionally filter by page or status), grouped by page with each page\'s verdict — the pick-a-page-to-work-on surface, and the hand-off to content-machine. With id: update one to-do — set status (open → researching → drafted → shipped → dismissed) and/or append a dated annotation note (your own observations, the history). On status:shipped with remeasure:true it re-fetches the SERP and records the outcome, so you can see whether the fix moved you from AIO-uncited to cited, or up the organic ranks.',
|
|
@@ -1325,7 +1223,6 @@ export function createServer() {
|
|
|
1325
1223
|
let outcome = row.outcome;
|
|
1326
1224
|
let outcomeNote = '';
|
|
1327
1225
|
if (status === 'shipped' && remeasure && dfs) {
|
|
1328
|
-
// Same depth + async-AIO handling as recon_targets, so before/after are comparable.
|
|
1329
1226
|
const serpResp = await dfs.serpOrganic(row.query, location, 'en', 20, { loadAsyncAiOverview: true });
|
|
1330
1227
|
const serp = parseSerpForRecon(serpResp, dfsHost(siteUrl), 20);
|
|
1331
1228
|
const cited = (v) => (v == null ? 'unknown' : v ? 'yes' : 'no');
|
|
@@ -1371,8 +1268,6 @@ export function createServer() {
|
|
|
1371
1268
|
db.close();
|
|
1372
1269
|
}
|
|
1373
1270
|
});
|
|
1374
|
-
// keyword_list — demand-first clustering ("list mode"): a keyword list becomes topics
|
|
1375
|
-
// with own/weak/absent verdicts, clustered by the URL Google already answers them with.
|
|
1376
1271
|
server.registerTool('keyword_list', {
|
|
1377
1272
|
title: 'Cluster a keyword list into topics with own/weak/absent verdicts',
|
|
1378
1273
|
description: 'Demand-first keyword clustering from stored data - NO paid calls. Give it a keyword list (or omit keywords to use your top 500 GSC queries) and it clusters them by the page Google ALREADY answers them with (two keywords that rank via the same URL belong together - the strongest clustering signal, free from your own GSC data), then groups non-ranking keywords lexically. Each cluster gets a deterministic verdict: OWN (best position <=3), WEAK (4-20), ABSENT (no ranking page), plus the ranking URL, summed 90-day impressions and clicks. The keyword-research workhorse: paste a client keyword list, get the topic map and where you stand. Chain with keyword_volume for market volumes on the interesting clusters, or draft_content for the absent ones.',
|
|
@@ -1501,7 +1396,7 @@ export function createServer() {
|
|
|
1501
1396
|
const cleaned = target.replace(/^sc-domain:/, '').replace(/^https?:\/\//, '').replace(/^www\./, '').replace(/\/+$/, '');
|
|
1502
1397
|
const r = await client.competitorsDomain(cleaned, location, languageCode, limit ?? 20);
|
|
1503
1398
|
const items = (r.tasks[0]?.result?.[0]?.items ?? [])
|
|
1504
|
-
.filter((it) => it.domain && it.domain.replace(/^www\./, '') !== cleaned)
|
|
1399
|
+
.filter((it) => it.domain && it.domain.replace(/^www\./, '') !== cleaned)
|
|
1505
1400
|
.map((it) => ({
|
|
1506
1401
|
domain: it.domain,
|
|
1507
1402
|
intersections: it.intersections ?? null,
|
|
@@ -1529,7 +1424,7 @@ export function createServer() {
|
|
|
1529
1424
|
const r = await client.pageIntersection(competitorUrls, excludePages ?? [], location, languageCode, limit ?? 100);
|
|
1530
1425
|
const items = (r.tasks[0]?.result?.[0]?.items ?? [])
|
|
1531
1426
|
.map((it) => {
|
|
1532
|
-
const kd = it.keyword_data ?? {};
|
|
1427
|
+
const kd = it.keyword_data ?? {};
|
|
1533
1428
|
return {
|
|
1534
1429
|
keyword: kd.keyword ?? null,
|
|
1535
1430
|
searchVolume: kd.keyword_info?.search_volume ?? null,
|
|
@@ -1540,16 +1435,13 @@ export function createServer() {
|
|
|
1540
1435
|
})),
|
|
1541
1436
|
};
|
|
1542
1437
|
})
|
|
1543
|
-
.sort((a, b) => (b.searchVolume ?? 0) - (a.searchVolume ?? 0));
|
|
1438
|
+
.sort((a, b) => (b.searchVolume ?? 0) - (a.searchVolume ?? 0));
|
|
1544
1439
|
return {
|
|
1545
1440
|
content: [{ type: 'text', text: `${items.length} gap keywords${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}` }],
|
|
1546
1441
|
structuredContent: { gaps: items, cached: r.cached, cost: r.cost },
|
|
1547
1442
|
};
|
|
1548
1443
|
});
|
|
1549
|
-
// Strip any scheme/sc-domain:/path down to a bare host for Labs domain targets.
|
|
1550
|
-
// Always strips leading www. — Labs treats www.example.com as a subdomain, not the domain.
|
|
1551
1444
|
const dfsHost = (t) => t.replace(/^sc-domain:/, '').replace(/^https?:\/\//, '').replace(/\/.*$/, '').replace(/^www\./, '').trim();
|
|
1552
|
-
// Cap a markdown table well under the ~40k model-facing ceiling: keep whole rows, note the rest.
|
|
1553
1445
|
const capMdRows = (header, rows, footer = '', maxChars = 30000) => {
|
|
1554
1446
|
let out = header;
|
|
1555
1447
|
let used = 0;
|
|
@@ -1581,7 +1473,6 @@ export function createServer() {
|
|
|
1581
1473
|
const r = await client.historicalRankOverview(cleaned, location, 'en', languageName);
|
|
1582
1474
|
const items = r.tasks[0]?.result?.[0]?.items ?? [];
|
|
1583
1475
|
const n = (v) => Number(v) || 0;
|
|
1584
|
-
// Guard year/month — a malformed item would emit an "undefined-NaN" period row.
|
|
1585
1476
|
const series = items
|
|
1586
1477
|
.filter((it) => Number.isInteger(it?.year) && Number.isInteger(it?.month))
|
|
1587
1478
|
.map((it) => {
|
|
@@ -1641,7 +1532,7 @@ export function createServer() {
|
|
|
1641
1532
|
.map((it) => {
|
|
1642
1533
|
const page = it.page_address ?? it.relative_url ?? null;
|
|
1643
1534
|
if (!page)
|
|
1644
|
-
return null;
|
|
1535
|
+
return null;
|
|
1645
1536
|
const o = it.metrics?.organic ?? {};
|
|
1646
1537
|
const top3 = n(o.pos_1) + n(o.pos_2_3);
|
|
1647
1538
|
return {
|
|
@@ -1698,14 +1589,9 @@ export function createServer() {
|
|
|
1698
1589
|
const folderPath = mode === 'folder'
|
|
1699
1590
|
? (folder.trim().startsWith('/') ? folder.trim() : '/' + folder.trim())
|
|
1700
1591
|
: null;
|
|
1701
|
-
// aioOnly: only rows where the target's ranked element is an AI Overview citation —
|
|
1702
|
-
// expressed as a serp_item.type filter (item_types is invalid on this Labs endpoint).
|
|
1703
1592
|
const clauses = [];
|
|
1704
1593
|
if (folderPath)
|
|
1705
1594
|
clauses.push(['ranked_serp_element.serp_item.relative_url', 'like', `${folderPath}%`]);
|
|
1706
|
-
// Exposure, not citation: Labs ranked_keywords carries no citation fields (verified via
|
|
1707
|
-
// available_filters) — this filters to keywords whose SERP CONTAINS an AI Overview.
|
|
1708
|
-
// Per-keyword citation checking (are WE a reference?) is the SERP-advanced recipe.
|
|
1709
1595
|
if (aioOnly)
|
|
1710
1596
|
clauses.push(['keyword_data.serp_info.serp_item_types', 'has', 'ai_overview']);
|
|
1711
1597
|
const filters = clauses.length === 0 ? undefined : clauses.length === 1 ? clauses : [clauses[0], 'and', clauses[1]];
|
|
@@ -1717,7 +1603,7 @@ export function createServer() {
|
|
|
1717
1603
|
const kd = it.keyword_data ?? {};
|
|
1718
1604
|
const serp = it.ranked_serp_element?.serp_item ?? {};
|
|
1719
1605
|
if (!kd.keyword)
|
|
1720
|
-
return null;
|
|
1606
|
+
return null;
|
|
1721
1607
|
const serpFeatures = Array.isArray(kd.serp_info?.serp_item_types) ? kd.serp_info.serp_item_types : null;
|
|
1722
1608
|
return {
|
|
1723
1609
|
keyword: kd.keyword,
|
|
@@ -1725,7 +1611,6 @@ export function createServer() {
|
|
|
1725
1611
|
searchVolume: kd.keyword_info?.search_volume ?? null,
|
|
1726
1612
|
etv: serp.etv != null ? Math.round(Number(serp.etv) * 100) / 100 : null,
|
|
1727
1613
|
url: serp.relative_url ?? serp.url ?? null,
|
|
1728
|
-
// Fields we already pay for but previously dropped (plan/data-utilisation.md Part 2 §4):
|
|
1729
1614
|
keywordDifficulty: kd.keyword_properties?.keyword_difficulty ?? null,
|
|
1730
1615
|
intent: kd.search_intent_info?.main_intent ?? null,
|
|
1731
1616
|
serpFeatures,
|
|
@@ -1744,7 +1629,6 @@ export function createServer() {
|
|
|
1744
1629
|
}
|
|
1745
1630
|
const totalCount = Number(result.total_count) || kws.length;
|
|
1746
1631
|
const sumEtv = kws.reduce((s, k) => s + (k.etv ?? 0), 0);
|
|
1747
|
-
// Optional columns — only where the response actually carries the field (older cache entries won't).
|
|
1748
1632
|
const hasKd = kws.some(k => k.keywordDifficulty != null);
|
|
1749
1633
|
const hasIntent = kws.some(k => k.intent != null);
|
|
1750
1634
|
const hasFeatures = kws.some(k => (k.serpFeatures && k.serpFeatures.length) || k.isFeaturedSnippet);
|
|
@@ -1767,9 +1651,6 @@ export function createServer() {
|
|
|
1767
1651
|
structuredContent: { target: dfsTarget, scope: mode, aioOnly: aioOnly ?? false, totalCount, rowsTotal: kws.length, keywords: kws.slice(0, 100), cached: r.cached, cost: r.cost },
|
|
1768
1652
|
};
|
|
1769
1653
|
});
|
|
1770
|
-
// serp_features — the SERP-feature footprint: "how much of my market do AI Overviews,
|
|
1771
|
-
// snippets and other features sit on, and do I already rank page 1 there?" One cached
|
|
1772
|
-
// Labs pull, volume-weighted; persisted so the dashboard can chart it.
|
|
1773
1654
|
server.registerTool('serp_features', {
|
|
1774
1655
|
title: 'SERP-feature footprint (AI Overviews, snippets, PAA)',
|
|
1775
1656
|
description: '[Paid: Labs, ONE cached call | Use for: AI-Overview / zero-click exposure with real numbers] How much of your keyword universe carries each SERP feature - AI Overviews, featured snippets, People Also Ask, shopping, video - weighted by search volume, and how much of that volume you already rank page 1 for. ONE DataForSEO Labs ranked_keywords pull (top-volume sample, cached 20 days, never per-keyword SERP loops). Answers "how exposed are we to AI Overviews / zero-click?" with real numbers. Deterministic: feature PRESENCE (from the Labs index). Judgement proxy: page-1 rank stands in for feature ownership - true ownership needs per-keyword SERP calls (see ranked_keywords aioOnly + the cookbook). Persists to the property DB so the dashboard charts it.',
|
|
@@ -1814,8 +1695,6 @@ export function createServer() {
|
|
|
1814
1695
|
structuredContent: fp,
|
|
1815
1696
|
};
|
|
1816
1697
|
});
|
|
1817
|
-
// market_sizing — Market Sizing and Prioritisation: the organic market read that opens
|
|
1818
|
-
// an engagement. Top-down Labs pulls only (client + <=4 rivals), cached, ~$0.65 worst case.
|
|
1819
1698
|
server.registerTool('market_sizing', {
|
|
1820
1699
|
title: 'Market Sizing and Prioritisation (share of voice vs competitors)',
|
|
1821
1700
|
description: '[Paid: Labs, one cached call per domain (<=5) | Use for: sizing the organic market and who owns it] Build the organic market map: your domain plus up to 4 named competitors, ONE cached Labs ranked_keywords pull each, unioned into a keyword universe. Returns total monthly demand (deduplicated search volume), each domain\'s share of voice (ETV share) overall and per topic cluster, and the leader per cluster - the "here is the market, here is who owns it, here is where to attack" table that opens an engagement. Deterministic: the competitor set and ranked keywords (Labs index). Judgement: ETV is DataForSEO\'s CTR-curve traffic estimate - the SoV percentages inherit that. Persists to the property DB for the dashboard chart. Get the competitor set from competitors_domain first if unsure.',
|
|
@@ -1833,7 +1712,7 @@ export function createServer() {
|
|
|
1833
1712
|
const inputs = [];
|
|
1834
1713
|
let cost = 0;
|
|
1835
1714
|
let cachedAll = true;
|
|
1836
|
-
for (const domain of domains) {
|
|
1715
|
+
for (const domain of domains) {
|
|
1837
1716
|
const r = await client.rankedKeywords(domain, location, languageCode ?? 'en', limitPerDomain ?? 1000, 'keyword_data.keyword_info.search_volume,desc');
|
|
1838
1717
|
cost += r.cost;
|
|
1839
1718
|
cachedAll = cachedAll && r.cached;
|
|
@@ -1899,8 +1778,6 @@ export function createServer() {
|
|
|
1899
1778
|
const ourHost = dfsHost(siteUrl);
|
|
1900
1779
|
let totalCost = 0, liveCalls = 0, cachedCalls = 0;
|
|
1901
1780
|
const tally = (r) => { totalCost += r.cost; r.cached ? cachedCalls++ : liveCalls++; };
|
|
1902
|
-
// Competitors: explicit (≤3) or derived top-2 by keyword overlap. Total Labs
|
|
1903
|
-
// budget stays ≤4 calls: at most 1 discovery + at most 3 footprint pulls.
|
|
1904
1781
|
let comps = [...new Set((competitors ?? []).map(dfsHost).filter(c => c && c !== ourHost))].slice(0, 3);
|
|
1905
1782
|
let derived = false;
|
|
1906
1783
|
if (!comps.length) {
|
|
@@ -1919,7 +1796,6 @@ export function createServer() {
|
|
|
1919
1796
|
};
|
|
1920
1797
|
}
|
|
1921
1798
|
}
|
|
1922
|
-
// One ranked_keywords pull per competitor (top 1000 rows by ETV — their money keywords).
|
|
1923
1799
|
const rows = [];
|
|
1924
1800
|
const perCompetitor = {};
|
|
1925
1801
|
for (const c of comps) {
|
|
@@ -1968,8 +1844,6 @@ export function createServer() {
|
|
|
1968
1844
|
`${res.competitorKeywords} distinct competitor keywords → ${res.afterSubtraction} survive subtraction of your footprint (${fmtVol(res.ourQueryCount)} GSC queries + page titles/H1s) → top ${res.clusters.length} topic clusters by volume × competitor coverage.\n\n` +
|
|
1969
1845
|
sections.join('\n\n') +
|
|
1970
1846
|
`\n\n_${costLine} Competitor rows: ${comps.map(c => `${c} ${perCompetitor[c] ?? 0}`).join(', ')}._`;
|
|
1971
|
-
// structuredContent: keyword lists capped at 10 per cluster with explicit totals
|
|
1972
|
-
// (host ~60k model-facing ceiling) — the counts always state the true size.
|
|
1973
1847
|
const scClusters = res.clusters.map(c => ({
|
|
1974
1848
|
...c,
|
|
1975
1849
|
keywords: c.keywords.slice(0, 10),
|
|
@@ -1989,16 +1863,12 @@ export function createServer() {
|
|
|
1989
1863
|
db.close();
|
|
1990
1864
|
}
|
|
1991
1865
|
});
|
|
1992
|
-
// ── Dashboard (MCP App UI — houtini design + ECharts) ───────────────────
|
|
1993
1866
|
registerAppTool(server, 'get_dashboard', {
|
|
1994
1867
|
title: 'SEO dashboard',
|
|
1995
1868
|
description: 'Interactive dashboard for a property: summary metrics, rank & clicks over time, and top-keyword performance (click a keyword for related terms). Needs synced GSC data — run refresh_property first.',
|
|
1996
1869
|
inputSchema: { siteUrl: z.string() },
|
|
1997
1870
|
_meta: { ui: { resourceUri: DASHBOARD_URI } },
|
|
1998
1871
|
}, async ({ siteUrl }) => {
|
|
1999
|
-
// Return only a TINY model-facing result + the siteUrl; the widget fetches the full
|
|
2000
|
-
// (large) dataset itself via the app-only get_dashboard_data tool, which keeps the
|
|
2001
|
-
// big payload OUT of the model's context/token limit (per the MCP Apps large-data pattern).
|
|
2002
1872
|
const data = getDashboardData(dataDir(), siteUrl);
|
|
2003
1873
|
if (data.empty) {
|
|
2004
1874
|
return { content: [{ type: 'text', text: `No synced data for ${siteUrl} yet — run refresh_property.` }], structuredContent: { siteUrl, empty: true } };
|
|
@@ -2008,9 +1878,6 @@ export function createServer() {
|
|
|
2008
1878
|
`${data.findings ? `, ${data.findings.total} audit findings` : ''}. Interactive charts + findings render in the widget.` + browserLink(siteUrl);
|
|
2009
1879
|
return { content: [{ type: 'text', text: summary }], structuredContent: { siteUrl } };
|
|
2010
1880
|
});
|
|
2011
|
-
// App-only data tool: the dashboard widget calls this via app.callServerTool to fetch its
|
|
2012
|
-
// full dataset. visibility:['app'] hides it from the model; results route to the iframe,
|
|
2013
|
-
// bypassing the model token cap that a large model-facing result would hit.
|
|
2014
1881
|
registerAppTool(server, 'get_dashboard_data', {
|
|
2015
1882
|
title: 'Dashboard data (internal)',
|
|
2016
1883
|
description: 'Full dashboard dataset for the UI widget. App-only — not for direct use.',
|
|
@@ -2020,9 +1887,6 @@ export function createServer() {
|
|
|
2020
1887
|
const data = { ...getDashboardData(dataDir(), siteUrl), apiKeys: apiKeysStatus() };
|
|
2021
1888
|
return { content: [{ type: 'text', text: 'ok' }], structuredContent: data };
|
|
2022
1889
|
});
|
|
2023
|
-
// export_report — the dependable deliverable: a self-contained interactive dashboard
|
|
2024
|
-
// HTML (data inlined) the user opens in any browser / emails to a client. Works
|
|
2025
|
-
// regardless of whether the host renders MCP-App widgets inline.
|
|
2026
1890
|
server.registerTool('export_report', {
|
|
2027
1891
|
title: 'Export a shareable dashboard report (HTML)',
|
|
2028
1892
|
description: 'Write a self-contained, interactive dashboard HTML for a property (all data + charts inlined) to the reports folder, and return the file path. Open it in any browser or send it to a client — no server, no MCP-App host support needed. Run refresh_property (+ run_audit for findings) first.',
|
|
@@ -2033,10 +1897,8 @@ export function createServer() {
|
|
|
2033
1897
|
return { content: [{ type: 'text', text: `No synced data for ${siteUrl} — run refresh_property first.` }], structuredContent: { error: 'empty', siteUrl } };
|
|
2034
1898
|
}
|
|
2035
1899
|
const tpl = readFileSync(path.join(__dirname, 'src', 'ui', 'dashboard.html'), 'utf8');
|
|
2036
|
-
const json = JSON.stringify(data).replace(/</g, '\\u003c');
|
|
1900
|
+
const json = JSON.stringify(data).replace(/</g, '\\u003c');
|
|
2037
1901
|
const inject = `<script>window.__DASH_FIXTURE__=${json};window.__DASH_THEME__=${JSON.stringify(theme ?? 'light')};</script>`;
|
|
2038
|
-
// Replacement FUNCTION, not string — crawl data containing $& / $' would otherwise
|
|
2039
|
-
// be interpreted as String.replace substitution patterns and corrupt the report.
|
|
2040
1902
|
const html = tpl.replace(/<head([^>]*)>/i, (_m, attrs) => `<head${attrs}>${inject}`);
|
|
2041
1903
|
const dir = path.join(dataDir(), 'reports');
|
|
2042
1904
|
mkdirSync(dir, { recursive: true });
|
|
@@ -2047,8 +1909,6 @@ export function createServer() {
|
|
|
2047
1909
|
structuredContent: { path: file, siteUrl, findings: data.findings?.total ?? 0, bytes: html.length },
|
|
2048
1910
|
};
|
|
2049
1911
|
});
|
|
2050
|
-
// serve_dashboard — the local webserver delivery surface: the full dashboard in a real
|
|
2051
|
-
// browser tab (live data, property switcher, native downloads), no MCP-App host needed.
|
|
2052
1912
|
server.registerTool('serve_dashboard', {
|
|
2053
1913
|
title: 'Serve the dashboard on a local webserver',
|
|
2054
1914
|
description: 'Start a localhost-only webserver and return a URL that opens the full interactive dashboard in your browser — live data straight from the local database (always current, unlike export_report snapshots), a property switcher, working CSV downloads, and no host widget limits. The server stays up while the MCP server runs; call again with stop=true to shut it down. Localhost only — nothing is exposed to the network.',
|
|
@@ -2072,8 +1932,6 @@ export function createServer() {
|
|
|
2072
1932
|
structuredContent: { url: open, base: url },
|
|
2073
1933
|
};
|
|
2074
1934
|
});
|
|
2075
|
-
// pull_backlinks — on-demand backlink profile (DataForSEO, paid + 20-day cached). Powers
|
|
2076
|
-
// backlinks-to-404 (the big quick win), top-linked pages, and true-orphan detection.
|
|
2077
1935
|
server.registerTool('pull_backlinks', {
|
|
2078
1936
|
title: 'Pull backlink profile (DataForSEO)',
|
|
2079
1937
|
description: '[Paid: Backlinks subscription (separate - 40204 = not activated), cached 20d | Use for: authority + dead-backlink recovery] Fetch the property’s backlink profile (overall summary — total backlinks, referring domains, Domain Rank, broken backlinks/pages, nofollow share — plus per-page backlink/referring-domain counts) into page_backlinks, and resolve each backlinked page’s live HTTP status so run_audit can flag external backlinks pointing to dead (4xx/5xx) pages. Paid DataForSEO call, 20-day cached, on-demand only. Async job — poll check_sync_status. Grain: one row per backlinked URL. Joins: url_key → pages/GSC; domain → Labs tools.',
|
|
@@ -2086,9 +1944,6 @@ export function createServer() {
|
|
|
2086
1944
|
structuredContent: { jobId, status: 'running', siteUrl },
|
|
2087
1945
|
};
|
|
2088
1946
|
});
|
|
2089
|
-
// link_intersect — "what links do our competitors have that we don't?". One DataForSEO
|
|
2090
|
-
// backlinks/domain_intersection call (paid, ~$0.024, 20-day cached), aggregated + prioritised
|
|
2091
|
-
// client-side. Default sort = the link-builder's: followed links first, then domain trust.
|
|
2092
1947
|
server.registerTool('link_intersect', {
|
|
2093
1948
|
title: 'Link intersect — links your competitors have that you don’t (DataForSEO + Majestic)',
|
|
2094
1949
|
description: '[Paid: Backlinks subscription (separate - 40204 = not activated), one cached call ~$0.024 | Use for: prospect list for link outreach] Answers "what links do our competitors have that we don’t?" - and for a SINGLE company, "what links does company X have that we don’t?" (pass competitors:["companyx.com"]). One DataForSEO domain_intersection call over the target set (excluding your domain), aggregated per prospect domain: how many of the targets it links to, its DataForSEO domain trust (rank 0-1000), worst spam score, whether the link is followed, the anchor/link-type mix. Default sort is the link-builder’s view - FOLLOWED links first, then domain trust - with spam filtered out. When MAJESTIC_API_KEY is set, prospects are enriched with Majestic Trust Flow + Topical Trust Flow and RE-SORTED by Trust Flow (the directory-killer: a DataForSEO rank-227 domain is often Trust Flow 0). Results persist to link_prospects. Pass targets explicitly (max 20) or let it derive the top few via competitors_domain. Grain: one row per prospect domain. Join: domain.',
|
|
@@ -2107,8 +1962,6 @@ export function createServer() {
|
|
|
2107
1962
|
}, async ({ siteUrl, competitors, location, poolLimit, minIntersections, maxSpamScore, dofollowOnly, topN, sort, enrichLimit }) => {
|
|
2108
1963
|
const li = requireDfs(linkIntersect);
|
|
2109
1964
|
const ownHost = dfsHost(siteUrl);
|
|
2110
|
-
// Prior-run awareness: if this property already has link_prospects, frame the run as an
|
|
2111
|
-
// update (the previous set ages as competitors keep earning links). Read-only, cheap.
|
|
2112
1965
|
let priorNote = '';
|
|
2113
1966
|
{
|
|
2114
1967
|
const d = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
@@ -2119,12 +1972,11 @@ export function createServer() {
|
|
|
2119
1972
|
priorNote = ` _(updating a prior intersect of ${prior.n} prospects from ${prior.t.slice(0, 10)}, ${ageDays}d ago)_`;
|
|
2120
1973
|
}
|
|
2121
1974
|
}
|
|
2122
|
-
catch {
|
|
1975
|
+
catch { }
|
|
2123
1976
|
finally {
|
|
2124
1977
|
d.close();
|
|
2125
1978
|
}
|
|
2126
1979
|
}
|
|
2127
|
-
// Competitors: explicit (deduped, minus self) or derived top few by keyword overlap.
|
|
2128
1980
|
let comps = [...new Set((competitors ?? []).map(dfsHost).filter(c => c && c !== ownHost))];
|
|
2129
1981
|
let derived = false;
|
|
2130
1982
|
let deriveCost = 0;
|
|
@@ -2161,7 +2013,6 @@ export function createServer() {
|
|
|
2161
2013
|
...(enrichLimit != null ? { enrichLimit } : {}),
|
|
2162
2014
|
}, majestic);
|
|
2163
2015
|
const enriched = result.majesticEnriched > 0;
|
|
2164
|
-
// With Majestic: show Trust Flow (0-100) + the domain's top topic. Without: DataForSEO domain rank.
|
|
2165
2016
|
const trustHead = enriched ? 'Trust Flow' : 'Domain trust';
|
|
2166
2017
|
const rows = result.prospects.map(p => {
|
|
2167
2018
|
const trust = enriched ? (p.trustFlow ?? '–') : (p.domainTrust ?? '–');
|
|
@@ -2193,7 +2044,6 @@ export function createServer() {
|
|
|
2193
2044
|
structuredContent: { jobId, status: 'running', siteUrl },
|
|
2194
2045
|
};
|
|
2195
2046
|
});
|
|
2196
|
-
// ── Static reference resources (same content the tools return — one source of truth) ──
|
|
2197
2047
|
server.registerResource('checks-reference', 'seo-audit://checks-reference', {
|
|
2198
2048
|
title: 'Check registry reference',
|
|
2199
2049
|
description: 'The full audit check catalogue (the list_checks data) rendered as markdown: every check with category, severity, labels, certainty, fix type and its one-line fix.',
|