@houtini/seo-audit-console 0.3.0 → 0.4.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +8 -3
  2. package/dist/audit/recon.d.ts +124 -0
  3. package/dist/audit/recon.d.ts.map +1 -0
  4. package/dist/audit/recon.js +257 -0
  5. package/dist/audit/recon.js.map +1 -0
  6. package/dist/core/AuditDatabase.d.ts.map +1 -1
  7. package/dist/core/AuditDatabase.js +39 -0
  8. package/dist/core/AuditDatabase.js.map +1 -1
  9. package/dist/core/DataForSeoClient.d.ts +9 -2
  10. package/dist/core/DataForSeoClient.d.ts.map +1 -1
  11. package/dist/core/DataForSeoClient.js +11 -3
  12. package/dist/core/DataForSeoClient.js.map +1 -1
  13. package/dist/core/FirecrawlClient.d.ts +39 -0
  14. package/dist/core/FirecrawlClient.d.ts.map +1 -0
  15. package/dist/core/FirecrawlClient.js +86 -0
  16. package/dist/core/FirecrawlClient.js.map +1 -0
  17. package/dist/core/SupadataClient.d.ts +29 -0
  18. package/dist/core/SupadataClient.d.ts.map +1 -0
  19. package/dist/core/SupadataClient.js +63 -0
  20. package/dist/core/SupadataClient.js.map +1 -0
  21. package/dist/core/passageScore.d.ts +40 -9
  22. package/dist/core/passageScore.d.ts.map +1 -1
  23. package/dist/core/passageScore.js +29 -3
  24. package/dist/core/passageScore.js.map +1 -1
  25. package/dist/core/reconFetch.d.ts +29 -0
  26. package/dist/core/reconFetch.d.ts.map +1 -0
  27. package/dist/core/reconFetch.js +49 -0
  28. package/dist/core/reconFetch.js.map +1 -0
  29. package/dist/core/reconResearch.d.ts +50 -0
  30. package/dist/core/reconResearch.d.ts.map +1 -0
  31. package/dist/core/reconResearch.js +122 -0
  32. package/dist/core/reconResearch.js.map +1 -0
  33. package/dist/core/serpRecon.d.ts +52 -0
  34. package/dist/core/serpRecon.d.ts.map +1 -0
  35. package/dist/core/serpRecon.js +99 -0
  36. package/dist/core/serpRecon.js.map +1 -0
  37. package/dist/server.d.ts.map +1 -1
  38. package/dist/server.js +334 -6
  39. package/dist/server.js.map +1 -1
  40. package/package.json +1 -1
  41. package/server.json +18 -4
package/dist/server.js CHANGED
@@ -11,6 +11,12 @@ import { getDashboardData } from './core/dashboardData.js';
11
11
  import { startDashboardServer, stopDashboardServer, dashboardServerUrl, listLocalProperties } from './core/webServer.js';
12
12
  import { computeSerpFootprint, persistSerpFootprint } from './core/serpFootprint.js';
13
13
  import { computeMarketSizing, persistMarketSizing } from './core/marketSizing.js';
14
+ import { FirecrawlClient } from './core/FirecrawlClient.js';
15
+ import { SupadataClient } from './core/SupadataClient.js';
16
+ import { fetchCompetitorContent, routeFor } from './core/reconResearch.js';
17
+ import { fetchOwnPage } from './core/reconFetch.js';
18
+ import { parseSerpForRecon, reconVerdict } from './core/serpRecon.js';
19
+ import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos, pageState, crawlRealityOverride, clusterCannibalisation, opportunityBasis } from './audit/recon.js';
14
20
  /** A clickable browser-dashboard link appended to tool outputs — the user should always
15
21
  * know the full interactive report is one click away (or one serve_dashboard call away). */
16
22
  function browserLink(siteUrl) {
@@ -165,6 +171,7 @@ Raw access: query_audit runs any single check with full evidence; every table ab
165
171
  - **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
166
172
  - **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
167
173
  - **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
174
+ - **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
168
175
  - **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
169
176
 
170
177
  Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
@@ -246,6 +253,12 @@ export function createServer() {
246
253
  : null;
247
254
  const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
248
255
  const backlinks = dfs ? new Backlinks(dfs, dataDir()) : null;
256
+ // Firecrawl (competitor-page scraping for content recon) — optional; degrades gracefully.
257
+ const firecrawlKey = process.env.FIRECRAWL_API_KEY;
258
+ const firecrawl = firecrawlKey ? new FirecrawlClient(firecrawlKey, path.join(dataDir(), 'firecrawl-cache.db')) : null;
259
+ // Supadata (transcribes the ranking videos for content recon) — optional; degrades gracefully.
260
+ const supadataKey = process.env.SUPADATA_API_KEY;
261
+ const supadata = supadataKey ? new SupadataClient(supadataKey, path.join(dataDir(), 'supadata-cache.db')) : null;
249
262
  const entities = new Entities(new WikidataClient(path.join(dataDir(), 'wikidata-cache.db')), dataDir());
250
263
  const refresh = new Refresh(sync, crawler, inspector, rankTracker);
251
264
  const requireGsc = (v) => {
@@ -440,14 +453,21 @@ export function createServer() {
440
453
  });
441
454
  server.registerTool('score_passages', {
442
455
  title: 'Score passage relevance (AI-search readiness)',
443
- description: 'Run a local cross-encoder reranker (ms-marco-MiniLM-L-6-v2, downloaded once to a cache, no Python) over each ranking page\'s heading chunks against its top Search Console query, and persist the single best-passage relevance score. It SCORES relevance (a classifier — it cannot hallucinate). Powers the `weak-passage-answer` check — the "RAG snippetability" test: does any passage confidently answer the query, the way AI/passage search re-ranks? On-demand + ML-heavy (~0.6s/page), bounded to pages that already rank. `limit` caps pages (highest-impression first); `minImpressions` sets the floor (default 50). First run downloads ~25MB. Re-run after a re-crawl.',
444
- inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(2000).optional(), minImpressions: z.number().int().min(0).optional() },
445
- }, async ({ siteUrl, limit, minImpressions }) => {
456
+ description: 'Run a local cross-encoder reranker (ms-marco-MiniLM-L-6-v2, downloaded once to a cache, no Python) over each ranking page\'s heading chunks against its top Search Console query, and persist the single best-passage relevance score. It SCORES relevance (a classifier — it cannot hallucinate). Powers the `weak-passage-answer` check — the "RAG snippetability" test: does any passage confidently answer the query, the way AI/passage search re-ranks? Returns the score for EVERY page it scored (not just the failures) with the flag threshold, the band convention and the run\'s own distribution, so a passing page\'s score is readable without a second query_data hop. On-demand + ML-heavy (~0.6s/page), bounded to pages that already rank. `limit` caps pages (highest-impression first); `minImpressions` sets the floor (default 50). First run downloads ~25MB. Re-run after a re-crawl.',
457
+ inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(2000).optional(), minImpressions: z.number().int().min(0).optional(), urls: z.array(z.string()).optional().describe('Only report these pages (they are still scored site-wide; this filters the returned list)') },
458
+ }, async ({ siteUrl, limit, minImpressions, urls }) => {
446
459
  const r = await scoreSitePassages(dataDir(), siteUrl, { limit, minImpressions });
447
- const lines = r.weakest.slice(0, 15).map(w => ` ${w.score} ${w.url.replace(/^https?:\/\/[^/]+/, '')} · "${w.query}"`).join('\n');
460
+ const keys = urls?.length ? new Set(urls.map(u => urlKey(u, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' }))) : null;
461
+ const shown = keys ? r.pages.filter(p => keys.has(p.url)) : r.pages;
462
+ const rows = shown.slice(0, 25).map(p => ` ${String(p.score).padStart(7)} ${p.band.padEnd(8)} ${p.url.replace(/^https?:\/\/[^/]+/, '')} · "${p.query}" (${p.impressions} impr)`).join('\n');
463
+ const d = r.distribution;
464
+ const text = `Scored ${r.scored} ranking pages; ${r.flagged} below the weak threshold of ${r.threshold} — run_audit surfaces those as weak-passage-answer.\n\n` +
465
+ `Scale: ${r.scale}\n` +
466
+ (d ? `This run: min ${d.min} · p25 ${d.p25} · median ${d.median} · p75 ${d.p75} · max ${d.max}\n` : '') +
467
+ (rows ? `\nScores (${shown.length === r.pagesTotal ? `all ${r.pagesTotal}` : `${shown.length} of ${r.pagesTotal}`}, showing up to 25):\n${rows}` : '');
448
468
  return {
449
- content: [{ type: 'text', text: `Scored ${r.scored} ranking pages; ${r.flagged} have no strongly-relevant passage (score < 3) — run_audit surfaces them as weak-passage-answer.${lines ? `\n\nWeakest:\n${lines}` : ''}` }],
450
- structuredContent: r,
469
+ content: [{ type: 'text', text }],
470
+ structuredContent: { ...r, ...(keys ? { pages: shown, pagesReturned: shown.length } : {}) },
451
471
  };
452
472
  });
453
473
  server.registerTool('draft_content', {
@@ -874,6 +894,314 @@ export function createServer() {
874
894
  },
875
895
  };
876
896
  });
897
+ // ── Content recon (recon_targets) — the data-intensive "why are we losing, what to do" mission ──
898
+ server.registerTool('recon_targets', {
899
+ title: 'Content recon: why a page is losing, and what to do about it',
900
+ description: '[Paid: DataForSEO SERP per page (~$0.004 each, plus a small refundable surcharge for loading async AI Overviews), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced, depth 20 organic), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, competitive-gap, or page-cannot-rank (the crawl says the URL is noindex/canonicalised away, so its GSC history is legacy and the SERP read belongs to another page). Honesty rules baked in: every GSC figure carries its 28-day window; organicRank:null means "absent from the top 20 ORGANIC results", never a position; and an AI Overview whose citations could not be resolved reports aioCitesUs:null (UNKNOWN) rather than "not cited". To-dos are prioritised by OPPORTUNITY (impressions x the CTR gap between where you rank and a realistic target, damped by the verdict), not by raw impressions, and cannibalisation is counted across the whole query cluster. Set scrapeCompetitors:true to also pull the top competitors as content, routed by host: YouTube/video → Supadata transcript (SUPADATA_API_KEY), Reddit → its .json, other pages → Firecrawl (FIRECRAWL_API_KEY) with a free HTTP fallback; competitorLimit (default 5) caps how many are fetched and everything above the cap is listed as skipped. Cloudflare-challenge sites (e.g. PCMag) still can\'t be fetched from a server and come back as a per-URL error — the SERP still tells you they rank; if one matters, ask the user to paste its copy or supply a text file and diff that in. Transcribe the videos (usually what wins these SERPs) and use the reachable pages. Then write findings back with save_recon_todo; track with recon_todos. **Async job:** returns a jobId immediately - poll check_sync_status; the finished job carries the per-page verdicts, to-dos and a summary. (Each page is persisted to the ledger as it completes, so recon_todos shows results even mid-run.)',
901
+ inputSchema: {
902
+ siteUrl: z.string(),
903
+ limit: z.number().int().min(1).max(50).optional().describe('Pages per batch (default 5; async so large batches are fine)'),
904
+ minImpressions: z.number().int().min(1).optional(),
905
+ location: z.union([z.string(), z.number()]).optional(),
906
+ urls: z.array(z.string()).optional().describe('Analyse these exact pages instead of auto-selecting'),
907
+ scrapeCompetitors: z.boolean().optional().describe('Also fetch the top competitors (video→transcript, pages→markdown/HTML)'),
908
+ competitorLimit: z.number().int().min(1).max(12).optional().describe('How many competitors to fetch per page when scrapeCompetitors is on (default 5). Whatever is not fetched is listed as skipped, never dropped silently.'),
909
+ crawlAs: z.enum(['browser', 'googlebot']).optional().describe('UA for the free HTTP fetch: browser (default, mimics a visit from Google — gets Reddit + mid-tier) or googlebot'),
910
+ },
911
+ }, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, competitorLimit, crawlAs }) => {
912
+ const client = requireDfs(dfs); // fail fast if no DataForSEO creds, before starting the job
913
+ const count = urls?.length ?? limit ?? 5;
914
+ // Async job: N live page-fetches + N serialised SERP calls exceed the ~60s MCP ceiling
915
+ // past a handful of pages, so return a jobId and poll (like refresh_property).
916
+ const jobId = jobs.start('recon', async (update, signal) => {
917
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
918
+ try {
919
+ const fresh = gscFreshness(db.db);
920
+ if (!fresh.effectiveMax)
921
+ return { siteUrl, empty: true, note: 'No synced GSC data — run refresh_property first.' };
922
+ const today = new Date().toISOString().slice(0, 10);
923
+ const ownDomain = dfsHost(siteUrl);
924
+ const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
925
+ const SERP_DEPTH = 20;
926
+ // Every GSC number below is this window and only this window. Undeclared, a 28-day figure
927
+ // read against an all-time baseline looks exactly like a decline.
928
+ const windowStart = db.db.prepare(`SELECT date(?, '-27 days') d`).get(fresh.effectiveMax);
929
+ const gscWindow = { start: windowStart.d, end: fresh.effectiveMax, days: 28, metric: 'Search Console, last 28 days of synced data (not all-time)' };
930
+ let targets;
931
+ if (urls?.length) {
932
+ targets = [];
933
+ for (const u of urls) {
934
+ const key = urlKey(u, { hostForm });
935
+ const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
936
+ FROM search_analytics WHERE page_key=? AND query IS NOT NULL AND date > date(?, '-28 days') AND date <= ? GROUP BY query ORDER BY imp DESC LIMIT 1`)
937
+ .get(key, fresh.effectiveMax, fresh.effectiveMax);
938
+ if (row?.query)
939
+ targets.push({ urlKey: key, query: row.query, impressions: row.imp, clicks: row.clk, position: Math.round(row.pos * 10) / 10, priorPosition: null, slipped: false, competingUrls: 0 });
940
+ }
941
+ }
942
+ else {
943
+ targets = selectReconTargets(db.db, { ...(limit != null ? { limit } : {}), ...(minImpressions != null ? { minImpressions } : {}), maxDate: fresh.effectiveMax });
944
+ }
945
+ if (!targets.length)
946
+ return { siteUrl, targets: [], note: `No recon targets (declining/striking pages, position 3-15, ${minImpressions ?? 300}+ impressions). Lower minImpressions or pass urls.` };
947
+ const results = [];
948
+ let cost = 0;
949
+ for (let i = 0; i < targets.length; i++) {
950
+ if (signal.aborted)
951
+ break;
952
+ const t = targets[i];
953
+ update({ phase: 'recon', done: i, total: targets.length, current: t.urlKey });
954
+ let own = null;
955
+ try {
956
+ own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
957
+ }
958
+ catch { /* page unreachable — classify on SERP alone */ }
959
+ const serpResp = await client.serpOrganic(t.query, location, 'en', SERP_DEPTH, { loadAsyncAiOverview: true });
960
+ cost += serpResp.cost;
961
+ const serp = parseSerpForRecon(serpResp, ownDomain, SERP_DEPTH);
962
+ // The crawl row overrides the SERP/GSC read: GSC keeps reporting impressions for URLs
963
+ // that have since been canonicalised away or set noindex.
964
+ const state = pageState(db.db, t.urlKey);
965
+ const serpVerdict = reconVerdict(serp);
966
+ const verdict = crawlRealityOverride(serpVerdict, state) ?? serpVerdict;
967
+ const cluster = clusterCannibalisation(db.db, t.query, fresh.effectiveMax);
968
+ const basis = opportunityBasis(t.impressions, serp.ourOrganicRank, t.position, verdict.verdict);
969
+ const baseline = { organicRank: serp.ourOrganicRank, serpDepth: SERP_DEPTH, aioCitesUs: serp.aioCitesUs, gscPosition: t.position, gscImpressions: t.impressions, window: gscWindow, at: today };
970
+ persistReconPage(db.db, {
971
+ urlKey: t.urlKey, query: t.query, verdict: verdict.verdict, verdictNote: verdict.note,
972
+ organicRank: serp.ourOrganicRank, gscPosition: t.position, gscImpressions: t.impressions,
973
+ aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, videoPresent: serp.videoPresent,
974
+ hasProductSchema: own?.hasProductOrReview ?? false, schemaTypes: own?.jsonLdTypes ?? [],
975
+ competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
976
+ });
977
+ const todos = own ? deterministicTodos(t, own, serp, verdict, today, { state, cluster }) : [];
978
+ const inserted = todos.length ? insertTodos(db.db, t.urlKey, t.query, todos, baseline, 'auto') : 0;
979
+ let competitorContent = undefined;
980
+ let competitorFetch = undefined;
981
+ if (scrapeCompetitors && (firecrawl || supadata)) {
982
+ // Build a routed candidate set: organic-above + AI-Overview references + one ranking
983
+ // video, deduped, our own domain removed. Each URL routes by HOST (YouTube→supadata,
984
+ // Reddit→.json, else→firecrawl). What we don't fetch is REPORTED, not silently dropped
985
+ // — an undeclared cap of 3 is what made this look like it returned nothing.
986
+ const seen = new Set();
987
+ const candidates = [];
988
+ const add = (url, from) => {
989
+ if (url && !seen.has(url) && dfsHost(url) !== ownDomain) {
990
+ seen.add(url);
991
+ candidates.push({ url, from, route: routeFor(url) });
992
+ }
993
+ };
994
+ serp.organicAbove.forEach(o => add(o.url, `organic #${o.rank}`));
995
+ serp.aioReferences.slice(0, 6).forEach(r => add(r.url, 'aio-reference'));
996
+ serp.videoItems.slice(0, 2).forEach(v => add(v.url, 'video-pack'));
997
+ const lim = competitorLimit ?? 5;
998
+ // Organic-above first (the editorial pages actually worth diffing), but keep one slot
999
+ // for a ranking video — video is usually what wins these SERPs.
1000
+ const video = candidates.find(c => c.route === 'video');
1001
+ const rest = candidates.filter(c => c !== video);
1002
+ const reserve = video && lim >= 3 ? 1 : 0;
1003
+ const chosen = [...rest.slice(0, lim - reserve), ...(reserve && video ? [video] : [])];
1004
+ const skipped = candidates.filter(c => !chosen.includes(c));
1005
+ competitorContent = [];
1006
+ for (const c of chosen) {
1007
+ competitorContent.push(await fetchCompetitorContent(c.url, { firecrawl, supadata }, { maxChars: 4000, ua: crawlAs }));
1008
+ }
1009
+ competitorFetch = {
1010
+ limit: lim,
1011
+ identified: candidates.length,
1012
+ attempted: chosen.length,
1013
+ fetched: competitorContent.filter((c) => !c.error && c.content).length,
1014
+ failed: competitorContent.filter((c) => c.error).length,
1015
+ empty: competitorContent.filter((c) => !c.error && !c.content).length,
1016
+ routes: chosen.map(c => ({ url: c.url, from: c.from, route: c.route })),
1017
+ skipped: skipped.map(c => ({ url: c.url, from: c.from, route: c.route })),
1018
+ note: skipped.length ? `${skipped.length} identified competitor(s) not fetched (competitorLimit ${lim}) — raise competitorLimit or fetch them directly.` : undefined,
1019
+ };
1020
+ }
1021
+ results.push({
1022
+ urlKey: t.urlKey, query: t.query, window: gscWindow,
1023
+ impressions: t.impressions, gscPosition: t.position, priorPosition: t.priorPosition, slipped: t.slipped,
1024
+ organicRank: serp.ourOrganicRank, serpDepth: SERP_DEPTH,
1025
+ organicRankNote: serp.ourOrganicRank == null ? `absent from the top ${SERP_DEPTH} organic results — position beyond ${SERP_DEPTH} was not measured` : undefined,
1026
+ aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, aioResolved: serp.aioResolved, aioAsync: serp.aioAsync,
1027
+ aioNote: serp.aioCitesUs == null && serp.aioPresent ? 'AI Overview present but citations unresolved — citation status UNKNOWN, not "not cited"' : undefined,
1028
+ videoPresent: serp.videoPresent,
1029
+ verdict: verdict.verdict, verdictNote: verdict.note,
1030
+ ...(verdict.verdict === 'page-cannot-rank' ? { serpVerdict: serpVerdict.verdict, serpVerdictNote: serpVerdict.note } : {}),
1031
+ pageState: state, priorityBasis: basis, cannibalisationCluster: cluster,
1032
+ ownHeadings: own?.headings.map(h => h.heading) ?? null, schemaTypes: own?.jsonLdTypes ?? null, hasProductSchema: own?.hasProductOrReview ?? null, dateModified: own?.dateModified ?? null,
1033
+ todos, todosInserted: inserted,
1034
+ competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
1035
+ ...(competitorContent ? { competitorContent, competitorFetch } : {}),
1036
+ });
1037
+ }
1038
+ update({ phase: 'done', done: results.length, total: targets.length });
1039
+ const md = `# Content recon — ${siteUrl}\n\n${results.length} page(s), ${results.reduce((s, r) => s + r.todosInserted, 0)} to-dos saved. DataForSEO SERP cost $${cost.toFixed(4)}.\n` +
1040
+ `All Search Console figures below are **${gscWindow.start} to ${gscWindow.end}** (28 days) — not all-time; SERP ranks are live at depth ${SERP_DEPTH} organic.\n\n` +
1041
+ results.map(r => {
1042
+ const aio = !r.aioPresent ? 'no AIO'
1043
+ : r.aioCitesUs == null ? 'AIO: citation UNKNOWN (unresolved)'
1044
+ : r.aioCitesUs ? 'AIO: cited' : `AIO: NOT cited (${r.competitors.aioReferences.length} refs checked)`;
1045
+ const flags = [aio, r.videoPresent ? 'video pack' : null].filter(Boolean).join(' · ');
1046
+ const rank = r.organicRank != null ? `organic #${r.organicRank}` : `not in the top ${SERP_DEPTH} organic`;
1047
+ const cann = r.cannibalisationCluster?.cannibalising ? `\nCluster overlap: ${r.cannibalisationCluster.urls.length} of your URLs across "${r.cannibalisationCluster.core.join(' ')}", colliding on ${r.cannibalisationCluster.collidingQueries} of ${r.cannibalisationCluster.clusterQueries} queries.` : '';
1048
+ const cf = r.competitorFetch
1049
+ ? `\nCompetitors: ${r.competitorFetch.fetched}/${r.competitorFetch.attempted} fetched of ${r.competitorFetch.identified} identified (limit ${r.competitorFetch.limit})${r.competitorFetch.skipped.length ? `; skipped ${r.competitorFetch.skipped.map((s) => s.url).join(', ')}` : ''}.`
1050
+ : '';
1051
+ return `## ${r.urlKey}\n"${r.query}" — ${rank} (GSC avg ${r.gscPosition}${r.slipped ? `, slipped from ${r.priorPosition}` : ''}), ${r.impressions} impr. ${flags}\n` +
1052
+ `**${r.verdict}** — ${r.verdictNote}\nOpportunity: ~${r.priorityBasis.opportunityClicks} clicks/28d if it moved from ${r.priorityBasis.position} (${r.priorityBasis.positionSource}) to ${r.priorityBasis.targetPosition}.${cann}${cf}\n` +
1053
+ (r.todos.length ? '\nTo do:\n' + r.todos.map((t) => `- [${t.type}] ${t.action}`).join('\n') : '');
1054
+ }).join('\n\n') +
1055
+ (() => {
1056
+ const errs = results.flatMap(r => (r.competitorContent ?? []).filter((c) => c.error));
1057
+ if (!errs.length)
1058
+ return '';
1059
+ const needKey = errs.filter(c => /not set/i.test(c.error));
1060
+ const blocked = errs.filter(c => !/not set/i.test(c.error));
1061
+ let note = '';
1062
+ if (blocked.length)
1063
+ note += `\n\n${blocked.length} competitor(s) couldn't be fetched (bot-protection like Cloudflare, a block, or a fetch error): ${blocked.slice(0, 5).map(c => `${c.url} [${c.via}: ${c.error}]`).join('; ')}. If one matters, paste its copy here or drop a text file and I'll diff it in.`;
1064
+ if (needKey.length)
1065
+ note += `\n\n${needKey.length} competitor(s) skipped for a missing API key: ${[...new Set(needKey.map(c => c.error))].join('; ')}.`;
1066
+ return note;
1067
+ })() +
1068
+ `\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
1069
+ // The poll (check_sync_status) injects this result into the model's context, which the
1070
+ // host caps (~25k tokens). The full per-page detail (heading lists, competitor URL arrays,
1071
+ // full to-do objects, cannibalisation clusters) is already persisted to the ledger and read
1072
+ // back via recon_todos, so the job result returns the readable `summary` plus a SLIM per-page
1073
+ // view — verdict, ranks, opportunity — and keeps only competitorContent (the scraped research
1074
+ // payload, which lives nowhere else). This keeps a 10-page poll well under the cap.
1075
+ const slimTargets = results.map(r => ({
1076
+ urlKey: r.urlKey, query: r.query, window: r.window,
1077
+ impressions: r.impressions, gscPosition: r.gscPosition, priorPosition: r.priorPosition, slipped: r.slipped,
1078
+ organicRank: r.organicRank, serpDepth: r.serpDepth, organicRankNote: r.organicRankNote,
1079
+ aioPresent: r.aioPresent, aioCitesUs: r.aioCitesUs, aioResolved: r.aioResolved, aioNote: r.aioNote, videoPresent: r.videoPresent,
1080
+ verdict: r.verdict, verdictNote: r.verdictNote,
1081
+ opportunityClicks: r.priorityBasis?.opportunityClicks, positionSource: r.priorityBasis?.positionSource, targetPosition: r.priorityBasis?.targetPosition,
1082
+ hasProductSchema: r.hasProductSchema, dateModified: r.dateModified, todosInserted: r.todosInserted,
1083
+ ...(r.competitorContent ? { competitorContent: r.competitorContent, competitorFetch: r.competitorFetch } : {}),
1084
+ }));
1085
+ return { siteUrl, cost, window: gscWindow, serpDepth: SERP_DEPTH, targets: slimTargets, summary: md };
1086
+ }
1087
+ finally {
1088
+ db.close();
1089
+ }
1090
+ });
1091
+ return {
1092
+ content: [{ type: 'text', text: `Content recon started for ${count} page(s) — job ${jobId}. Each page is a live page-fetch + a serialised SERP call (a few seconds each), so poll check_sync_status with jobId "${jobId}"; the finished job's result carries the per-page verdicts, to-dos and a summary.` }],
1093
+ structuredContent: { jobId, status: 'running', siteUrl },
1094
+ };
1095
+ });
1096
+ // save_recon_todo — the research session writes its content-gap / originality findings back
1097
+ // into the ledger against a page (source: 'research').
1098
+ server.registerTool('save_recon_todo', {
1099
+ title: 'Save content-recon to-dos (research writeback)',
1100
+ description: 'Write content-recon findings back into the trackable ledger for a page — the gaps and originality the research session found by diffing competitors (firecrawl) and videos (supadata) against our content. Each to-do is an action with a type (content-gap / originality / schema / freshness / format), rationale and evidence. Snapshots the page baseline so the fix\'s effect on rank/AIO-citation is measurable later. Run recon_targets first (it classifies the page and seeds the deterministic to-dos); this adds the judgement ones. Track everything with recon_todos.',
1101
+ inputSchema: {
1102
+ siteUrl: z.string(),
1103
+ urlKey: z.string().describe('The page (any URL form — normalised to its key)'),
1104
+ todos: z.array(z.object({
1105
+ action: z.string(),
1106
+ type: z.string().optional(),
1107
+ rationale: z.string().optional(),
1108
+ evidence: z.record(z.any()).optional(),
1109
+ priority: z.number().optional(),
1110
+ })).min(1),
1111
+ },
1112
+ }, async ({ siteUrl, urlKey: rawUrl, todos }) => {
1113
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1114
+ try {
1115
+ const key = urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' });
1116
+ const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position, gsc_impressions, verdict FROM recon_page WHERE url_key=?`).get(key);
1117
+ // aio_cites_us is NULL when the AI Overview never resolved — keep that as unknown rather
1118
+ // than coercing it to false, or the outcome diff reads "no → yes" off a missing fetch.
1119
+ const baseline = page ? { organicRank: page.organic_rank, aioCitesUs: page.aio_cites_us == null ? null : !!page.aio_cites_us, gscPosition: page.gsc_position, at: new Date().toISOString().slice(0, 10) } : {};
1120
+ // Default priority to the PAGE's opportunity so research to-dos sort on the same scale as
1121
+ // the deterministic ones (a default of 0 buried every judgement finding at the bottom).
1122
+ const basis = page ? opportunityBasis(page.gsc_impressions ?? 0, page.organic_rank, page.gsc_position ?? 20, page.verdict) : null;
1123
+ const fallbackPriority = basis ? Math.max(1, Math.round(basis.opportunityClicks * basis.verdictFactor)) : 0;
1124
+ const drafts = todos.map(t => ({ action: t.action, type: t.type ?? 'content-gap', rationale: t.rationale ?? '', evidence: t.evidence ?? {}, priority: t.priority ?? fallbackPriority }));
1125
+ const n = insertTodos(db.db, key, page?.query ?? '', drafts, baseline, 'research');
1126
+ return { content: [{ type: 'text', text: `Saved ${n} recon to-do(s) for ${key}${n < drafts.length ? ` (${drafts.length - n} already open)` : ''}. Track with recon_todos.` }], structuredContent: { urlKey: key, inserted: n } };
1127
+ }
1128
+ finally {
1129
+ db.close();
1130
+ }
1131
+ });
1132
+ // recon_todos — list, track and annotate the ledger. No id → list (optionally filtered);
1133
+ // id → update status and/or append a dated annotation, and optionally re-measure the outcome.
1134
+ server.registerTool('recon_todos', {
1135
+ title: 'List, track and annotate content-recon to-dos',
1136
+ description: 'The content-recon to-do board. With no id: list to-dos (optionally filter by page or status), grouped by page with each page\'s verdict — the pick-a-page-to-work-on surface, and the hand-off to content-machine. With id: update one to-do — set status (open → researching → drafted → shipped → dismissed) and/or append a dated annotation note (your own observations, the history). On status:shipped with remeasure:true it re-fetches the SERP and records the outcome, so you can see whether the fix moved you from AIO-uncited to cited, or up the organic ranks.',
1137
+ inputSchema: {
1138
+ siteUrl: z.string(),
1139
+ urlKey: z.string().optional().describe('Filter the list to one page'),
1140
+ status: z.enum(['open', 'researching', 'drafted', 'shipped', 'dismissed']).optional().describe('Filter the list, or the new status when id is set'),
1141
+ id: z.number().int().optional().describe('Update this to-do'),
1142
+ note: z.string().optional().describe('Append a dated annotation to this to-do'),
1143
+ remeasure: z.boolean().optional().describe('On status:shipped, re-fetch the SERP and record the outcome (paid)'),
1144
+ location: z.union([z.string(), z.number()]).optional(),
1145
+ },
1146
+ }, async ({ siteUrl, urlKey: rawUrl, status, id, note, remeasure, location }) => {
1147
+ const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
1148
+ try {
1149
+ if (id != null) {
1150
+ const row = db.db.prepare(`SELECT * FROM recon_todo WHERE id=?`).get(id);
1151
+ if (!row)
1152
+ throw new Error(`No recon to-do #${id}.`);
1153
+ let notes = row.notes ?? '';
1154
+ if (note)
1155
+ notes = (notes ? notes + '\n' : '') + `[${new Date().toISOString().slice(0, 10)}] ${note}`;
1156
+ let outcome = row.outcome;
1157
+ let outcomeNote = '';
1158
+ if (status === 'shipped' && remeasure && dfs) {
1159
+ // Same depth + async-AIO handling as recon_targets, so before/after are comparable.
1160
+ const serpResp = await dfs.serpOrganic(row.query, location, 'en', 20, { loadAsyncAiOverview: true });
1161
+ const serp = parseSerpForRecon(serpResp, dfsHost(siteUrl), 20);
1162
+ const cited = (v) => (v == null ? 'unknown' : v ? 'yes' : 'no');
1163
+ outcome = JSON.stringify({ organicRank: serp.ourOrganicRank, serpDepth: 20, aioCitesUs: serp.aioCitesUs, aioResolved: serp.aioResolved, at: new Date().toISOString().slice(0, 10) });
1164
+ const base = row.baseline ? JSON.parse(row.baseline) : {};
1165
+ const rankStr = (r) => (r == null ? 'not in top 20' : `#${r}`);
1166
+ outcomeNote = ` Outcome: organic ${rankStr(base.organicRank)}→${rankStr(serp.ourOrganicRank)}, AIO cited ${cited(base.aioCitesUs)}→${cited(serp.aioCitesUs)}.`;
1167
+ }
1168
+ db.db.prepare(`UPDATE recon_todo SET status=COALESCE(?,status), notes=?, outcome=COALESCE(?,outcome), updated_at=datetime('now') WHERE id=?`)
1169
+ .run(status ?? null, notes, outcome ?? null, id);
1170
+ return { content: [{ type: 'text', text: `Updated to-do #${id}${status ? ` → ${status}` : ''}${note ? ' (note added)' : ''}.${outcomeNote}` }], structuredContent: { id, status: status ?? row.status, outcome: outcome ? JSON.parse(outcome) : null } };
1171
+ }
1172
+ const key = rawUrl ? urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' }) : null;
1173
+ const where = [];
1174
+ const args = [];
1175
+ if (key) {
1176
+ where.push('url_key=?');
1177
+ args.push(key);
1178
+ }
1179
+ if (status) {
1180
+ where.push('status=?');
1181
+ args.push(status);
1182
+ }
1183
+ const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
1184
+ FROM recon_todo ${where.length ? 'WHERE ' + where.join(' AND ') : ''} ORDER BY url_key, priority DESC`).all(...args);
1185
+ if (!rows.length)
1186
+ return { content: [{ type: 'text', text: `No recon to-dos${key ? ` for ${key}` : ''}${status ? ` with status ${status}` : ''}. Run recon_targets to generate some.` }], structuredContent: { todos: [] } };
1187
+ const byPage = new Map();
1188
+ for (const r of rows)
1189
+ (byPage.get(r.url_key) ?? byPage.set(r.url_key, []).get(r.url_key)).push(r);
1190
+ const verdictStmt = db.db.prepare(`SELECT verdict, verdict_note FROM recon_page WHERE url_key=?`);
1191
+ const md = `# Content-recon to-dos — ${rows.length} item(s)${status ? `, status ${status}` : ''}\n\n` +
1192
+ [...byPage.entries()].map(([url, items]) => {
1193
+ const v = verdictStmt.get(url);
1194
+ return `## ${url}${v ? `\n_${v.verdict}_ — ${v.verdict_note}` : ''}\n\n| # | Status | Type | Action |\n|---|---|---|---|\n` +
1195
+ items.map(i => `| ${i.id} | ${i.status} | ${i.type ?? ''} | ${String(i.action).replace(/\|/g, '\\|')} |`).join('\n') +
1196
+ (items.some(i => i.notes) ? '\n\nNotes:\n' + items.filter(i => i.notes).map(i => `- #${i.id}: ${String(i.notes).replace(/\n/g, ' / ')}`).join('\n') : '');
1197
+ }).join('\n\n') +
1198
+ browserLink(siteUrl);
1199
+ return { content: [{ type: 'text', text: md }], structuredContent: { todos: rows } };
1200
+ }
1201
+ finally {
1202
+ db.close();
1203
+ }
1204
+ });
877
1205
  // keyword_list — demand-first clustering ("list mode"): a keyword list becomes topics
878
1206
  // with own/weak/absent verdicts, clustered by the URL Google already answers them with.
879
1207
  server.registerTool('keyword_list', {