@houtini/seo-audit-console 0.4.3 → 0.4.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/audit/checks.d.ts.map +1 -1
- package/dist/audit/checks.js +5 -4
- package/dist/audit/checks.js.map +1 -1
- package/dist/audit/drift.d.ts +7 -0
- package/dist/audit/drift.d.ts.map +1 -1
- package/dist/audit/drift.js +31 -3
- package/dist/audit/drift.js.map +1 -1
- package/dist/audit/engine.d.ts +9 -0
- package/dist/audit/engine.d.ts.map +1 -1
- package/dist/audit/engine.js +24 -4
- package/dist/audit/engine.js.map +1 -1
- package/dist/audit/recon.d.ts +70 -2
- package/dist/audit/recon.d.ts.map +1 -1
- package/dist/audit/recon.js +129 -6
- package/dist/audit/recon.js.map +1 -1
- package/dist/core/Crawler.d.ts.map +1 -1
- package/dist/core/Crawler.js +16 -0
- package/dist/core/Crawler.js.map +1 -1
- package/dist/core/DataForSeoClient.d.ts +9 -2
- package/dist/core/DataForSeoClient.d.ts.map +1 -1
- package/dist/core/DataForSeoClient.js +11 -3
- package/dist/core/DataForSeoClient.js.map +1 -1
- package/dist/core/FirecrawlClient.d.ts +6 -2
- package/dist/core/FirecrawlClient.d.ts.map +1 -1
- package/dist/core/FirecrawlClient.js +20 -3
- package/dist/core/FirecrawlClient.js.map +1 -1
- package/dist/core/SupadataClient.d.ts +3 -0
- package/dist/core/SupadataClient.d.ts.map +1 -1
- package/dist/core/SupadataClient.js +10 -4
- package/dist/core/SupadataClient.js.map +1 -1
- package/dist/core/dataStorage.d.ts +2 -0
- package/dist/core/dataStorage.d.ts.map +1 -1
- package/dist/core/dataStorage.js +8 -1
- package/dist/core/dataStorage.js.map +1 -1
- package/dist/core/passageScore.d.ts +40 -9
- package/dist/core/passageScore.d.ts.map +1 -1
- package/dist/core/passageScore.js +29 -3
- package/dist/core/passageScore.js.map +1 -1
- package/dist/core/paths.d.ts +3 -1
- package/dist/core/paths.d.ts.map +1 -1
- package/dist/core/paths.js +0 -0
- package/dist/core/paths.js.map +1 -1
- package/dist/core/reconResearch.d.ts +8 -0
- package/dist/core/reconResearch.d.ts.map +1 -1
- package/dist/core/reconResearch.js +35 -14
- package/dist/core/reconResearch.js.map +1 -1
- package/dist/core/serpRecon.d.ts +9 -3
- package/dist/core/serpRecon.d.ts.map +1 -1
- package/dist/core/serpRecon.js +46 -18
- package/dist/core/serpRecon.js.map +1 -1
- package/dist/core/webServer.d.ts.map +1 -1
- package/dist/core/webServer.js +3 -1
- package/dist/core/webServer.js.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +213 -115
- package/dist/server.js.map +1 -1
- package/package.json +1 -1
- package/server.json +2 -2
package/dist/server.js
CHANGED
|
@@ -13,10 +13,10 @@ import { computeSerpFootprint, persistSerpFootprint } from './core/serpFootprint
|
|
|
13
13
|
import { computeMarketSizing, persistMarketSizing } from './core/marketSizing.js';
|
|
14
14
|
import { FirecrawlClient } from './core/FirecrawlClient.js';
|
|
15
15
|
import { SupadataClient } from './core/SupadataClient.js';
|
|
16
|
-
import { fetchCompetitorContent } from './core/reconResearch.js';
|
|
16
|
+
import { fetchCompetitorContent, routeFor } from './core/reconResearch.js';
|
|
17
17
|
import { fetchOwnPage } from './core/reconFetch.js';
|
|
18
18
|
import { parseSerpForRecon, reconVerdict } from './core/serpRecon.js';
|
|
19
|
-
import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos } from './audit/recon.js';
|
|
19
|
+
import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos, pageState, crawlRealityOverride, clusterCannibalisation, opportunityBasis } from './audit/recon.js';
|
|
20
20
|
/** A clickable browser-dashboard link appended to tool outputs — the user should always
|
|
21
21
|
* know the full interactive report is one click away (or one serve_dashboard call away). */
|
|
22
22
|
function browserLink(siteUrl) {
|
|
@@ -453,14 +453,21 @@ export function createServer() {
|
|
|
453
453
|
});
|
|
454
454
|
server.registerTool('score_passages', {
|
|
455
455
|
title: 'Score passage relevance (AI-search readiness)',
|
|
456
|
-
description: 'Run a local cross-encoder reranker (ms-marco-MiniLM-L-6-v2, downloaded once to a cache, no Python) over each ranking page\'s heading chunks against its top Search Console query, and persist the single best-passage relevance score. It SCORES relevance (a classifier — it cannot hallucinate). Powers the `weak-passage-answer` check — the "RAG snippetability" test: does any passage confidently answer the query, the way AI/passage search re-ranks? On-demand + ML-heavy (~0.6s/page), bounded to pages that already rank. `limit` caps pages (highest-impression first); `minImpressions` sets the floor (default 50). First run downloads ~25MB. Re-run after a re-crawl.',
|
|
457
|
-
inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(2000).optional(), minImpressions: z.number().int().min(0).optional() },
|
|
458
|
-
}, async ({ siteUrl, limit, minImpressions }) => {
|
|
456
|
+
description: 'Run a local cross-encoder reranker (ms-marco-MiniLM-L-6-v2, downloaded once to a cache, no Python) over each ranking page\'s heading chunks against its top Search Console query, and persist the single best-passage relevance score. It SCORES relevance (a classifier — it cannot hallucinate). Powers the `weak-passage-answer` check — the "RAG snippetability" test: does any passage confidently answer the query, the way AI/passage search re-ranks? Returns the score for EVERY page it scored (not just the failures) with the flag threshold, the band convention and the run\'s own distribution, so a passing page\'s score is readable without a second query_data hop. On-demand + ML-heavy (~0.6s/page), bounded to pages that already rank. `limit` caps pages (highest-impression first); `minImpressions` sets the floor (default 50). First run downloads ~25MB. Re-run after a re-crawl.',
|
|
457
|
+
inputSchema: { siteUrl: z.string(), limit: z.number().int().min(1).max(2000).optional(), minImpressions: z.number().int().min(0).optional(), urls: z.array(z.string()).optional().describe('Only report these pages (they are still scored site-wide; this filters the returned list)') },
|
|
458
|
+
}, async ({ siteUrl, limit, minImpressions, urls }) => {
|
|
459
459
|
const r = await scoreSitePassages(dataDir(), siteUrl, { limit, minImpressions });
|
|
460
|
-
const
|
|
460
|
+
const keys = urls?.length ? new Set(urls.map(u => urlKey(u, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' }))) : null;
|
|
461
|
+
const shown = keys ? r.pages.filter(p => keys.has(p.url)) : r.pages;
|
|
462
|
+
const rows = shown.slice(0, 25).map(p => ` ${String(p.score).padStart(7)} ${p.band.padEnd(8)} ${p.url.replace(/^https?:\/\/[^/]+/, '')} · "${p.query}" (${p.impressions} impr)`).join('\n');
|
|
463
|
+
const d = r.distribution;
|
|
464
|
+
const text = `Scored ${r.scored} ranking pages; ${r.flagged} below the weak threshold of ${r.threshold} — run_audit surfaces those as weak-passage-answer.\n\n` +
|
|
465
|
+
`Scale: ${r.scale}\n` +
|
|
466
|
+
(d ? `This run: min ${d.min} · p25 ${d.p25} · median ${d.median} · p75 ${d.p75} · max ${d.max}\n` : '') +
|
|
467
|
+
(rows ? `\nScores (${shown.length === r.pagesTotal ? `all ${r.pagesTotal}` : `${shown.length} of ${r.pagesTotal}`}, showing up to 25):\n${rows}` : '');
|
|
461
468
|
return {
|
|
462
|
-
content: [{ type: 'text', text
|
|
463
|
-
structuredContent: r,
|
|
469
|
+
content: [{ type: 'text', text }],
|
|
470
|
+
structuredContent: { ...r, ...(keys ? { pages: shown, pagesReturned: shown.length } : {}) },
|
|
464
471
|
};
|
|
465
472
|
});
|
|
466
473
|
server.registerTool('draft_content', {
|
|
@@ -890,119 +897,201 @@ export function createServer() {
|
|
|
890
897
|
// ── Content recon (recon_targets) — the data-intensive "why are we losing, what to do" mission ──
|
|
891
898
|
server.registerTool('recon_targets', {
|
|
892
899
|
title: 'Content recon: why a page is losing, and what to do about it',
|
|
893
|
-
description: '[Paid: DataForSEO SERP per page (~$0.004 each), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, or
|
|
900
|
+
description: '[Paid: DataForSEO SERP per page (~$0.004 each, plus a small refundable surcharge for loading async AI Overviews), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced, depth 20 organic), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, competitive-gap, or page-cannot-rank (the crawl says the URL is noindex/canonicalised away, so its GSC history is legacy and the SERP read belongs to another page). Honesty rules baked in: every GSC figure carries its 28-day window; organicRank:null means "absent from the top 20 ORGANIC results", never a position; and an AI Overview whose citations could not be resolved reports aioCitesUs:null (UNKNOWN) rather than "not cited". To-dos are prioritised by OPPORTUNITY (impressions x the CTR gap between where you rank and a realistic target, damped by the verdict), not by raw impressions, and cannibalisation is counted across the whole query cluster. Set scrapeCompetitors:true to also pull the top competitors as content, routed by host: YouTube/video → Supadata transcript (SUPADATA_API_KEY), Reddit → its .json, other pages → Firecrawl (FIRECRAWL_API_KEY) with a free HTTP fallback; competitorLimit (default 5) caps how many are fetched and everything above the cap is listed as skipped. Cloudflare-challenge sites (e.g. PCMag) still can\'t be fetched from a server and come back as a per-URL error — the SERP still tells you they rank; if one matters, ask the user to paste its copy or supply a text file and diff that in. Transcribe the videos (usually what wins these SERPs) and use the reachable pages. Then write findings back with save_recon_todo; track with recon_todos. **Async job:** returns a jobId immediately - poll check_sync_status; the finished job carries the per-page verdicts, to-dos and a summary. (Each page is persisted to the ledger as it completes, so recon_todos shows results even mid-run.)',
|
|
894
901
|
inputSchema: {
|
|
895
902
|
siteUrl: z.string(),
|
|
896
|
-
limit: z.number().int().min(1).max(
|
|
903
|
+
limit: z.number().int().min(1).max(50).optional().describe('Pages per batch (default 5; async so large batches are fine)'),
|
|
897
904
|
minImpressions: z.number().int().min(1).optional(),
|
|
898
905
|
location: z.union([z.string(), z.number()]).optional(),
|
|
899
906
|
urls: z.array(z.string()).optional().describe('Analyse these exact pages instead of auto-selecting'),
|
|
900
907
|
scrapeCompetitors: z.boolean().optional().describe('Also fetch the top competitors (video→transcript, pages→markdown/HTML)'),
|
|
908
|
+
competitorLimit: z.number().int().min(1).max(12).optional().describe('How many competitors to fetch per page when scrapeCompetitors is on (default 5). Whatever is not fetched is listed as skipped, never dropped silently.'),
|
|
901
909
|
crawlAs: z.enum(['browser', 'googlebot']).optional().describe('UA for the free HTTP fetch: browser (default, mimics a visit from Google — gets Reddit + mid-tier) or googlebot'),
|
|
902
910
|
},
|
|
903
|
-
}, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, crawlAs }) => {
|
|
904
|
-
const client = requireDfs(dfs);
|
|
905
|
-
const
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
911
|
+
}, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, competitorLimit, crawlAs }) => {
|
|
912
|
+
const client = requireDfs(dfs); // fail fast if no DataForSEO creds, before starting the job
|
|
913
|
+
const count = urls?.length ?? limit ?? 5;
|
|
914
|
+
// Async job: N live page-fetches + N serialised SERP calls exceed the ~60s MCP ceiling
|
|
915
|
+
// past a handful of pages, so return a jobId and poll (like refresh_property).
|
|
916
|
+
const jobId = jobs.start('recon', async (update, signal) => {
|
|
917
|
+
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
918
|
+
try {
|
|
919
|
+
const fresh = gscFreshness(db.db);
|
|
920
|
+
if (!fresh.effectiveMax)
|
|
921
|
+
return { siteUrl, empty: true, note: 'No synced GSC data — run refresh_property first.' };
|
|
922
|
+
const today = new Date().toISOString().slice(0, 10);
|
|
923
|
+
const ownDomain = dfsHost(siteUrl);
|
|
924
|
+
const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
|
|
925
|
+
const SERP_DEPTH = 20;
|
|
926
|
+
// Every GSC number below is this window and only this window. Undeclared, a 28-day figure
|
|
927
|
+
// read against an all-time baseline looks exactly like a decline.
|
|
928
|
+
const windowStart = db.db.prepare(`SELECT date(?, '-27 days') d`).get(fresh.effectiveMax);
|
|
929
|
+
const gscWindow = { start: windowStart.d, end: fresh.effectiveMax, days: 28, metric: 'Search Console, last 28 days of synced data (not all-time)' };
|
|
930
|
+
let targets;
|
|
931
|
+
if (urls?.length) {
|
|
932
|
+
targets = [];
|
|
933
|
+
for (const u of urls) {
|
|
934
|
+
const key = urlKey(u, { hostForm });
|
|
935
|
+
const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
|
|
936
|
+
FROM search_analytics WHERE page_key=? AND query IS NOT NULL AND date > date(?, '-28 days') AND date <= ? GROUP BY query ORDER BY imp DESC LIMIT 1`)
|
|
937
|
+
.get(key, fresh.effectiveMax, fresh.effectiveMax);
|
|
938
|
+
if (row?.query)
|
|
939
|
+
targets.push({ urlKey: key, query: row.query, impressions: row.imp, clicks: row.clk, position: Math.round(row.pos * 10) / 10, priorPosition: null, slipped: false, competingUrls: 0 });
|
|
940
|
+
}
|
|
923
941
|
}
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
targets = selectReconTargets(db.db, { ...(limit != null ? { limit } : {}), ...(minImpressions != null ? { minImpressions } : {}), maxDate: fresh.effectiveMax });
|
|
927
|
-
}
|
|
928
|
-
if (!targets.length)
|
|
929
|
-
return { content: [{ type: 'text', text: `No recon targets found (declining/striking-distance pages, position 3-15, ${minImpressions ?? 300}+ impressions). Try a lower minImpressions or pass explicit urls.` }], structuredContent: { siteUrl, targets: [] } };
|
|
930
|
-
const results = [];
|
|
931
|
-
let cost = 0;
|
|
932
|
-
for (const t of targets) {
|
|
933
|
-
let own = null;
|
|
934
|
-
try {
|
|
935
|
-
own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
|
|
942
|
+
else {
|
|
943
|
+
targets = selectReconTargets(db.db, { ...(limit != null ? { limit } : {}), ...(minImpressions != null ? { minImpressions } : {}), maxDate: fresh.effectiveMax });
|
|
936
944
|
}
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
//
|
|
955
|
-
//
|
|
956
|
-
|
|
957
|
-
const
|
|
958
|
-
const
|
|
959
|
-
const
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
945
|
+
if (!targets.length)
|
|
946
|
+
return { siteUrl, targets: [], note: `No recon targets (declining/striking pages, position 3-15, ${minImpressions ?? 300}+ impressions). Lower minImpressions or pass urls.` };
|
|
947
|
+
const results = [];
|
|
948
|
+
let cost = 0;
|
|
949
|
+
for (let i = 0; i < targets.length; i++) {
|
|
950
|
+
if (signal.aborted)
|
|
951
|
+
break;
|
|
952
|
+
const t = targets[i];
|
|
953
|
+
update({ phase: 'recon', done: i, total: targets.length, current: t.urlKey });
|
|
954
|
+
let own = null;
|
|
955
|
+
try {
|
|
956
|
+
own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
|
|
957
|
+
}
|
|
958
|
+
catch { /* page unreachable — classify on SERP alone */ }
|
|
959
|
+
const serpResp = await client.serpOrganic(t.query, location, 'en', SERP_DEPTH, { loadAsyncAiOverview: true });
|
|
960
|
+
cost += serpResp.cost;
|
|
961
|
+
const serp = parseSerpForRecon(serpResp, ownDomain, SERP_DEPTH);
|
|
962
|
+
// The crawl row overrides the SERP/GSC read: GSC keeps reporting impressions for URLs
|
|
963
|
+
// that have since been canonicalised away or set noindex.
|
|
964
|
+
const state = pageState(db.db, t.urlKey);
|
|
965
|
+
const serpVerdict = reconVerdict(serp);
|
|
966
|
+
const verdict = crawlRealityOverride(serpVerdict, state) ?? serpVerdict;
|
|
967
|
+
const cluster = clusterCannibalisation(db.db, t.query, fresh.effectiveMax);
|
|
968
|
+
const basis = opportunityBasis(t.impressions, serp.ourOrganicRank, t.position, verdict.verdict);
|
|
969
|
+
const baseline = { organicRank: serp.ourOrganicRank, serpDepth: SERP_DEPTH, aioCitesUs: serp.aioCitesUs, gscPosition: t.position, gscImpressions: t.impressions, window: gscWindow, at: today };
|
|
970
|
+
persistReconPage(db.db, {
|
|
971
|
+
urlKey: t.urlKey, query: t.query, verdict: verdict.verdict, verdictNote: verdict.note,
|
|
972
|
+
organicRank: serp.ourOrganicRank, gscPosition: t.position, gscImpressions: t.impressions,
|
|
973
|
+
aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, videoPresent: serp.videoPresent,
|
|
974
|
+
hasProductSchema: own?.hasProductOrReview ?? false, schemaTypes: own?.jsonLdTypes ?? [],
|
|
975
|
+
competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
|
|
976
|
+
});
|
|
977
|
+
const todos = own ? deterministicTodos(t, own, serp, verdict, today, { state, cluster }) : [];
|
|
978
|
+
const inserted = todos.length ? insertTodos(db.db, t.urlKey, t.query, todos, baseline, 'auto') : 0;
|
|
979
|
+
let competitorContent = undefined;
|
|
980
|
+
let competitorFetch = undefined;
|
|
981
|
+
if (scrapeCompetitors && (firecrawl || supadata)) {
|
|
982
|
+
// Build a routed candidate set: organic-above + AI-Overview references + one ranking
|
|
983
|
+
// video, deduped, our own domain removed. Each URL routes by HOST (YouTube→supadata,
|
|
984
|
+
// Reddit→.json, else→firecrawl). What we don't fetch is REPORTED, not silently dropped
|
|
985
|
+
// — an undeclared cap of 3 is what made this look like it returned nothing.
|
|
986
|
+
const seen = new Set();
|
|
987
|
+
const candidates = [];
|
|
988
|
+
const add = (url, from) => {
|
|
989
|
+
if (url && !seen.has(url) && dfsHost(url) !== ownDomain) {
|
|
990
|
+
seen.add(url);
|
|
991
|
+
candidates.push({ url, from, route: routeFor(url) });
|
|
992
|
+
}
|
|
993
|
+
};
|
|
994
|
+
serp.organicAbove.forEach(o => add(o.url, `organic #${o.rank}`));
|
|
995
|
+
serp.aioReferences.slice(0, 6).forEach(r => add(r.url, 'aio-reference'));
|
|
996
|
+
serp.videoItems.slice(0, 2).forEach(v => add(v.url, 'video-pack'));
|
|
997
|
+
const lim = competitorLimit ?? 5;
|
|
998
|
+
// Organic-above first (the editorial pages actually worth diffing), but keep one slot
|
|
999
|
+
// for a ranking video — video is usually what wins these SERPs.
|
|
1000
|
+
const video = candidates.find(c => c.route === 'video');
|
|
1001
|
+
const rest = candidates.filter(c => c !== video);
|
|
1002
|
+
const reserve = video && lim >= 3 ? 1 : 0;
|
|
1003
|
+
const chosen = [...rest.slice(0, lim - reserve), ...(reserve && video ? [video] : [])];
|
|
1004
|
+
const skipped = candidates.filter(c => !chosen.includes(c));
|
|
1005
|
+
competitorContent = [];
|
|
1006
|
+
for (const c of chosen) {
|
|
1007
|
+
competitorContent.push(await fetchCompetitorContent(c.url, { firecrawl, supadata }, { maxChars: 4000, ua: crawlAs }));
|
|
1008
|
+
}
|
|
1009
|
+
competitorFetch = {
|
|
1010
|
+
limit: lim,
|
|
1011
|
+
identified: candidates.length,
|
|
1012
|
+
attempted: chosen.length,
|
|
1013
|
+
fetched: competitorContent.filter((c) => !c.error && c.content).length,
|
|
1014
|
+
failed: competitorContent.filter((c) => c.error).length,
|
|
1015
|
+
empty: competitorContent.filter((c) => !c.error && !c.content).length,
|
|
1016
|
+
routes: chosen.map(c => ({ url: c.url, from: c.from, route: c.route })),
|
|
1017
|
+
skipped: skipped.map(c => ({ url: c.url, from: c.from, route: c.route })),
|
|
1018
|
+
note: skipped.length ? `${skipped.length} identified competitor(s) not fetched (competitorLimit ${lim}) — raise competitorLimit or fetch them directly.` : undefined,
|
|
1019
|
+
};
|
|
969
1020
|
}
|
|
1021
|
+
results.push({
|
|
1022
|
+
urlKey: t.urlKey, query: t.query, window: gscWindow,
|
|
1023
|
+
impressions: t.impressions, gscPosition: t.position, priorPosition: t.priorPosition, slipped: t.slipped,
|
|
1024
|
+
organicRank: serp.ourOrganicRank, serpDepth: SERP_DEPTH,
|
|
1025
|
+
organicRankNote: serp.ourOrganicRank == null ? `absent from the top ${SERP_DEPTH} organic results — position beyond ${SERP_DEPTH} was not measured` : undefined,
|
|
1026
|
+
aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, aioResolved: serp.aioResolved, aioAsync: serp.aioAsync,
|
|
1027
|
+
aioNote: serp.aioCitesUs == null && serp.aioPresent ? 'AI Overview present but citations unresolved — citation status UNKNOWN, not "not cited"' : undefined,
|
|
1028
|
+
videoPresent: serp.videoPresent,
|
|
1029
|
+
verdict: verdict.verdict, verdictNote: verdict.note,
|
|
1030
|
+
...(verdict.verdict === 'page-cannot-rank' ? { serpVerdict: serpVerdict.verdict, serpVerdictNote: serpVerdict.note } : {}),
|
|
1031
|
+
pageState: state, priorityBasis: basis, cannibalisationCluster: cluster,
|
|
1032
|
+
ownHeadings: own?.headings.map(h => h.heading) ?? null, schemaTypes: own?.jsonLdTypes ?? null, hasProductSchema: own?.hasProductOrReview ?? null, dateModified: own?.dateModified ?? null,
|
|
1033
|
+
todos, todosInserted: inserted,
|
|
1034
|
+
competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
|
|
1035
|
+
...(competitorContent ? { competitorContent, competitorFetch } : {}),
|
|
1036
|
+
});
|
|
970
1037
|
}
|
|
971
|
-
results.
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
1038
|
+
update({ phase: 'done', done: results.length, total: targets.length });
|
|
1039
|
+
const md = `# Content recon — ${siteUrl}\n\n${results.length} page(s), ${results.reduce((s, r) => s + r.todosInserted, 0)} to-dos saved. DataForSEO SERP cost $${cost.toFixed(4)}.\n` +
|
|
1040
|
+
`All Search Console figures below are **${gscWindow.start} to ${gscWindow.end}** (28 days) — not all-time; SERP ranks are live at depth ${SERP_DEPTH} organic.\n\n` +
|
|
1041
|
+
results.map(r => {
|
|
1042
|
+
const aio = !r.aioPresent ? 'no AIO'
|
|
1043
|
+
: r.aioCitesUs == null ? 'AIO: citation UNKNOWN (unresolved)'
|
|
1044
|
+
: r.aioCitesUs ? 'AIO: cited' : `AIO: NOT cited (${r.competitors.aioReferences.length} refs checked)`;
|
|
1045
|
+
const flags = [aio, r.videoPresent ? 'video pack' : null].filter(Boolean).join(' · ');
|
|
1046
|
+
const rank = r.organicRank != null ? `organic #${r.organicRank}` : `not in the top ${SERP_DEPTH} organic`;
|
|
1047
|
+
const cann = r.cannibalisationCluster?.cannibalising ? `\nCluster overlap: ${r.cannibalisationCluster.urls.length} of your URLs across "${r.cannibalisationCluster.core.join(' ')}", colliding on ${r.cannibalisationCluster.collidingQueries} of ${r.cannibalisationCluster.clusterQueries} queries.` : '';
|
|
1048
|
+
const cf = r.competitorFetch
|
|
1049
|
+
? `\nCompetitors: ${r.competitorFetch.fetched}/${r.competitorFetch.attempted} fetched of ${r.competitorFetch.identified} identified (limit ${r.competitorFetch.limit})${r.competitorFetch.skipped.length ? `; skipped ${r.competitorFetch.skipped.map((s) => s.url).join(', ')}` : ''}.`
|
|
1050
|
+
: '';
|
|
1051
|
+
return `## ${r.urlKey}\n"${r.query}" — ${rank} (GSC avg ${r.gscPosition}${r.slipped ? `, slipped from ${r.priorPosition}` : ''}), ${r.impressions} impr. ${flags}\n` +
|
|
1052
|
+
`**${r.verdict}** — ${r.verdictNote}\nOpportunity: ~${r.priorityBasis.opportunityClicks} clicks/28d if it moved from ${r.priorityBasis.position} (${r.priorityBasis.positionSource}) to ${r.priorityBasis.targetPosition}.${cann}${cf}\n` +
|
|
1053
|
+
(r.todos.length ? '\nTo do:\n' + r.todos.map((t) => `- [${t.type}] ${t.action}`).join('\n') : '');
|
|
1054
|
+
}).join('\n\n') +
|
|
1055
|
+
(() => {
|
|
1056
|
+
const errs = results.flatMap(r => (r.competitorContent ?? []).filter((c) => c.error));
|
|
1057
|
+
if (!errs.length)
|
|
1058
|
+
return '';
|
|
1059
|
+
const needKey = errs.filter(c => /not set/i.test(c.error));
|
|
1060
|
+
const blocked = errs.filter(c => !/not set/i.test(c.error));
|
|
1061
|
+
let note = '';
|
|
1062
|
+
if (blocked.length)
|
|
1063
|
+
note += `\n\n${blocked.length} competitor(s) couldn't be fetched (bot-protection like Cloudflare, a block, or a fetch error): ${blocked.slice(0, 5).map(c => `${c.url} [${c.via}: ${c.error}]`).join('; ')}. If one matters, paste its copy here or drop a text file and I'll diff it in.`;
|
|
1064
|
+
if (needKey.length)
|
|
1065
|
+
note += `\n\n${needKey.length} competitor(s) skipped for a missing API key: ${[...new Set(needKey.map(c => c.error))].join('; ')}.`;
|
|
1066
|
+
return note;
|
|
1067
|
+
})() +
|
|
1068
|
+
`\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
|
|
1069
|
+
// The poll (check_sync_status) injects this result into the model's context, which the
|
|
1070
|
+
// host caps (~25k tokens). The full per-page detail (heading lists, competitor URL arrays,
|
|
1071
|
+
// full to-do objects, cannibalisation clusters) is already persisted to the ledger and read
|
|
1072
|
+
// back via recon_todos, so the job result returns the readable `summary` plus a SLIM per-page
|
|
1073
|
+
// view — verdict, ranks, opportunity — and keeps only competitorContent (the scraped research
|
|
1074
|
+
// payload, which lives nowhere else). This keeps a 10-page poll well under the cap.
|
|
1075
|
+
const slimTargets = results.map(r => ({
|
|
1076
|
+
urlKey: r.urlKey, query: r.query, window: r.window,
|
|
1077
|
+
impressions: r.impressions, gscPosition: r.gscPosition, priorPosition: r.priorPosition, slipped: r.slipped,
|
|
1078
|
+
organicRank: r.organicRank, serpDepth: r.serpDepth, organicRankNote: r.organicRankNote,
|
|
1079
|
+
aioPresent: r.aioPresent, aioCitesUs: r.aioCitesUs, aioResolved: r.aioResolved, aioNote: r.aioNote, videoPresent: r.videoPresent,
|
|
1080
|
+
verdict: r.verdict, verdictNote: r.verdictNote,
|
|
1081
|
+
opportunityClicks: r.priorityBasis?.opportunityClicks, positionSource: r.priorityBasis?.positionSource, targetPosition: r.priorityBasis?.targetPosition,
|
|
1082
|
+
hasProductSchema: r.hasProductSchema, dateModified: r.dateModified, todosInserted: r.todosInserted,
|
|
1083
|
+
...(r.competitorContent ? { competitorContent: r.competitorContent, competitorFetch: r.competitorFetch } : {}),
|
|
1084
|
+
}));
|
|
1085
|
+
return { siteUrl, cost, window: gscWindow, serpDepth: SERP_DEPTH, targets: slimTargets, summary: md };
|
|
980
1086
|
}
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
if (!errs.length)
|
|
990
|
-
return '';
|
|
991
|
-
const needKey = errs.filter(c => /not set/i.test(c.error));
|
|
992
|
-
const blocked = errs.filter(c => !/not set/i.test(c.error));
|
|
993
|
-
let note = '';
|
|
994
|
-
if (blocked.length)
|
|
995
|
-
note += `\n\n${blocked.length} competitor(s) couldn't be fetched (bot-protection like Cloudflare, a block, or a fetch error): ${blocked.slice(0, 5).map(c => c.url).join(', ')}. If one matters, paste its copy here or drop a text file and I'll diff it in.`;
|
|
996
|
-
if (needKey.length)
|
|
997
|
-
note += `\n\n${needKey.length} competitor(s) skipped for a missing API key: ${[...new Set(needKey.map(c => c.error))].join('; ')}.`;
|
|
998
|
-
return note;
|
|
999
|
-
})() +
|
|
1000
|
-
`\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
|
|
1001
|
-
return { content: [{ type: 'text', text: md }], structuredContent: { siteUrl, cost, targets: results } };
|
|
1002
|
-
}
|
|
1003
|
-
finally {
|
|
1004
|
-
db.close();
|
|
1005
|
-
}
|
|
1087
|
+
finally {
|
|
1088
|
+
db.close();
|
|
1089
|
+
}
|
|
1090
|
+
});
|
|
1091
|
+
return {
|
|
1092
|
+
content: [{ type: 'text', text: `Content recon started for ${count} page(s) — job ${jobId}. Each page is a live page-fetch + a serialised SERP call (a few seconds each), so poll check_sync_status with jobId "${jobId}"; the finished job's result carries the per-page verdicts, to-dos and a summary.` }],
|
|
1093
|
+
structuredContent: { jobId, status: 'running', siteUrl },
|
|
1094
|
+
};
|
|
1006
1095
|
});
|
|
1007
1096
|
// save_recon_todo — the research session writes its content-gap / originality findings back
|
|
1008
1097
|
// into the ledger against a page (source: 'research').
|
|
@@ -1024,9 +1113,15 @@ export function createServer() {
|
|
|
1024
1113
|
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
1025
1114
|
try {
|
|
1026
1115
|
const key = urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' });
|
|
1027
|
-
const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position FROM recon_page WHERE url_key=?`).get(key);
|
|
1028
|
-
|
|
1029
|
-
|
|
1116
|
+
const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position, gsc_impressions, verdict FROM recon_page WHERE url_key=?`).get(key);
|
|
1117
|
+
// aio_cites_us is NULL when the AI Overview never resolved — keep that as unknown rather
|
|
1118
|
+
// than coercing it to false, or the outcome diff reads "no → yes" off a missing fetch.
|
|
1119
|
+
const baseline = page ? { organicRank: page.organic_rank, aioCitesUs: page.aio_cites_us == null ? null : !!page.aio_cites_us, gscPosition: page.gsc_position, at: new Date().toISOString().slice(0, 10) } : {};
|
|
1120
|
+
// Default priority to the PAGE's opportunity so research to-dos sort on the same scale as
|
|
1121
|
+
// the deterministic ones (a default of 0 buried every judgement finding at the bottom).
|
|
1122
|
+
const basis = page ? opportunityBasis(page.gsc_impressions ?? 0, page.organic_rank, page.gsc_position ?? 20, page.verdict) : null;
|
|
1123
|
+
const fallbackPriority = basis ? Math.max(1, Math.round(basis.opportunityClicks * basis.verdictFactor)) : 0;
|
|
1124
|
+
const drafts = todos.map(t => ({ action: t.action, type: t.type ?? 'content-gap', rationale: t.rationale ?? '', evidence: t.evidence ?? {}, priority: t.priority ?? fallbackPriority }));
|
|
1030
1125
|
const n = insertTodos(db.db, key, page?.query ?? '', drafts, baseline, 'research');
|
|
1031
1126
|
return { content: [{ type: 'text', text: `Saved ${n} recon to-do(s) for ${key}${n < drafts.length ? ` (${drafts.length - n} already open)` : ''}. Track with recon_todos.` }], structuredContent: { urlKey: key, inserted: n } };
|
|
1032
1127
|
}
|
|
@@ -1061,11 +1156,14 @@ export function createServer() {
|
|
|
1061
1156
|
let outcome = row.outcome;
|
|
1062
1157
|
let outcomeNote = '';
|
|
1063
1158
|
if (status === 'shipped' && remeasure && dfs) {
|
|
1064
|
-
|
|
1065
|
-
const
|
|
1066
|
-
|
|
1159
|
+
// Same depth + async-AIO handling as recon_targets, so before/after are comparable.
|
|
1160
|
+
const serpResp = await dfs.serpOrganic(row.query, location, 'en', 20, { loadAsyncAiOverview: true });
|
|
1161
|
+
const serp = parseSerpForRecon(serpResp, dfsHost(siteUrl), 20);
|
|
1162
|
+
const cited = (v) => (v == null ? 'unknown' : v ? 'yes' : 'no');
|
|
1163
|
+
outcome = JSON.stringify({ organicRank: serp.ourOrganicRank, serpDepth: 20, aioCitesUs: serp.aioCitesUs, aioResolved: serp.aioResolved, at: new Date().toISOString().slice(0, 10) });
|
|
1067
1164
|
const base = row.baseline ? JSON.parse(row.baseline) : {};
|
|
1068
|
-
|
|
1165
|
+
const rankStr = (r) => (r == null ? 'not in top 20' : `#${r}`);
|
|
1166
|
+
outcomeNote = ` Outcome: organic ${rankStr(base.organicRank)}→${rankStr(serp.ourOrganicRank)}, AIO cited ${cited(base.aioCitesUs)}→${cited(serp.aioCitesUs)}.`;
|
|
1069
1167
|
}
|
|
1070
1168
|
db.db.prepare(`UPDATE recon_todo SET status=COALESCE(?,status), notes=?, outcome=COALESCE(?,outcome), updated_at=datetime('now') WHERE id=?`)
|
|
1071
1169
|
.run(status ?? null, notes, outcome ?? null, id);
|