@houtini/seo-audit-console 0.3.0 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -3
- package/dist/audit/recon.d.ts +56 -0
- package/dist/audit/recon.d.ts.map +1 -0
- package/dist/audit/recon.js +134 -0
- package/dist/audit/recon.js.map +1 -0
- package/dist/core/AuditDatabase.d.ts.map +1 -1
- package/dist/core/AuditDatabase.js +39 -0
- package/dist/core/AuditDatabase.js.map +1 -1
- package/dist/core/FirecrawlClient.d.ts +35 -0
- package/dist/core/FirecrawlClient.d.ts.map +1 -0
- package/dist/core/FirecrawlClient.js +69 -0
- package/dist/core/FirecrawlClient.js.map +1 -0
- package/dist/core/SupadataClient.d.ts +26 -0
- package/dist/core/SupadataClient.d.ts.map +1 -0
- package/dist/core/SupadataClient.js +57 -0
- package/dist/core/SupadataClient.js.map +1 -0
- package/dist/core/reconFetch.d.ts +29 -0
- package/dist/core/reconFetch.d.ts.map +1 -0
- package/dist/core/reconFetch.js +49 -0
- package/dist/core/reconFetch.js.map +1 -0
- package/dist/core/reconResearch.d.ts +42 -0
- package/dist/core/reconResearch.d.ts.map +1 -0
- package/dist/core/reconResearch.js +101 -0
- package/dist/core/reconResearch.js.map +1 -0
- package/dist/core/serpRecon.d.ts +46 -0
- package/dist/core/serpRecon.d.ts.map +1 -0
- package/dist/core/serpRecon.js +71 -0
- package/dist/core/serpRecon.js.map +1 -0
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +230 -0
- package/dist/server.js.map +1 -1
- package/package.json +1 -1
- package/server.json +18 -4
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import { extractPage } from './extract.js';
|
|
2
|
+
export async function fetchOwnPage(url, hostForm = 'asis', timeoutMs = 20000) {
|
|
3
|
+
const host = new URL(url).hostname.replace(/^www\./, '');
|
|
4
|
+
const res = await fetch(url, {
|
|
5
|
+
headers: { 'user-agent': 'seo-audit-console recon (+https://github.com/houtini-ai/seo-audit)' },
|
|
6
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
7
|
+
});
|
|
8
|
+
const html = await res.text();
|
|
9
|
+
const ex = extractPage(html, url, host, { hostForm }, res.headers.get('x-robots-tag'));
|
|
10
|
+
const chunks = ex.bodyChunks ?? [];
|
|
11
|
+
const fulltextLower = chunks.map(c => `${c.heading} ${c.text}`).join(' ').toLowerCase();
|
|
12
|
+
// jsonLd is a JSON *string* (array of raw blocks) — parse ONCE, then each block (itself a
|
|
13
|
+
// JSON string or object) may carry an @graph. (Build lesson: don't iterate the raw string.)
|
|
14
|
+
const types = new Set();
|
|
15
|
+
let dateModified = null, datePublished = null;
|
|
16
|
+
try {
|
|
17
|
+
const blocks = ex.jsonLd ? JSON.parse(ex.jsonLd) : [];
|
|
18
|
+
for (const block of blocks) {
|
|
19
|
+
const obj = typeof block === 'string' ? JSON.parse(block) : block;
|
|
20
|
+
for (const node of (obj['@graph'] ?? [obj])) {
|
|
21
|
+
const t = node['@type'];
|
|
22
|
+
(Array.isArray(t) ? t : [t]).forEach((x) => x && types.add(x));
|
|
23
|
+
if (node.dateModified && !dateModified)
|
|
24
|
+
dateModified = String(node.dateModified).slice(0, 10);
|
|
25
|
+
if (node.datePublished && !datePublished)
|
|
26
|
+
datePublished = String(node.datePublished).slice(0, 10);
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
catch { /* malformed JSON-LD — leave types empty rather than throw */ }
|
|
31
|
+
return {
|
|
32
|
+
url,
|
|
33
|
+
status: res.status,
|
|
34
|
+
title: ex.title,
|
|
35
|
+
h1: ex.h1,
|
|
36
|
+
wordCount: ex.wordCount ?? fulltextLower.split(/\s+/).length,
|
|
37
|
+
headings: chunks.filter(c => c.heading).map(c => ({ level: c.level, heading: c.heading, chars: c.text.length })),
|
|
38
|
+
fulltextLower,
|
|
39
|
+
jsonLdTypes: [...types],
|
|
40
|
+
hasProductOrReview: types.has('Product') || types.has('Review') || types.has('AggregateRating'),
|
|
41
|
+
dateModified,
|
|
42
|
+
datePublished,
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
/** Which of these terms does the page text actually contain? Powers the coverage-gap probe. */
|
|
46
|
+
export function coverageProbe(fulltextLower, terms) {
|
|
47
|
+
return terms.map(t => ({ term: t, present: fulltextLower.includes(t.toLowerCase()) }));
|
|
48
|
+
}
|
|
49
|
+
//# sourceMappingURL=reconFetch.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reconFetch.js","sourceRoot":"","sources":["../../src/core/reconFetch.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,cAAc,CAAC;AAqB3C,MAAM,CAAC,KAAK,UAAU,YAAY,CAAC,GAAW,EAAE,WAAoC,MAAM,EAAE,SAAS,GAAG,KAAK;IAC3G,MAAM,IAAI,GAAG,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,OAAO,CAAC,QAAQ,EAAE,EAAE,CAAC,CAAC;IACzD,MAAM,GAAG,GAAG,MAAM,KAAK,CAAC,GAAG,EAAE;QAC3B,OAAO,EAAE,EAAE,YAAY,EAAE,oEAAoE,EAAE;QAC/F,MAAM,EAAE,WAAW,CAAC,OAAO,CAAC,SAAS,CAAC;KACvC,CAAC,CAAC;IACH,MAAM,IAAI,GAAG,MAAM,GAAG,CAAC,IAAI,EAAE,CAAC;IAC9B,MAAM,EAAE,GAAG,WAAW,CAAC,IAAI,EAAE,GAAG,EAAE,IAAI,EAAE,EAAE,QAAQ,EAAE,EAAE,GAAG,CAAC,OAAO,CAAC,GAAG,CAAC,cAAc,CAAC,CAAC,CAAC;IAEvF,MAAM,MAAM,GAAG,EAAE,CAAC,UAAU,IAAI,EAAE,CAAC;IACnC,MAAM,aAAa,GAAG,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,GAAG,CAAC,CAAC,OAAO,IAAI,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,WAAW,EAAE,CAAC;IAExF,0FAA0F;IAC1F,4FAA4F;IAC5F,MAAM,KAAK,GAAG,IAAI,GAAG,EAAU,CAAC;IAChC,IAAI,YAAY,GAAkB,IAAI,EAAE,aAAa,GAAkB,IAAI,CAAC;IAC5E,IAAI,CAAC;QACH,MAAM,MAAM,GAAG,EAAE,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;QACtD,KAAK,MAAM,KAAK,IAAI,MAAM,EAAE,CAAC;YAC3B,MAAM,GAAG,GAAG,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC;YAClE,KAAK,MAAM,IAAI,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC;gBAC5C,MAAM,CAAC,GAAG,IAAI,CAAC,OAAO,CAAC,CAAC;gBACxB,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAS,EAAE,EAAE,CAAC,CAAC,IAAI,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;gBACvE,IAAI,IAAI,CAAC,YAAY,IAAI,CAAC,YAAY;oBAAE,YAAY,GAAG,MAAM,CAAC,IAAI,CAAC,YAAY,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;gBAC9F,IAAI,IAAI,CAAC,aAAa,IAAI,CAAC,aAAa;oBAAE,aAAa,GAAG,MAAM,CAAC,IAAI,CAAC,aAAa,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;YACpG,CAAC;QACH,CAAC;IACH,CAAC;IAAC,MAAM,CAAC,CAAC,6DAA6D,CAAC,CAAC;IAEzE,OAAO;QACL,GAAG;QACH,MAAM,EAAE,GAAG,CAAC,MAAM;QAClB,KAAK,EAAE,EAAE,CAAC,KAAK;QACf,EAAE,EAAE,EAAE,CAAC,EAAE;QACT,SAAS,EAAE,EAAE,CAAC,SAAS,IAAI,aAAa,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM;QAC5D,QAAQ,EAAE,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,KAAK,EAAE,CAAC,CAAC,IAAI,CAAC,MAAM,EAAE,CAAC,CAAC;QAChH,aAAa;QACb,WAAW,EAAE,CAAC,GAAG,KAAK,CAAC;QACvB,kBAAkB,EAAE,KAAK,CAAC,GAAG,CAAC,SAAS,CAAC,IAAI,KAAK,CAAC,GAAG,CAAC,QAAQ,CAAC,IAAI,KAAK,CAAC,GAAG,CAAC,iBAAiB,CAAC;QAC/F,YAAY;QACZ,aAAa;KACd,CAAC;AACJ,CAAC;AAED,+FAA+F;AAC/F,MAAM,UAAU,aAAa,CAAC,aAAqB,EAAE,KAAe;IAClE,OAAO,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,EAAE,IAAI,EAAE,CAAC,EAAE,OAAO,EAAE,aAAa,CAAC,QAAQ,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC;AACzF,CAAC"}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
import type { FirecrawlClient } from './FirecrawlClient.js';
|
|
2
|
+
import type { SupadataClient } from './SupadataClient.js';
|
|
3
|
+
/** User-agent modes for the free HTTP fetch. 'browser' mimics a real visit arriving from Google
|
|
4
|
+
* (referer + cross-site fetch metadata) — this is what gets past Reddit and the mid-tier.
|
|
5
|
+
* 'googlebot' spoofs Googlebot's UA (UA-gated sites only; strict-verified Cloudflare rejects it).
|
|
6
|
+
* Cloudflare *managed-challenge* sites (e.g. PCMag) reject every HTTP-only client regardless. */
|
|
7
|
+
export type UaMode = 'browser' | 'googlebot';
|
|
8
|
+
/**
|
|
9
|
+
* Route a competitor URL to the right fetcher by who owns the source:
|
|
10
|
+
* YouTube/TikTok/X → Supadata (transcript) — Firecrawl can't read video
|
|
11
|
+
* Reddit → its own .json endpoint — Firecrawl hard-blocks Reddit
|
|
12
|
+
* everything else → Firecrawl (markdown, proxy:auto for bot-protected sites)
|
|
13
|
+
* Each returns a uniform shape so the recon packet is easy for the research session to diff.
|
|
14
|
+
*/
|
|
15
|
+
export interface CompetitorContent {
|
|
16
|
+
url: string;
|
|
17
|
+
kind: 'video' | 'reddit' | 'page';
|
|
18
|
+
title: string | null;
|
|
19
|
+
content: string;
|
|
20
|
+
cached: boolean;
|
|
21
|
+
error?: string;
|
|
22
|
+
}
|
|
23
|
+
/** Reddit serves clean JSON at url.json — title + selftext + top comments, no scraper needed.
|
|
24
|
+
* Needs the browser header profile (UA + Google referer) — a bot UA gets a 403. */
|
|
25
|
+
export declare function fetchRedditJson(url: string, maxChars?: number, ua?: UaMode): Promise<{
|
|
26
|
+
title: string | null;
|
|
27
|
+
content: string;
|
|
28
|
+
}>;
|
|
29
|
+
/** Free HTTP fallback for pages Firecrawl refuses (its blocklist covers major publishers).
|
|
30
|
+
* Fetches with a browser-like UA and reuses our own extractor for clean main-content text. */
|
|
31
|
+
export declare function fetchPlainPage(url: string, maxChars?: number, ua?: UaMode): Promise<{
|
|
32
|
+
title: string | null;
|
|
33
|
+
content: string;
|
|
34
|
+
}>;
|
|
35
|
+
export declare function fetchCompetitorContent(url: string, clients: {
|
|
36
|
+
firecrawl: FirecrawlClient | null;
|
|
37
|
+
supadata: SupadataClient | null;
|
|
38
|
+
}, opts?: {
|
|
39
|
+
maxChars?: number;
|
|
40
|
+
ua?: UaMode;
|
|
41
|
+
}): Promise<CompetitorContent>;
|
|
42
|
+
//# sourceMappingURL=reconResearch.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reconResearch.d.ts","sourceRoot":"","sources":["../../src/core/reconResearch.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,sBAAsB,CAAC;AAC5D,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,qBAAqB,CAAC;AAQ1D;;;iGAGiG;AACjG,MAAM,MAAM,MAAM,GAAG,SAAS,GAAG,WAAW,CAAC;AAmB7C;;;;;;GAMG;AACH,MAAM,WAAW,iBAAiB;IAChC,GAAG,EAAE,MAAM,CAAC;IACZ,IAAI,EAAE,OAAO,GAAG,QAAQ,GAAG,MAAM,CAAC;IAClC,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;IACrB,OAAO,EAAE,MAAM,CAAC;IAChB,MAAM,EAAE,OAAO,CAAC;IAChB,KAAK,CAAC,EAAE,MAAM,CAAC;CAChB;AAKD;mFACmF;AACnF,wBAAsB,eAAe,CAAC,GAAG,EAAE,MAAM,EAAE,QAAQ,SAAO,EAAE,EAAE,GAAE,MAAkB,GAAG,OAAO,CAAC;IAAE,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC,CAkB9I;AAED;8FAC8F;AAC9F,wBAAsB,cAAc,CAAC,GAAG,EAAE,MAAM,EAAE,QAAQ,SAAO,EAAE,EAAE,GAAE,MAAkB,GAAG,OAAO,CAAC;IAAE,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC,CAQ7I;AAED,wBAAsB,sBAAsB,CAC1C,GAAG,EAAE,MAAM,EACX,OAAO,EAAE;IAAE,SAAS,EAAE,eAAe,GAAG,IAAI,CAAC;IAAC,QAAQ,EAAE,cAAc,GAAG,IAAI,CAAA;CAAE,EAC/E,IAAI,GAAE;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;IAAC,EAAE,CAAC,EAAE,MAAM,CAAA;CAAO,GAC5C,OAAO,CAAC,iBAAiB,CAAC,CA2B5B"}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import { Agent, fetch as ufetch } from 'undici';
|
|
2
|
+
import { hostOf } from './serpRecon.js';
|
|
3
|
+
import { extractPage } from './extract.js';
|
|
4
|
+
const httpAgent = new Agent({ allowH2: true, connections: 6, keepAliveTimeout: 30_000 });
|
|
5
|
+
const CHROME_UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36';
|
|
6
|
+
const GOOGLEBOT_UA = 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html) Chrome/131.0.0.0 Safari/537.36';
|
|
7
|
+
function browserHeaders(mode) {
|
|
8
|
+
if (mode === 'googlebot') {
|
|
9
|
+
return { 'user-agent': GOOGLEBOT_UA, 'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', 'accept-encoding': 'gzip, deflate, br', 'from': 'googlebot(at)googlebot.com' };
|
|
10
|
+
}
|
|
11
|
+
return {
|
|
12
|
+
'user-agent': CHROME_UA,
|
|
13
|
+
'accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
|
|
14
|
+
'accept-language': 'en-GB,en-US;q=0.9,en;q=0.8',
|
|
15
|
+
'accept-encoding': 'gzip, deflate, br',
|
|
16
|
+
'referer': 'https://www.google.com/',
|
|
17
|
+
'sec-ch-ua': '"Google Chrome";v="131", "Chromium";v="131", "Not_A Brand";v="24"',
|
|
18
|
+
'sec-ch-ua-mobile': '?0', 'sec-ch-ua-platform': '"Windows"',
|
|
19
|
+
'sec-fetch-dest': 'document', 'sec-fetch-mode': 'navigate', 'sec-fetch-site': 'cross-site', 'sec-fetch-user': '?1',
|
|
20
|
+
'upgrade-insecure-requests': '1',
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
const isVideo = (host) => /youtube|youtu\.be|tiktok|twitter\.com|x\.com/.test(host);
|
|
24
|
+
const isReddit = (host) => /reddit\.com/.test(host);
|
|
25
|
+
/** Reddit serves clean JSON at url.json — title + selftext + top comments, no scraper needed.
|
|
26
|
+
* Needs the browser header profile (UA + Google referer) — a bot UA gets a 403. */
|
|
27
|
+
export async function fetchRedditJson(url, maxChars = 4000, ua = 'browser') {
|
|
28
|
+
const jsonUrl = url.split('?')[0].replace(/\/$/, '') + '.json';
|
|
29
|
+
const res = await ufetch(jsonUrl, { headers: browserHeaders(ua), dispatcher: httpAgent, signal: AbortSignal.timeout(20000) });
|
|
30
|
+
if (!res.ok)
|
|
31
|
+
throw new Error(`Reddit .json ${res.status}`);
|
|
32
|
+
const data = await res.json();
|
|
33
|
+
const post = data?.[0]?.data?.children?.[0]?.data ?? {};
|
|
34
|
+
const comments = data?.[1]?.data?.children ?? [];
|
|
35
|
+
const parts = [];
|
|
36
|
+
if (post.title)
|
|
37
|
+
parts.push(`# ${post.title}`);
|
|
38
|
+
if (post.selftext)
|
|
39
|
+
parts.push(post.selftext);
|
|
40
|
+
const top = comments
|
|
41
|
+
.map(c => c?.data)
|
|
42
|
+
.filter(d => d && d.body && d.body !== '[deleted]')
|
|
43
|
+
.sort((a, b) => (b.score ?? 0) - (a.score ?? 0))
|
|
44
|
+
.slice(0, 12)
|
|
45
|
+
.map(d => `- (${d.score ?? 0}) ${d.body}`);
|
|
46
|
+
if (top.length)
|
|
47
|
+
parts.push('## Top comments', ...top);
|
|
48
|
+
return { title: post.title ?? null, content: parts.join('\n\n').slice(0, maxChars) };
|
|
49
|
+
}
|
|
50
|
+
/** Free HTTP fallback for pages Firecrawl refuses (its blocklist covers major publishers).
|
|
51
|
+
* Fetches with a browser-like UA and reuses our own extractor for clean main-content text. */
|
|
52
|
+
export async function fetchPlainPage(url, maxChars = 4000, ua = 'browser') {
|
|
53
|
+
const res = await ufetch(url, { headers: browserHeaders(ua), dispatcher: httpAgent, signal: AbortSignal.timeout(20000) });
|
|
54
|
+
if (!res.ok)
|
|
55
|
+
throw new Error(`plain fetch ${res.status}`);
|
|
56
|
+
const html = await res.text();
|
|
57
|
+
const host = new URL(url).hostname.replace(/^www\./, '');
|
|
58
|
+
const ex = extractPage(html, url, host, { hostForm: 'asis' }, null);
|
|
59
|
+
const text = (ex.bodyChunks ?? []).map(c => `${c.heading} ${c.text}`).join('\n').trim();
|
|
60
|
+
return { title: ex.title, content: text.slice(0, maxChars) };
|
|
61
|
+
}
|
|
62
|
+
export async function fetchCompetitorContent(url, clients, opts = {}) {
|
|
63
|
+
const host = hostOf(url);
|
|
64
|
+
const max = opts.maxChars ?? 4000;
|
|
65
|
+
const ua = opts.ua ?? 'browser';
|
|
66
|
+
try {
|
|
67
|
+
if (isVideo(host)) {
|
|
68
|
+
if (!clients.supadata)
|
|
69
|
+
return { url, kind: 'video', title: null, content: '', cached: false, error: 'SUPADATA_API_KEY not set — cannot transcribe video' };
|
|
70
|
+
const t = await clients.supadata.transcript(url, { maxChars: Math.max(max, 5000) });
|
|
71
|
+
return { url, kind: 'video', title: null, content: t.content, cached: t.cached };
|
|
72
|
+
}
|
|
73
|
+
if (isReddit(host)) {
|
|
74
|
+
// .json first (cleaner structured data); HTML fallback. Both need the browser profile.
|
|
75
|
+
try {
|
|
76
|
+
const r = await fetchRedditJson(url, max, ua);
|
|
77
|
+
return { url, kind: 'reddit', title: r.title, content: r.content, cached: false };
|
|
78
|
+
}
|
|
79
|
+
catch {
|
|
80
|
+
const r = await fetchPlainPage(url, max, ua);
|
|
81
|
+
return { url, kind: 'reddit', title: r.title, content: r.content, cached: false };
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
// Pages: Firecrawl first (cleaner, handles JS); fall back to the free browser-profile fetch
|
|
85
|
+
// for sites Firecrawl refuses ("we do not support this site") or when no key is set.
|
|
86
|
+
if (clients.firecrawl) {
|
|
87
|
+
try {
|
|
88
|
+
const s = await clients.firecrawl.scrape(url, { maxChars: max, proxy: 'auto' });
|
|
89
|
+
return { url, kind: 'page', title: s.title, content: s.markdown, cached: s.cached };
|
|
90
|
+
}
|
|
91
|
+
catch { /* fall through to the browser-profile fetch */ }
|
|
92
|
+
}
|
|
93
|
+
const p = await fetchPlainPage(url, max, ua);
|
|
94
|
+
return { url, kind: 'page', title: p.title, content: p.content, cached: false };
|
|
95
|
+
}
|
|
96
|
+
catch (e) {
|
|
97
|
+
const kind = isVideo(host) ? 'video' : isReddit(host) ? 'reddit' : 'page';
|
|
98
|
+
return { url, kind, title: null, content: '', cached: false, error: e instanceof Error ? e.message : String(e) };
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
//# sourceMappingURL=reconResearch.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"reconResearch.js","sourceRoot":"","sources":["../../src/core/reconResearch.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,KAAK,EAAE,KAAK,IAAI,MAAM,EAAE,MAAM,QAAQ,CAAC;AAGhD,OAAO,EAAE,MAAM,EAAE,MAAM,gBAAgB,CAAC;AACxC,OAAO,EAAE,WAAW,EAAE,MAAM,cAAc,CAAC;AAE3C,MAAM,SAAS,GAAG,IAAI,KAAK,CAAC,EAAE,OAAO,EAAE,IAAI,EAAE,WAAW,EAAE,CAAC,EAAE,gBAAgB,EAAE,MAAM,EAAE,CAAC,CAAC;AACzF,MAAM,SAAS,GAAG,iHAAiH,CAAC;AACpI,MAAM,YAAY,GAAG,+IAA+I,CAAC;AAQrK,SAAS,cAAc,CAAC,IAAY;IAClC,IAAI,IAAI,KAAK,WAAW,EAAE,CAAC;QACzB,OAAO,EAAE,YAAY,EAAE,YAAY,EAAE,QAAQ,EAAE,iEAAiE,EAAE,iBAAiB,EAAE,mBAAmB,EAAE,MAAM,EAAE,4BAA4B,EAAE,CAAC;IACnM,CAAC;IACD,OAAO;QACL,YAAY,EAAE,SAAS;QACvB,QAAQ,EAAE,kGAAkG;QAC5G,iBAAiB,EAAE,4BAA4B;QAC/C,iBAAiB,EAAE,mBAAmB;QACtC,SAAS,EAAE,yBAAyB;QACpC,WAAW,EAAE,mEAAmE;QAChF,kBAAkB,EAAE,IAAI,EAAE,oBAAoB,EAAE,WAAW;QAC3D,gBAAgB,EAAE,UAAU,EAAE,gBAAgB,EAAE,UAAU,EAAE,gBAAgB,EAAE,YAAY,EAAE,gBAAgB,EAAE,IAAI;QAClH,2BAA2B,EAAE,GAAG;KACjC,CAAC;AACJ,CAAC;AAkBD,MAAM,OAAO,GAAG,CAAC,IAAY,EAAW,EAAE,CAAC,8CAA8C,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AACrG,MAAM,QAAQ,GAAG,CAAC,IAAY,EAAW,EAAE,CAAC,aAAa,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAErE;mFACmF;AACnF,MAAM,CAAC,KAAK,UAAU,eAAe,CAAC,GAAW,EAAE,QAAQ,GAAG,IAAI,EAAE,KAAa,SAAS;IACxF,MAAM,OAAO,GAAG,GAAG,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,KAAK,EAAE,EAAE,CAAC,GAAG,OAAO,CAAC;IAC/D,MAAM,GAAG,GAAG,MAAM,MAAM,CAAC,OAAO,EAAE,EAAE,OAAO,EAAE,cAAc,CAAC,EAAE,CAAC,EAAE,UAAU,EAAE,SAAS,EAAE,MAAM,EAAE,WAAW,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;IAC9H,IAAI,CAAC,GAAG,CAAC,EAAE;QAAE,MAAM,IAAI,KAAK,CAAC,gBAAgB,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC;IAC3D,MAAM,IAAI,GAAQ,MAAM,GAAG,CAAC,IAAI,EAAE,CAAC;IACnC,MAAM,IAAI,GAAG,IAAI,EAAE,CAAC,CAAC,CAAC,EAAE,IAAI,EAAE,QAAQ,EAAE,CAAC,CAAC,CAAC,EAAE,IAAI,IAAI,EAAE,CAAC;IACxD,MAAM,QAAQ,GAAU,IAAI,EAAE,CAAC,CAAC,CAAC,EAAE,IAAI,EAAE,QAAQ,IAAI,EAAE,CAAC;IACxD,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,IAAI,CAAC,KAAK;QAAE,KAAK,CAAC,IAAI,CAAC,KAAK,IAAI,CAAC,KAAK,EAAE,CAAC,CAAC;IAC9C,IAAI,IAAI,CAAC,QAAQ;QAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC;IAC7C,MAAM,GAAG,GAAG,QAAQ;SACjB,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,EAAE,IAAI,CAAC;SACjB,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,IAAI,CAAC,CAAC,IAAI,IAAI,CAAC,CAAC,IAAI,KAAK,WAAW,CAAC;SAClD,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC;SAC/C,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC;SACZ,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,MAAM,CAAC,CAAC,KAAK,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;IAC7C,IAAI,GAAG,CAAC,MAAM;QAAE,KAAK,CAAC,IAAI,CAAC,iBAAiB,EAAE,GAAG,GAAG,CAAC,CAAC;IACtD,OAAO,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,IAAI,IAAI,EAAE,OAAO,EAAE,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,QAAQ,CAAC,EAAE,CAAC;AACvF,CAAC;AAED;8FAC8F;AAC9F,MAAM,CAAC,KAAK,UAAU,cAAc,CAAC,GAAW,EAAE,QAAQ,GAAG,IAAI,EAAE,KAAa,SAAS;IACvF,MAAM,GAAG,GAAG,MAAM,MAAM,CAAC,GAAG,EAAE,EAAE,OAAO,EAAE,cAAc,CAAC,EAAE,CAAC,EAAE,UAAU,EAAE,SAAS,EAAE,MAAM,EAAE,WAAW,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;IAC1H,IAAI,CAAC,GAAG,CAAC,EAAE;QAAE,MAAM,IAAI,KAAK,CAAC,eAAe,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC;IAC1D,MAAM,IAAI,GAAG,MAAM,GAAG,CAAC,IAAI,EAAE,CAAC;IAC9B,MAAM,IAAI,GAAG,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,OAAO,CAAC,QAAQ,EAAE,EAAE,CAAC,CAAC;IACzD,MAAM,EAAE,GAAG,WAAW,CAAC,IAAI,EAAE,GAAG,EAAE,IAAI,EAAE,EAAE,QAAQ,EAAE,MAAM,EAAE,EAAE,IAAI,CAAC,CAAC;IACpE,MAAM,IAAI,GAAG,CAAC,EAAE,CAAC,UAAU,IAAI,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,GAAG,CAAC,CAAC,OAAO,IAAI,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,IAAI,EAAE,CAAC;IACxF,OAAO,EAAE,KAAK,EAAE,EAAE,CAAC,KAAK,EAAE,OAAO,EAAE,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,QAAQ,CAAC,EAAE,CAAC;AAC/D,CAAC;AAED,MAAM,CAAC,KAAK,UAAU,sBAAsB,CAC1C,GAAW,EACX,OAA+E,EAC/E,OAA2C,EAAE;IAE7C,MAAM,IAAI,GAAG,MAAM,CAAC,GAAG,CAAC,CAAC;IACzB,MAAM,GAAG,GAAG,IAAI,CAAC,QAAQ,IAAI,IAAI,CAAC;IAClC,MAAM,EAAE,GAAG,IAAI,CAAC,EAAE,IAAI,SAAS,CAAC;IAChC,IAAI,CAAC;QACH,IAAI,OAAO,CAAC,IAAI,CAAC,EAAE,CAAC;YAClB,IAAI,CAAC,OAAO,CAAC,QAAQ;gBAAE,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,EAAE,EAAE,MAAM,EAAE,KAAK,EAAE,KAAK,EAAE,oDAAoD,EAAE,CAAC;YAC3J,MAAM,CAAC,GAAG,MAAM,OAAO,CAAC,QAAQ,CAAC,UAAU,CAAC,GAAG,EAAE,EAAE,QAAQ,EAAE,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,IAAI,CAAC,EAAE,CAAC,CAAC;YACpF,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC;QACnF,CAAC;QACD,IAAI,QAAQ,CAAC,IAAI,CAAC,EAAE,CAAC;YACnB,uFAAuF;YACvF,IAAI,CAAC;gBAAC,MAAM,CAAC,GAAG,MAAM,eAAe,CAAC,GAAG,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC;gBAAC,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,QAAQ,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC;YAAC,CAAC;YACzI,MAAM,CAAC;gBAAC,MAAM,CAAC,GAAG,MAAM,cAAc,CAAC,GAAG,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC;gBAAC,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,QAAQ,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC;YAAC,CAAC;QAC5I,CAAC;QACD,4FAA4F;QAC5F,qFAAqF;QACrF,IAAI,OAAO,CAAC,SAAS,EAAE,CAAC;YACtB,IAAI,CAAC;gBAAC,MAAM,CAAC,GAAG,MAAM,OAAO,CAAC,SAAS,CAAC,MAAM,CAAC,GAAG,EAAE,EAAE,QAAQ,EAAE,GAAG,EAAE,KAAK,EAAE,MAAM,EAAE,CAAC,CAAC;gBAAC,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC,QAAQ,EAAE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC;YAAC,CAAC;YAC7K,MAAM,CAAC,CAAC,+CAA+C,CAAC,CAAC;QAC3D,CAAC;QACD,MAAM,CAAC,GAAG,MAAM,cAAc,CAAC,GAAG,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC;QAC7C,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,CAAC;IAClF,CAAC;IAAC,OAAO,CAAC,EAAE,CAAC;QACX,MAAM,IAAI,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,MAAM,CAAC;QAC1E,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,EAAE,EAAE,MAAM,EAAE,KAAK,EAAE,KAAK,EAAE,CAAC,YAAY,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC;IACnH,CAAC;AACH,CAAC"}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import type { DfsResponse } from './DataForSeoClient.js';
|
|
2
|
+
/**
|
|
3
|
+
* Parse a DataForSEO SERP-advanced response into the fields content recon needs:
|
|
4
|
+
* where WE actually rank in organic, whether the AI Overview cites us and who it does
|
|
5
|
+
* cite, and how much of the SERP is video/product (format signals). This is the single
|
|
6
|
+
* fetch that powers the organic-rank x AIO-citation verdict matrix.
|
|
7
|
+
*/
|
|
8
|
+
export interface SerpReconResult {
|
|
9
|
+
itemTypes: string[];
|
|
10
|
+
ourOrganicRank: number | null;
|
|
11
|
+
aioPresent: boolean;
|
|
12
|
+
aioCitesUs: boolean;
|
|
13
|
+
aioAsync: boolean;
|
|
14
|
+
aioReferences: {
|
|
15
|
+
rank: number | null;
|
|
16
|
+
domain: string;
|
|
17
|
+
url: string;
|
|
18
|
+
title: string;
|
|
19
|
+
}[];
|
|
20
|
+
videoPresent: boolean;
|
|
21
|
+
organicAbove: {
|
|
22
|
+
rank: number;
|
|
23
|
+
domain: string;
|
|
24
|
+
url: string;
|
|
25
|
+
title: string;
|
|
26
|
+
}[];
|
|
27
|
+
videoItems: {
|
|
28
|
+
url: string;
|
|
29
|
+
title: string;
|
|
30
|
+
source: string;
|
|
31
|
+
}[];
|
|
32
|
+
}
|
|
33
|
+
/** Registrable-ish host: lowercase, strip scheme/path/www. Good enough to match a domain field. */
|
|
34
|
+
export declare function hostOf(s: string | undefined | null): string;
|
|
35
|
+
export declare function parseSerpForRecon(resp: DfsResponse, ownDomain: string): SerpReconResult;
|
|
36
|
+
/**
|
|
37
|
+
* The deterministic verdict from organic rank x AIO citation (Richard's rule):
|
|
38
|
+
* strong organic but NOT in the AIO = a data-accuracy/freshness problem — Google ranks
|
|
39
|
+
* us but won't quote us.
|
|
40
|
+
*/
|
|
41
|
+
export type ReconVerdict = 'defend-and-deepen' | 'accuracy-or-freshness' | 'consolidate-weak-page' | 'competitive-gap';
|
|
42
|
+
export declare function reconVerdict(r: SerpReconResult): {
|
|
43
|
+
verdict: ReconVerdict;
|
|
44
|
+
note: string;
|
|
45
|
+
};
|
|
46
|
+
//# sourceMappingURL=serpRecon.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"serpRecon.d.ts","sourceRoot":"","sources":["../../src/core/serpRecon.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,uBAAuB,CAAC;AAEzD;;;;;GAKG;AACH,MAAM,WAAW,eAAe;IAC9B,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,UAAU,EAAE,OAAO,CAAC;IACpB,UAAU,EAAE,OAAO,CAAC;IACpB,QAAQ,EAAE,OAAO,CAAC;IAClB,aAAa,EAAE;QAAE,IAAI,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,MAAM,EAAE,MAAM,CAAC;QAAC,GAAG,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,EAAE,CAAC;IACrF,YAAY,EAAE,OAAO,CAAC;IACtB,YAAY,EAAE;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAC;QAAC,GAAG,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,EAAE,CAAC;IAC7E,UAAU,EAAE;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,EAAE,CAAC;CAC9D;AAED,mGAAmG;AACnG,wBAAgB,MAAM,CAAC,CAAC,EAAE,MAAM,GAAG,SAAS,GAAG,IAAI,GAAG,MAAM,CAG3D;AAED,wBAAgB,iBAAiB,CAAC,IAAI,EAAE,WAAW,EAAE,SAAS,EAAE,MAAM,GAAG,eAAe,CA2CvF;AAED;;;;GAIG;AACH,MAAM,MAAM,YAAY,GAAG,mBAAmB,GAAG,uBAAuB,GAAG,uBAAuB,GAAG,iBAAiB,CAAC;AAEvH,wBAAgB,YAAY,CAAC,CAAC,EAAE,eAAe,GAAG;IAAE,OAAO,EAAE,YAAY,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAkBxF"}
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
/** Registrable-ish host: lowercase, strip scheme/path/www. Good enough to match a domain field. */
|
|
2
|
+
export function hostOf(s) {
|
|
3
|
+
if (!s)
|
|
4
|
+
return '';
|
|
5
|
+
return String(s).replace(/^https?:\/\//i, '').replace(/\/.*$/, '').replace(/^www\./i, '').toLowerCase().trim();
|
|
6
|
+
}
|
|
7
|
+
export function parseSerpForRecon(resp, ownDomain) {
|
|
8
|
+
const own = hostOf(ownDomain);
|
|
9
|
+
const result = resp.tasks?.[0]?.result?.[0] ?? {};
|
|
10
|
+
const items = result.items ?? [];
|
|
11
|
+
const itemTypes = result.item_types ?? [...new Set(items.map(i => i.type))];
|
|
12
|
+
let ourOrganicRank = null;
|
|
13
|
+
const organic = [];
|
|
14
|
+
let aioPresent = false, aioCitesUs = false, aioAsync = false, videoPresent = false;
|
|
15
|
+
const aioReferences = [];
|
|
16
|
+
const videoItems = [];
|
|
17
|
+
for (const it of items) {
|
|
18
|
+
const t = it.type;
|
|
19
|
+
if (t === 'organic') {
|
|
20
|
+
const domain = hostOf(it.domain || it.url);
|
|
21
|
+
const rank = Number(it.rank_group) || organic.length + 1;
|
|
22
|
+
organic.push({ rank, domain, url: it.url ?? '', title: it.title ?? '' });
|
|
23
|
+
if (own && domain === own && ourOrganicRank == null)
|
|
24
|
+
ourOrganicRank = rank;
|
|
25
|
+
}
|
|
26
|
+
else if (t === 'ai_overview') {
|
|
27
|
+
aioPresent = true;
|
|
28
|
+
aioAsync = !!it.asynchronous_ai_overview;
|
|
29
|
+
const refs = it.references ?? [];
|
|
30
|
+
let i = 0;
|
|
31
|
+
for (const ref of refs) {
|
|
32
|
+
const domain = hostOf(ref.domain || ref.url);
|
|
33
|
+
aioReferences.push({ rank: Number(ref.rank_absolute) || ++i, domain, url: ref.url ?? '', title: ref.title ?? '' });
|
|
34
|
+
if (own && domain === own)
|
|
35
|
+
aioCitesUs = true;
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
else if (t === 'video' || t === 'short_videos') {
|
|
39
|
+
videoPresent = true;
|
|
40
|
+
for (const v of it.items ?? []) {
|
|
41
|
+
if (v?.url)
|
|
42
|
+
videoItems.push({ url: v.url, title: v.title ?? '', source: hostOf(v.url) });
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
// Organic ranked above us (the ones actually beating us); if we're absent, the whole top block.
|
|
47
|
+
const organicAbove = ourOrganicRank == null
|
|
48
|
+
? organic.slice(0, 8)
|
|
49
|
+
: organic.filter(o => o.rank < ourOrganicRank);
|
|
50
|
+
return { itemTypes, ourOrganicRank, aioPresent, aioCitesUs, aioAsync, aioReferences, videoPresent, organicAbove, videoItems };
|
|
51
|
+
}
|
|
52
|
+
export function reconVerdict(r) {
|
|
53
|
+
const strongOrganic = r.ourOrganicRank != null && r.ourOrganicRank <= 5;
|
|
54
|
+
const shape = r.videoPresent ? ' The SERP carries a video pack, so some click loss is format, not content.' : '';
|
|
55
|
+
if (!r.aioPresent) {
|
|
56
|
+
return strongOrganic
|
|
57
|
+
? { verdict: 'defend-and-deepen', note: `No AI Overview on this SERP; you rank organic #${r.ourOrganicRank}. Defend and deepen.${shape}` }
|
|
58
|
+
: { verdict: 'competitive-gap', note: `No AI Overview; you rank organic ${r.ourOrganicRank ?? 'outside top results'}. Close the competitive gap.${shape}` };
|
|
59
|
+
}
|
|
60
|
+
if (strongOrganic && r.aioCitesUs) {
|
|
61
|
+
return { verdict: 'defend-and-deepen', note: `You rank organic #${r.ourOrganicRank} and the AI Overview already cites you. Deepen to become the primary source.${shape}` };
|
|
62
|
+
}
|
|
63
|
+
if (strongOrganic && !r.aioCitesUs) {
|
|
64
|
+
return { verdict: 'accuracy-or-freshness', note: `You rank organic #${r.ourOrganicRank} but the AI Overview does NOT cite you — a data-accuracy or freshness signal. Make the facts current, correct and marked up so they are liftable.${shape}` };
|
|
65
|
+
}
|
|
66
|
+
if (!strongOrganic && r.aioCitesUs) {
|
|
67
|
+
return { verdict: 'consolidate-weak-page', note: `The AI Overview cites you but your organic rank is weak (${r.ourOrganicRank ?? 'absent'}) — you have a quotable nugget on an under-powered page; consolidate and strengthen it.${shape}` };
|
|
68
|
+
}
|
|
69
|
+
return { verdict: 'competitive-gap', note: `Not cited in the AI Overview and organic rank is ${r.ourOrganicRank ?? 'absent'} — full coverage gap; run the deep competitor diff.${shape}` };
|
|
70
|
+
}
|
|
71
|
+
//# sourceMappingURL=serpRecon.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"serpRecon.js","sourceRoot":"","sources":["../../src/core/serpRecon.ts"],"names":[],"mappings":"AAoBA,mGAAmG;AACnG,MAAM,UAAU,MAAM,CAAC,CAA4B;IACjD,IAAI,CAAC,CAAC;QAAE,OAAO,EAAE,CAAC;IAClB,OAAO,MAAM,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,eAAe,EAAE,EAAE,CAAC,CAAC,OAAO,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,OAAO,CAAC,SAAS,EAAE,EAAE,CAAC,CAAC,WAAW,EAAE,CAAC,IAAI,EAAE,CAAC;AACjH,CAAC;AAED,MAAM,UAAU,iBAAiB,CAAC,IAAiB,EAAE,SAAiB;IACpE,MAAM,GAAG,GAAG,MAAM,CAAC,SAAS,CAAC,CAAC;IAC9B,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,EAAE,CAAC,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;IAClD,MAAM,KAAK,GAAU,MAAM,CAAC,KAAK,IAAI,EAAE,CAAC;IACxC,MAAM,SAAS,GAAa,MAAM,CAAC,UAAU,IAAI,CAAC,GAAG,IAAI,GAAG,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IAEtF,IAAI,cAAc,GAAkB,IAAI,CAAC;IACzC,MAAM,OAAO,GAAmE,EAAE,CAAC;IACnF,IAAI,UAAU,GAAG,KAAK,EAAE,UAAU,GAAG,KAAK,EAAE,QAAQ,GAAG,KAAK,EAAE,YAAY,GAAG,KAAK,CAAC;IACnF,MAAM,aAAa,GAAqC,EAAE,CAAC;IAC3D,MAAM,UAAU,GAAkC,EAAE,CAAC;IAErD,KAAK,MAAM,EAAE,IAAI,KAAK,EAAE,CAAC;QACvB,MAAM,CAAC,GAAG,EAAE,CAAC,IAAI,CAAC;QAClB,IAAI,CAAC,KAAK,SAAS,EAAE,CAAC;YACpB,MAAM,MAAM,GAAG,MAAM,CAAC,EAAE,CAAC,MAAM,IAAI,EAAE,CAAC,GAAG,CAAC,CAAC;YAC3C,MAAM,IAAI,GAAG,MAAM,CAAC,EAAE,CAAC,UAAU,CAAC,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC,CAAC;YACzD,OAAO,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,GAAG,EAAE,EAAE,CAAC,GAAG,IAAI,EAAE,EAAE,KAAK,EAAE,EAAE,CAAC,KAAK,IAAI,EAAE,EAAE,CAAC,CAAC;YACzE,IAAI,GAAG,IAAI,MAAM,KAAK,GAAG,IAAI,cAAc,IAAI,IAAI;gBAAE,cAAc,GAAG,IAAI,CAAC;QAC7E,CAAC;aAAM,IAAI,CAAC,KAAK,aAAa,EAAE,CAAC;YAC/B,UAAU,GAAG,IAAI,CAAC;YAClB,QAAQ,GAAG,CAAC,CAAC,EAAE,CAAC,wBAAwB,CAAC;YACzC,MAAM,IAAI,GAAG,EAAE,CAAC,UAAU,IAAI,EAAE,CAAC;YACjC,IAAI,CAAC,GAAG,CAAC,CAAC;YACV,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;gBACvB,MAAM,MAAM,GAAG,MAAM,CAAC,GAAG,CAAC,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC;gBAC7C,aAAa,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,MAAM,CAAC,GAAG,CAAC,aAAa,CAAC,IAAI,EAAE,CAAC,EAAE,MAAM,EAAE,GAAG,EAAE,GAAG,CAAC,GAAG,IAAI,EAAE,EAAE,KAAK,EAAE,GAAG,CAAC,KAAK,IAAI,EAAE,EAAE,CAAC,CAAC;gBACnH,IAAI,GAAG,IAAI,MAAM,KAAK,GAAG;oBAAE,UAAU,GAAG,IAAI,CAAC;YAC/C,CAAC;QACH,CAAC;aAAM,IAAI,CAAC,KAAK,OAAO,IAAI,CAAC,KAAK,cAAc,EAAE,CAAC;YACjD,YAAY,GAAG,IAAI,CAAC;YACpB,KAAK,MAAM,CAAC,IAAI,EAAE,CAAC,KAAK,IAAI,EAAE,EAAE,CAAC;gBAC/B,IAAI,CAAC,EAAE,GAAG;oBAAE,UAAU,CAAC,IAAI,CAAC,EAAE,GAAG,EAAE,CAAC,CAAC,GAAG,EAAE,KAAK,EAAE,CAAC,CAAC,KAAK,IAAI,EAAE,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;YAC3F,CAAC;QACH,CAAC;IACH,CAAC;IAED,gGAAgG;IAChG,MAAM,YAAY,GAAG,cAAc,IAAI,IAAI;QACzC,CAAC,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC;QACrB,CAAC,CAAC,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,IAAI,GAAG,cAAe,CAAC,CAAC;IAElD,OAAO,EAAE,SAAS,EAAE,cAAc,EAAE,UAAU,EAAE,UAAU,EAAE,QAAQ,EAAE,aAAa,EAAE,YAAY,EAAE,YAAY,EAAE,UAAU,EAAE,CAAC;AAChI,CAAC;AASD,MAAM,UAAU,YAAY,CAAC,CAAkB;IAC7C,MAAM,aAAa,GAAG,CAAC,CAAC,cAAc,IAAI,IAAI,IAAI,CAAC,CAAC,cAAc,IAAI,CAAC,CAAC;IACxE,MAAM,KAAK,GAAG,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,4EAA4E,CAAC,CAAC,CAAC,EAAE,CAAC;IACjH,IAAI,CAAC,CAAC,CAAC,UAAU,EAAE,CAAC;QAClB,OAAO,aAAa;YAClB,CAAC,CAAC,EAAE,OAAO,EAAE,mBAAmB,EAAE,IAAI,EAAE,kDAAkD,CAAC,CAAC,cAAc,uBAAuB,KAAK,EAAE,EAAE;YAC1I,CAAC,CAAC,EAAE,OAAO,EAAE,iBAAiB,EAAE,IAAI,EAAE,oCAAoC,CAAC,CAAC,cAAc,IAAI,qBAAqB,+BAA+B,KAAK,EAAE,EAAE,CAAC;IAChK,CAAC;IACD,IAAI,aAAa,IAAI,CAAC,CAAC,UAAU,EAAE,CAAC;QAClC,OAAO,EAAE,OAAO,EAAE,mBAAmB,EAAE,IAAI,EAAE,qBAAqB,CAAC,CAAC,cAAc,+EAA+E,KAAK,EAAE,EAAE,CAAC;IAC7K,CAAC;IACD,IAAI,aAAa,IAAI,CAAC,CAAC,CAAC,UAAU,EAAE,CAAC;QACnC,OAAO,EAAE,OAAO,EAAE,uBAAuB,EAAE,IAAI,EAAE,qBAAqB,CAAC,CAAC,cAAc,oJAAoJ,KAAK,EAAE,EAAE,CAAC;IACtP,CAAC;IACD,IAAI,CAAC,aAAa,IAAI,CAAC,CAAC,UAAU,EAAE,CAAC;QACnC,OAAO,EAAE,OAAO,EAAE,uBAAuB,EAAE,IAAI,EAAE,4DAA4D,CAAC,CAAC,cAAc,IAAI,QAAQ,0FAA0F,KAAK,EAAE,EAAE,CAAC;IAC/O,CAAC;IACD,OAAO,EAAE,OAAO,EAAE,iBAAiB,EAAE,IAAI,EAAE,oDAAoD,CAAC,CAAC,cAAc,IAAI,QAAQ,sDAAsD,KAAK,EAAE,EAAE,CAAC;AAC7L,CAAC"}
|
package/dist/server.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"server.d.ts","sourceRoot":"","sources":["../src/server.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,MAAM,yCAAyC,CAAC;
|
|
1
|
+
{"version":3,"file":"server.d.ts","sourceRoot":"","sources":["../src/server.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,MAAM,yCAAyC,CAAC;AA2EpE,wBAAgB,OAAO,IAAI,MAAM,CAGhC;AA8KD,wBAAgB,YAAY,IAAI;IAAE,MAAM,EAAE,SAAS,CAAC;IAAC,GAAG,EAAE,MAAM,OAAO,CAAC,IAAI,CAAC,CAAA;CAAE,CAw0D9E"}
|
package/dist/server.js
CHANGED
|
@@ -11,6 +11,12 @@ import { getDashboardData } from './core/dashboardData.js';
|
|
|
11
11
|
import { startDashboardServer, stopDashboardServer, dashboardServerUrl, listLocalProperties } from './core/webServer.js';
|
|
12
12
|
import { computeSerpFootprint, persistSerpFootprint } from './core/serpFootprint.js';
|
|
13
13
|
import { computeMarketSizing, persistMarketSizing } from './core/marketSizing.js';
|
|
14
|
+
import { FirecrawlClient } from './core/FirecrawlClient.js';
|
|
15
|
+
import { SupadataClient } from './core/SupadataClient.js';
|
|
16
|
+
import { fetchCompetitorContent } from './core/reconResearch.js';
|
|
17
|
+
import { fetchOwnPage } from './core/reconFetch.js';
|
|
18
|
+
import { parseSerpForRecon, reconVerdict } from './core/serpRecon.js';
|
|
19
|
+
import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos } from './audit/recon.js';
|
|
14
20
|
/** A clickable browser-dashboard link appended to tool outputs — the user should always
|
|
15
21
|
* know the full interactive report is one click away (or one serve_dashboard call away). */
|
|
16
22
|
function browserLink(siteUrl) {
|
|
@@ -165,6 +171,7 @@ Raw access: query_audit runs any single check with full evidence; every table ab
|
|
|
165
171
|
- **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
|
|
166
172
|
- **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
|
|
167
173
|
- **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
|
|
174
|
+
- **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
|
|
168
175
|
- **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
|
|
169
176
|
|
|
170
177
|
Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
|
|
@@ -246,6 +253,12 @@ export function createServer() {
|
|
|
246
253
|
: null;
|
|
247
254
|
const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
|
|
248
255
|
const backlinks = dfs ? new Backlinks(dfs, dataDir()) : null;
|
|
256
|
+
// Firecrawl (competitor-page scraping for content recon) — optional; degrades gracefully.
|
|
257
|
+
const firecrawlKey = process.env.FIRECRAWL_API_KEY;
|
|
258
|
+
const firecrawl = firecrawlKey ? new FirecrawlClient(firecrawlKey, path.join(dataDir(), 'firecrawl-cache.db')) : null;
|
|
259
|
+
// Supadata (transcribes the ranking videos for content recon) — optional; degrades gracefully.
|
|
260
|
+
const supadataKey = process.env.SUPADATA_API_KEY;
|
|
261
|
+
const supadata = supadataKey ? new SupadataClient(supadataKey, path.join(dataDir(), 'supadata-cache.db')) : null;
|
|
249
262
|
const entities = new Entities(new WikidataClient(path.join(dataDir(), 'wikidata-cache.db')), dataDir());
|
|
250
263
|
const refresh = new Refresh(sync, crawler, inspector, rankTracker);
|
|
251
264
|
const requireGsc = (v) => {
|
|
@@ -874,6 +887,223 @@ export function createServer() {
|
|
|
874
887
|
},
|
|
875
888
|
};
|
|
876
889
|
});
|
|
890
|
+
// ── Content recon (recon_targets) — the data-intensive "why are we losing, what to do" mission ──
|
|
891
|
+
server.registerTool('recon_targets', {
|
|
892
|
+
title: 'Content recon: why a page is losing, and what to do about it',
|
|
893
|
+
description: '[Paid: DataForSEO SERP per page (~$0.004 each), bounded to the batch | Use for: the deep "why are we behind and what to add" recon] For each of your worst declining / striking-distance pages (auto-selected by impressions x decline, position 3-15; or pass explicit urls), this fetches OUR live page with the crawler, pulls the live Google SERP (DataForSEO SERP-advanced), and classifies WHY we are behind using the organic-rank x AI-Overview-citation matrix: defend-and-deepen (cited + strong), accuracy-or-freshness (rank but the AIO will not quote us - the sharpest, most actionable class), consolidate-weak-page, or competitive-gap. It writes a per-page classification plus deterministic to-dos (schema gaps, freshness, cannibalisation, video format) into a trackable ledger, and returns the competitor set (organic-above + AI-Overview references + ranking videos) for the research session to diff. Set scrapeCompetitors:true to also pull the top competitors as content, routed by source: YouTube/video → Supadata transcript (SUPADATA_API_KEY), Reddit → its .json, other pages → Firecrawl (FIRECRAWL_API_KEY) with a free HTTP fallback. Cloudflare-challenge sites (e.g. PCMag) still can\'t be fetched from a server and come back as a per-URL error — the SERP still tells you they rank; if one matters, ask the user to paste its copy or supply a text file and diff that in. Transcribe the videos (usually what wins these SERPs) and use the reachable pages. Then write findings back with save_recon_todo; track with recon_todos.',
|
|
894
|
+
inputSchema: {
|
|
895
|
+
siteUrl: z.string(),
|
|
896
|
+
limit: z.number().int().min(1).max(15).optional().describe('Pages per batch (default 5)'),
|
|
897
|
+
minImpressions: z.number().int().min(1).optional(),
|
|
898
|
+
location: z.union([z.string(), z.number()]).optional(),
|
|
899
|
+
urls: z.array(z.string()).optional().describe('Analyse these exact pages instead of auto-selecting'),
|
|
900
|
+
scrapeCompetitors: z.boolean().optional().describe('Also fetch the top competitors (video→transcript, pages→markdown/HTML)'),
|
|
901
|
+
crawlAs: z.enum(['browser', 'googlebot']).optional().describe('UA for the free HTTP fetch: browser (default, mimics a visit from Google — gets Reddit + mid-tier) or googlebot'),
|
|
902
|
+
},
|
|
903
|
+
}, async ({ siteUrl, limit, minImpressions, location, urls, scrapeCompetitors, crawlAs }) => {
|
|
904
|
+
const client = requireDfs(dfs);
|
|
905
|
+
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
906
|
+
try {
|
|
907
|
+
const fresh = gscFreshness(db.db);
|
|
908
|
+
if (!fresh.effectiveMax)
|
|
909
|
+
return { content: [{ type: 'text', text: `No synced GSC data for ${siteUrl} — run refresh_property first.` }], structuredContent: { error: 'empty', siteUrl } };
|
|
910
|
+
const today = new Date().toISOString().slice(0, 10);
|
|
911
|
+
const ownDomain = dfsHost(siteUrl);
|
|
912
|
+
const hostForm = hostFormForProperty(siteUrl) ?? 'asis';
|
|
913
|
+
let targets;
|
|
914
|
+
if (urls?.length) {
|
|
915
|
+
targets = [];
|
|
916
|
+
for (const u of urls) {
|
|
917
|
+
const key = urlKey(u, { hostForm });
|
|
918
|
+
const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
|
|
919
|
+
FROM search_analytics WHERE page_key=? AND query IS NOT NULL AND date > date(?, '-28 days') AND date <= ? GROUP BY query ORDER BY imp DESC LIMIT 1`)
|
|
920
|
+
.get(key, fresh.effectiveMax, fresh.effectiveMax);
|
|
921
|
+
if (row?.query)
|
|
922
|
+
targets.push({ urlKey: key, query: row.query, impressions: row.imp, clicks: row.clk, position: Math.round(row.pos * 10) / 10, priorPosition: null, slipped: false, competingUrls: 0 });
|
|
923
|
+
}
|
|
924
|
+
}
|
|
925
|
+
else {
|
|
926
|
+
targets = selectReconTargets(db.db, { ...(limit != null ? { limit } : {}), ...(minImpressions != null ? { minImpressions } : {}), maxDate: fresh.effectiveMax });
|
|
927
|
+
}
|
|
928
|
+
if (!targets.length)
|
|
929
|
+
return { content: [{ type: 'text', text: `No recon targets found (declining/striking-distance pages, position 3-15, ${minImpressions ?? 300}+ impressions). Try a lower minImpressions or pass explicit urls.` }], structuredContent: { siteUrl, targets: [] } };
|
|
930
|
+
const results = [];
|
|
931
|
+
let cost = 0;
|
|
932
|
+
for (const t of targets) {
|
|
933
|
+
let own = null;
|
|
934
|
+
try {
|
|
935
|
+
own = await fetchOwnPage(t.urlKey, hostForm === 'asis' ? 'asis' : hostForm);
|
|
936
|
+
}
|
|
937
|
+
catch { /* page unreachable — classify on SERP alone */ }
|
|
938
|
+
const serpResp = await client.serpOrganic(t.query, location, 'en', 20);
|
|
939
|
+
cost += serpResp.cost;
|
|
940
|
+
const serp = parseSerpForRecon(serpResp, ownDomain);
|
|
941
|
+
const verdict = reconVerdict(serp);
|
|
942
|
+
const baseline = { organicRank: serp.ourOrganicRank, aioCitesUs: serp.aioCitesUs, gscPosition: t.position, gscImpressions: t.impressions, at: today };
|
|
943
|
+
persistReconPage(db.db, {
|
|
944
|
+
urlKey: t.urlKey, query: t.query, verdict: verdict.verdict, verdictNote: verdict.note,
|
|
945
|
+
organicRank: serp.ourOrganicRank, gscPosition: t.position, gscImpressions: t.impressions,
|
|
946
|
+
aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, videoPresent: serp.videoPresent,
|
|
947
|
+
hasProductSchema: own?.hasProductOrReview ?? false, schemaTypes: own?.jsonLdTypes ?? [],
|
|
948
|
+
competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
|
|
949
|
+
});
|
|
950
|
+
const todos = own ? deterministicTodos(t, own, serp, verdict, today) : [];
|
|
951
|
+
const inserted = todos.length ? insertTodos(db.db, t.urlKey, t.query, todos, baseline, 'auto') : 0;
|
|
952
|
+
let competitorContent = undefined;
|
|
953
|
+
if (scrapeCompetitors && (firecrawl || supadata)) {
|
|
954
|
+
// Build a routed candidate set: organic-above + a couple of AI-Overview references +
|
|
955
|
+
// one ranking video, deduped, our own domain removed. Each URL routes by domain
|
|
956
|
+
// (YouTube→supadata, Reddit→.json, else→firecrawl). Bounded to keep the payload sane.
|
|
957
|
+
const seen = new Set();
|
|
958
|
+
const candidates = [];
|
|
959
|
+
const add = (u) => { if (u && !seen.has(u) && dfsHost(u) !== ownDomain) {
|
|
960
|
+
seen.add(u);
|
|
961
|
+
candidates.push(u);
|
|
962
|
+
} };
|
|
963
|
+
serp.organicAbove.forEach(o => add(o.url));
|
|
964
|
+
serp.aioReferences.slice(0, 4).forEach(r => add(r.url));
|
|
965
|
+
serp.videoItems.slice(0, 1).forEach(v => add(v.url));
|
|
966
|
+
competitorContent = [];
|
|
967
|
+
for (const url of candidates.slice(0, 3)) {
|
|
968
|
+
competitorContent.push(await fetchCompetitorContent(url, { firecrawl, supadata }, { maxChars: 4000, ua: crawlAs }));
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
results.push({
|
|
972
|
+
urlKey: t.urlKey, query: t.query, impressions: t.impressions, gscPosition: t.position, priorPosition: t.priorPosition, slipped: t.slipped,
|
|
973
|
+
organicRank: serp.ourOrganicRank, aioPresent: serp.aioPresent, aioCitesUs: serp.aioCitesUs, videoPresent: serp.videoPresent,
|
|
974
|
+
verdict: verdict.verdict, verdictNote: verdict.note,
|
|
975
|
+
ownHeadings: own?.headings.map(h => h.heading) ?? null, schemaTypes: own?.jsonLdTypes ?? null, hasProductSchema: own?.hasProductOrReview ?? null, dateModified: own?.dateModified ?? null,
|
|
976
|
+
todos, todosInserted: inserted,
|
|
977
|
+
competitors: { organicAbove: serp.organicAbove, aioReferences: serp.aioReferences, videoItems: serp.videoItems },
|
|
978
|
+
...(competitorContent ? { competitorContent } : {}),
|
|
979
|
+
});
|
|
980
|
+
}
|
|
981
|
+
const md = `# Content recon — ${siteUrl}\n\n${results.length} page(s), ${results.reduce((s, r) => s + r.todosInserted, 0)} to-dos saved. DataForSEO SERP cost $${cost.toFixed(4)}.\n\n` +
|
|
982
|
+
results.map(r => {
|
|
983
|
+
const flags = [r.aioPresent ? (r.aioCitesUs ? 'AIO: cited' : 'AIO: NOT cited') : 'no AIO', r.videoPresent ? 'video pack' : null].filter(Boolean).join(' · ');
|
|
984
|
+
return `## ${r.urlKey}\n"${r.query}" — organic #${r.organicRank ?? '?'} (GSC avg ${r.gscPosition}${r.slipped ? `, slipped from ${r.priorPosition}` : ''}), ${r.impressions} impr. ${flags}\n**${r.verdict}** — ${r.verdictNote}\n` +
|
|
985
|
+
(r.todos.length ? '\nTo do:\n' + r.todos.map((t) => `- [${t.type}] ${t.action}`).join('\n') : '');
|
|
986
|
+
}).join('\n\n') +
|
|
987
|
+
(() => {
|
|
988
|
+
const errs = results.flatMap(r => (r.competitorContent ?? []).filter((c) => c.error));
|
|
989
|
+
if (!errs.length)
|
|
990
|
+
return '';
|
|
991
|
+
const needKey = errs.filter(c => /not set/i.test(c.error));
|
|
992
|
+
const blocked = errs.filter(c => !/not set/i.test(c.error));
|
|
993
|
+
let note = '';
|
|
994
|
+
if (blocked.length)
|
|
995
|
+
note += `\n\n${blocked.length} competitor(s) couldn't be fetched (bot-protection like Cloudflare, a block, or a fetch error): ${blocked.slice(0, 5).map(c => c.url).join(', ')}. If one matters, paste its copy here or drop a text file and I'll diff it in.`;
|
|
996
|
+
if (needKey.length)
|
|
997
|
+
note += `\n\n${needKey.length} competitor(s) skipped for a missing API key: ${[...new Set(needKey.map(c => c.error))].join('; ')}.`;
|
|
998
|
+
return note;
|
|
999
|
+
})() +
|
|
1000
|
+
`\n\nNext: research the competitors (transcribe the videos, read the reachable pages), write gaps back with save_recon_todo, and track with recon_todos.` + browserLink(siteUrl);
|
|
1001
|
+
return { content: [{ type: 'text', text: md }], structuredContent: { siteUrl, cost, targets: results } };
|
|
1002
|
+
}
|
|
1003
|
+
finally {
|
|
1004
|
+
db.close();
|
|
1005
|
+
}
|
|
1006
|
+
});
|
|
1007
|
+
// save_recon_todo — the research session writes its content-gap / originality findings back
|
|
1008
|
+
// into the ledger against a page (source: 'research').
|
|
1009
|
+
server.registerTool('save_recon_todo', {
|
|
1010
|
+
title: 'Save content-recon to-dos (research writeback)',
|
|
1011
|
+
description: 'Write content-recon findings back into the trackable ledger for a page — the gaps and originality the research session found by diffing competitors (firecrawl) and videos (supadata) against our content. Each to-do is an action with a type (content-gap / originality / schema / freshness / format), rationale and evidence. Snapshots the page baseline so the fix\'s effect on rank/AIO-citation is measurable later. Run recon_targets first (it classifies the page and seeds the deterministic to-dos); this adds the judgement ones. Track everything with recon_todos.',
|
|
1012
|
+
inputSchema: {
|
|
1013
|
+
siteUrl: z.string(),
|
|
1014
|
+
urlKey: z.string().describe('The page (any URL form — normalised to its key)'),
|
|
1015
|
+
todos: z.array(z.object({
|
|
1016
|
+
action: z.string(),
|
|
1017
|
+
type: z.string().optional(),
|
|
1018
|
+
rationale: z.string().optional(),
|
|
1019
|
+
evidence: z.record(z.any()).optional(),
|
|
1020
|
+
priority: z.number().optional(),
|
|
1021
|
+
})).min(1),
|
|
1022
|
+
},
|
|
1023
|
+
}, async ({ siteUrl, urlKey: rawUrl, todos }) => {
|
|
1024
|
+
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
1025
|
+
try {
|
|
1026
|
+
const key = urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' });
|
|
1027
|
+
const page = db.db.prepare(`SELECT query, organic_rank, aio_cites_us, gsc_position FROM recon_page WHERE url_key=?`).get(key);
|
|
1028
|
+
const baseline = page ? { organicRank: page.organic_rank, aioCitesUs: !!page.aio_cites_us, gscPosition: page.gsc_position, at: new Date().toISOString().slice(0, 10) } : {};
|
|
1029
|
+
const drafts = todos.map(t => ({ action: t.action, type: t.type ?? 'content-gap', rationale: t.rationale ?? '', evidence: t.evidence ?? {}, priority: t.priority ?? 0 }));
|
|
1030
|
+
const n = insertTodos(db.db, key, page?.query ?? '', drafts, baseline, 'research');
|
|
1031
|
+
return { content: [{ type: 'text', text: `Saved ${n} recon to-do(s) for ${key}${n < drafts.length ? ` (${drafts.length - n} already open)` : ''}. Track with recon_todos.` }], structuredContent: { urlKey: key, inserted: n } };
|
|
1032
|
+
}
|
|
1033
|
+
finally {
|
|
1034
|
+
db.close();
|
|
1035
|
+
}
|
|
1036
|
+
});
|
|
1037
|
+
// recon_todos — list, track and annotate the ledger. No id → list (optionally filtered);
|
|
1038
|
+
// id → update status and/or append a dated annotation, and optionally re-measure the outcome.
|
|
1039
|
+
server.registerTool('recon_todos', {
|
|
1040
|
+
title: 'List, track and annotate content-recon to-dos',
|
|
1041
|
+
description: 'The content-recon to-do board. With no id: list to-dos (optionally filter by page or status), grouped by page with each page\'s verdict — the pick-a-page-to-work-on surface, and the hand-off to content-machine. With id: update one to-do — set status (open → researching → drafted → shipped → dismissed) and/or append a dated annotation note (your own observations, the history). On status:shipped with remeasure:true it re-fetches the SERP and records the outcome, so you can see whether the fix moved you from AIO-uncited to cited, or up the organic ranks.',
|
|
1042
|
+
inputSchema: {
|
|
1043
|
+
siteUrl: z.string(),
|
|
1044
|
+
urlKey: z.string().optional().describe('Filter the list to one page'),
|
|
1045
|
+
status: z.enum(['open', 'researching', 'drafted', 'shipped', 'dismissed']).optional().describe('Filter the list, or the new status when id is set'),
|
|
1046
|
+
id: z.number().int().optional().describe('Update this to-do'),
|
|
1047
|
+
note: z.string().optional().describe('Append a dated annotation to this to-do'),
|
|
1048
|
+
remeasure: z.boolean().optional().describe('On status:shipped, re-fetch the SERP and record the outcome (paid)'),
|
|
1049
|
+
location: z.union([z.string(), z.number()]).optional(),
|
|
1050
|
+
},
|
|
1051
|
+
}, async ({ siteUrl, urlKey: rawUrl, status, id, note, remeasure, location }) => {
|
|
1052
|
+
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
1053
|
+
try {
|
|
1054
|
+
if (id != null) {
|
|
1055
|
+
const row = db.db.prepare(`SELECT * FROM recon_todo WHERE id=?`).get(id);
|
|
1056
|
+
if (!row)
|
|
1057
|
+
throw new Error(`No recon to-do #${id}.`);
|
|
1058
|
+
let notes = row.notes ?? '';
|
|
1059
|
+
if (note)
|
|
1060
|
+
notes = (notes ? notes + '\n' : '') + `[${new Date().toISOString().slice(0, 10)}] ${note}`;
|
|
1061
|
+
let outcome = row.outcome;
|
|
1062
|
+
let outcomeNote = '';
|
|
1063
|
+
if (status === 'shipped' && remeasure && dfs) {
|
|
1064
|
+
const serpResp = await dfs.serpOrganic(row.query, location, 'en', 20);
|
|
1065
|
+
const serp = parseSerpForRecon(serpResp, dfsHost(siteUrl));
|
|
1066
|
+
outcome = JSON.stringify({ organicRank: serp.ourOrganicRank, aioCitesUs: serp.aioCitesUs, at: new Date().toISOString().slice(0, 10) });
|
|
1067
|
+
const base = row.baseline ? JSON.parse(row.baseline) : {};
|
|
1068
|
+
outcomeNote = ` Outcome: organic ${base.organicRank ?? '?'}→${serp.ourOrganicRank ?? '?'}, AIO cited ${base.aioCitesUs ? 'yes' : 'no'}→${serp.aioCitesUs ? 'yes' : 'no'}.`;
|
|
1069
|
+
}
|
|
1070
|
+
db.db.prepare(`UPDATE recon_todo SET status=COALESCE(?,status), notes=?, outcome=COALESCE(?,outcome), updated_at=datetime('now') WHERE id=?`)
|
|
1071
|
+
.run(status ?? null, notes, outcome ?? null, id);
|
|
1072
|
+
return { content: [{ type: 'text', text: `Updated to-do #${id}${status ? ` → ${status}` : ''}${note ? ' (note added)' : ''}.${outcomeNote}` }], structuredContent: { id, status: status ?? row.status, outcome: outcome ? JSON.parse(outcome) : null } };
|
|
1073
|
+
}
|
|
1074
|
+
const key = rawUrl ? urlKey(rawUrl, { hostForm: hostFormForProperty(siteUrl) ?? 'asis' }) : null;
|
|
1075
|
+
const where = [];
|
|
1076
|
+
const args = [];
|
|
1077
|
+
if (key) {
|
|
1078
|
+
where.push('url_key=?');
|
|
1079
|
+
args.push(key);
|
|
1080
|
+
}
|
|
1081
|
+
if (status) {
|
|
1082
|
+
where.push('status=?');
|
|
1083
|
+
args.push(status);
|
|
1084
|
+
}
|
|
1085
|
+
const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
|
|
1086
|
+
FROM recon_todo ${where.length ? 'WHERE ' + where.join(' AND ') : ''} ORDER BY url_key, priority DESC`).all(...args);
|
|
1087
|
+
if (!rows.length)
|
|
1088
|
+
return { content: [{ type: 'text', text: `No recon to-dos${key ? ` for ${key}` : ''}${status ? ` with status ${status}` : ''}. Run recon_targets to generate some.` }], structuredContent: { todos: [] } };
|
|
1089
|
+
const byPage = new Map();
|
|
1090
|
+
for (const r of rows)
|
|
1091
|
+
(byPage.get(r.url_key) ?? byPage.set(r.url_key, []).get(r.url_key)).push(r);
|
|
1092
|
+
const verdictStmt = db.db.prepare(`SELECT verdict, verdict_note FROM recon_page WHERE url_key=?`);
|
|
1093
|
+
const md = `# Content-recon to-dos — ${rows.length} item(s)${status ? `, status ${status}` : ''}\n\n` +
|
|
1094
|
+
[...byPage.entries()].map(([url, items]) => {
|
|
1095
|
+
const v = verdictStmt.get(url);
|
|
1096
|
+
return `## ${url}${v ? `\n_${v.verdict}_ — ${v.verdict_note}` : ''}\n\n| # | Status | Type | Action |\n|---|---|---|---|\n` +
|
|
1097
|
+
items.map(i => `| ${i.id} | ${i.status} | ${i.type ?? ''} | ${String(i.action).replace(/\|/g, '\\|')} |`).join('\n') +
|
|
1098
|
+
(items.some(i => i.notes) ? '\n\nNotes:\n' + items.filter(i => i.notes).map(i => `- #${i.id}: ${String(i.notes).replace(/\n/g, ' / ')}`).join('\n') : '');
|
|
1099
|
+
}).join('\n\n') +
|
|
1100
|
+
browserLink(siteUrl);
|
|
1101
|
+
return { content: [{ type: 'text', text: md }], structuredContent: { todos: rows } };
|
|
1102
|
+
}
|
|
1103
|
+
finally {
|
|
1104
|
+
db.close();
|
|
1105
|
+
}
|
|
1106
|
+
});
|
|
877
1107
|
// keyword_list — demand-first clustering ("list mode"): a keyword list becomes topics
|
|
878
1108
|
// with own/weak/absent verdicts, clustered by the URL Google already answers them with.
|
|
879
1109
|
server.registerTool('keyword_list', {
|