@houtini/seo-audit-console 0.5.1 → 0.7.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"server.d.ts","sourceRoot":"","sources":["../src/server.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,MAAM,yCAAyC,CAAC;AA6EpE,wBAAgB,OAAO,IAAI,MAAM,CAGhC;AAiLD,wBAAgB,YAAY,IAAI;IAAE,MAAM,EAAE,SAAS,CAAC;IAAC,GAAG,EAAE,MAAM,OAAO,CAAC,IAAI,CAAC,CAAA;CAAE,CA6hE9E"}
1
+ {"version":3,"file":"server.d.ts","sourceRoot":"","sources":["../src/server.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAAE,MAAM,yCAAyC,CAAC;AAiFpE,wBAAgB,OAAO,IAAI,MAAM,CAGhC;AAkLD,wBAAgB,YAAY,IAAI;IAAE,MAAM,EAAE,SAAS,CAAC;IAAC,GAAG,EAAE,MAAM,OAAO,CAAC,IAAI,CAAC,CAAA;CAAE,CAiqE9E"}
package/dist/server.js CHANGED
@@ -22,8 +22,11 @@ import { selectReconTargets, deterministicTodos, persistReconPage, insertTodos,
22
22
  function browserLink(siteUrl) {
23
23
  const base = dashboardServerUrl();
24
24
  if (base)
25
- return `\n\nBrowser dashboard: ${base}/dashboard${siteUrl ? `?siteUrl=${encodeURIComponent(siteUrl)}` : ''}`;
26
- return `\n\nTip: run serve_dashboard to open the full interactive dashboard in your browser.`;
25
+ return `\n\nšŸ“Š Browser dashboard: ${base}/dashboard${siteUrl ? `?siteUrl=${encodeURIComponent(siteUrl)}` : ''}`;
26
+ // No live server: point at the dashboard surfaces that always work. get_dashboard renders the
27
+ // interactive dashboard in chat (works through the Docker gateway where a served port would not);
28
+ // serve_dashboard opens a browser tab locally; export_report writes a shareable HTML file.
29
+ return `\n\nšŸ“Š See it in the dashboard — run get_dashboard${siteUrl ? ` for ${siteUrl}` : ''} (interactive, in chat), serve_dashboard (browser tab), or export_report (shareable HTML).`;
27
30
  }
28
31
  import { runAudit, runSingleCheck, listChecks } from './audit/engine.js';
29
32
  import { buildAuditMarkdown } from './audit/report.js';
@@ -50,6 +53,7 @@ import { RankTracker } from './core/RankTracker.js';
50
53
  import { Backlinks } from './core/Backlinks.js';
51
54
  import { LinkIntersect } from './core/LinkIntersect.js';
52
55
  import { MajesticClient } from './core/MajesticClient.js';
56
+ import { fetchGoogleNews } from './core/googleNews.js';
53
57
  import { WikidataClient } from './core/WikidataClient.js';
54
58
  import { Entities } from './core/Entities.js';
55
59
  import { JobManager } from './core/JobManager.js';
@@ -114,8 +118,9 @@ ARCHETYPE CHAINS (compose along these lines):
114
118
 
115
119
  COST DISCIPLINE (behave like a strategist who knows the margins):
116
120
  - Free and instant, use liberally: everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
117
- - Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
118
- - Paid per-keyword (SERP): related_terms and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
121
+ - Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, topic_trend, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
122
+ - Paid per-keyword (SERP): related_terms, youtube_discovery (ranking videos for a topic — pair with a transcript tool), news_discovery (recent coverage / freshness), and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
123
+ - Content research (what to write / what changed): topic_trend (is a topic rising/seasonal) → youtube_discovery + news_discovery (what the winning videos/articles cover) → draft_content.
119
124
  - Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
120
125
 
121
126
  AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
@@ -252,17 +257,18 @@ export function createServer() {
252
257
  const crawler = new Crawler(dataDir()); // no GSC credentials required
253
258
  const dfsUser = process.env.DATAFORSEO_USERNAME;
254
259
  const dfsPass = process.env.DATAFORSEO_PASSWORD;
255
- const dfsCacheDays = Number(process.env.DATAFORSEO_CACHE_DAYS) || 20;
260
+ const dfsCacheDays = Number(process.env.DATAFORSEO_CACHE_DAYS) || 7;
256
261
  const dfs = dfsUser && dfsPass
257
262
  ? new DataForSeoClient(dfsUser, dfsPass, path.join(dataDir(), 'dataforseo-cache.db'), dfsCacheDays)
258
263
  : null;
259
264
  const rankTracker = dfs ? new RankTracker(dfs, dataDir()) : null;
260
- const backlinks = dfs ? new Backlinks(dfs, dataDir()) : null;
261
265
  const linkIntersect = dfs ? new LinkIntersect(dfs, dataDir()) : null;
262
- // Majestic (Trust Flow / Topical Trust Flow) — optional link_intersect enrichment tier.
266
+ // Majestic (Trust Flow / Topical Trust Flow) — optional link_intersect + trapped-authority tier.
263
267
  const majesticKey = process.env.MAJESTIC_API_KEY;
264
- const majesticCacheDays = Number(process.env.MAJESTIC_CACHE_DAYS) || 20;
268
+ const majesticCacheDays = Number(process.env.MAJESTIC_CACHE_DAYS) || 30;
265
269
  const majestic = majesticKey ? new MajesticClient(majesticKey, path.join(dataDir(), 'majestic-cache.db'), majesticCacheDays) : null;
270
+ // Backlinks after Majestic, so pull_backlinks can enrich per-URL pages with Trust Flow.
271
+ const backlinks = dfs ? new Backlinks(dfs, dataDir(), majestic) : null;
266
272
  // Firecrawl (competitor-page scraping for content recon) — optional; degrades gracefully.
267
273
  const firecrawlKey = process.env.FIRECRAWL_API_KEY;
268
274
  const firecrawl = firecrawlKey ? new FirecrawlClient(firecrawlKey, path.join(dataDir(), 'firecrawl-cache.db')) : null;
@@ -281,6 +287,10 @@ export function createServer() {
281
287
  throw new Error('DATAFORSEO_USERNAME / DATAFORSEO_PASSWORD not set — required for DataForSEO.');
282
288
  return v;
283
289
  };
290
+ // Which optional integrations have a key set — drives the dashboard's key-gated tabs
291
+ // (Links, Content research) and their affiliate-linked upsell states. The data layer has
292
+ // no env access, so it's injected here onto every dashboard payload surface.
293
+ const apiKeysStatus = () => ({ dataforseo: !!dfs, majestic: !!majestic, firecrawl: !!firecrawl, supadata: !!supadata });
284
294
  // ── Introspection (no data / creds required) ────────────────────────────
285
295
  server.registerTool('seo_audit_help', {
286
296
  title: 'Help — what this audit can do',
@@ -845,6 +855,111 @@ export function createServer() {
845
855
  structuredContent: r,
846
856
  };
847
857
  });
858
+ server.registerTool('youtube_discovery', {
859
+ title: 'YouTube video discovery (DataForSEO SERP)',
860
+ description: '[Paid: SERP call PER KEYWORD, cached 20d | Use for: finding the videos that rank for a topic — content research, competitor video scan, freshness] The ranking YouTube videos for a keyword (DataForSEO YouTube organic SERP). Returns title, URL, channel, view count, publish date, duration and shorts/live flags, in rank order. Pair with a transcript tool (Supadata) to read what the winning videos actually say. SERP scope, 20-day cache. Default location: United States (2840).',
861
+ inputSchema: {
862
+ keyword: z.string(),
863
+ location: z.union([z.string(), z.number()]).optional(),
864
+ languageCode: z.string().optional(),
865
+ blockDepth: z.number().int().min(1).max(200).optional().describe('How many ranking videos to collect (default 20)'),
866
+ },
867
+ }, async ({ keyword, location, languageCode, blockDepth }) => {
868
+ const client = requireDfs(dfs);
869
+ const r = await client.serpYoutube(keyword, location, languageCode, blockDepth ?? 20);
870
+ const top = r.videos.slice(0, 15).map(v => `• ${v.title ?? '(untitled)'}${v.channel ? ` — ${v.channel}` : ''}${v.views != null ? `, ${v.views.toLocaleString()} views` : ''}${v.published ? `, ${v.published}` : ''}\n ${v.url ?? ''}`).join('\n');
871
+ return {
872
+ content: [{ type: 'text', text: `${r.videos.length} YouTube videos for "${keyword}"${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}${r.videos.length ? `:\n${top}` : ''}` }],
873
+ structuredContent: { keyword, videos: r.videos, cached: r.cached, cost: r.cost },
874
+ };
875
+ });
876
+ server.registerTool('news_discovery', {
877
+ title: 'News discovery (Google News RSS + DataForSEO)',
878
+ description: '[Google News is FREE (no key); DataForSEO adds a paid, richer Google News SERP | Use for: what has been PUBLISHED on a topic — freshness, "what changed since {date}", competitor coverage] Recent news for a keyword from Google News (the free RSS search feed — Google\'s old News API is retired) and/or the DataForSEO Google News SERP. Returns title, source, publish time, URL. Default source "both" merges + dedupes; pass source:"google" for a free keyless lookup, "dataforseo" for the paid SERP only. Default location: United States.',
879
+ inputSchema: {
880
+ keyword: z.string(),
881
+ location: z.union([z.string(), z.number()]).optional(),
882
+ languageCode: z.string().optional(),
883
+ depth: z.number().int().min(1).max(200).optional().describe('How many news results (default 20)'),
884
+ source: z.enum(['google', 'dataforseo', 'both']).optional().describe('google = free Google News RSS; dataforseo = paid Google News SERP; both (default) merges + dedupes'),
885
+ },
886
+ }, async ({ keyword, location, languageCode, depth, source }) => {
887
+ const src = source ?? 'both';
888
+ const n = depth ?? 20;
889
+ const seen = new Set();
890
+ const articles = [];
891
+ const add = (a) => { const k = String(a.url || a.title || '').toLowerCase(); if (k && !seen.has(k)) {
892
+ seen.add(k);
893
+ articles.push(a);
894
+ } };
895
+ let cost = 0, cached = true, googleCount = 0, dfsCount = 0, googleError = null;
896
+ // Free Google News RSS.
897
+ if (src === 'google' || src === 'both') {
898
+ try {
899
+ const g = await fetchGoogleNews(keyword, { limit: n });
900
+ googleCount = g.articles.length;
901
+ for (const a of g.articles)
902
+ add({ title: a.title, source: a.source, timestamp: a.timestamp, url: a.url, via: 'google-news' });
903
+ }
904
+ catch (e) {
905
+ googleError = e instanceof Error ? e.message : 'failed';
906
+ }
907
+ }
908
+ // Paid DataForSEO Google News SERP (only when explicitly asked, or 'both' AND a key is set).
909
+ if (src === 'dataforseo' || (src === 'both' && !!dfs)) {
910
+ const r = await requireDfs(dfs).serpNews(keyword, location, languageCode, n);
911
+ cost += r.cost;
912
+ cached = cached && r.cached;
913
+ dfsCount = r.articles.length;
914
+ for (const a of r.articles)
915
+ add({ ...a, via: 'dataforseo' });
916
+ }
917
+ // Top sources: which publishers are covering this topic, ranked by article count — the
918
+ // "who is talking about X" read. Powers "trending news + sources for X" in one call.
919
+ const sourceCount = new Map();
920
+ for (const a of articles) {
921
+ const s = String(a.source ?? '').trim();
922
+ if (s)
923
+ sourceCount.set(s, (sourceCount.get(s) ?? 0) + 1);
924
+ }
925
+ const topSources = [...sourceCount.entries()].sort((x, y) => y[1] - x[1]).slice(0, 12).map(([source, count]) => ({ source, count }));
926
+ const top = articles.slice(0, 15).map(a => `• ${a.title ?? '(untitled)'}${a.source ? ` — ${a.source}` : ''}${a.timestamp ? `, ${a.timestamp}` : ''}\n ${a.url ?? ''}`).join('\n');
927
+ const srcNote = [googleCount ? `${googleCount} Google News` : '', dfsCount ? `${dfsCount} DataForSEO` : ''].filter(Boolean).join(' + ') || 'no results';
928
+ const sourcesLine = topSources.length ? `\n\nTop sources: ${topSources.map(s => `${s.source} (${s.count})`).join(', ')}` : '';
929
+ return {
930
+ content: [{ type: 'text', text: `${articles.length} news results for "${keyword}" (${srcNote}${cost ? `, $${cost.toFixed(4)}` : ', free'})${googleError ? ` [Google News error: ${googleError}]` : ''}${sourcesLine}${articles.length ? `\n\n${top}` : ''}\n\nCompose with topic_trend for direction (rising/falling); schedule this call daily/hourly for a topic radar.` }],
931
+ structuredContent: { keyword, articles, topSources, cached, cost, sources: { googleNews: googleCount, dataforseo: dfsCount }, googleError },
932
+ };
933
+ });
934
+ server.registerTool('topic_trend', {
935
+ title: 'Google Trends interest over time (DataForSEO)',
936
+ description: '[Paid: Keywords API, cheap, cached 20d | Use for: seasonality + "is this rising or fading" — should we update now, when to publish] Google Trends relative interest (0–100) over time for up to 5 keywords (DataForSEO KEYWORDS_DATA / google_trends). Returns a per-keyword time series plus a rising/falling/flat read, so you can see direction and seasonality. timeRange: past_7_days | past_30_days | past_90_days | past_12_months (default) | past_5_years | 2004_present. type: web (default) | news | youtube | images. Default location: United States.',
937
+ inputSchema: {
938
+ keywords: z.array(z.string()).min(1).max(5),
939
+ location: z.union([z.string(), z.number()]).optional(),
940
+ languageCode: z.string().optional(),
941
+ timeRange: z.enum(['past_7_days', 'past_30_days', 'past_90_days', 'past_12_months', 'past_5_years', '2004_present']).optional(),
942
+ type: z.enum(['web', 'news', 'youtube', 'images', 'froogle']).optional(),
943
+ },
944
+ }, async ({ keywords, location, languageCode, timeRange, type }) => {
945
+ const client = requireDfs(dfs);
946
+ const r = await client.googleTrends(keywords, location, languageCode, { ...(timeRange ? { timeRange } : {}), ...(type ? { type } : {}) });
947
+ const summarise = (k) => {
948
+ const vals = r.series.map(s => s.values[k]).filter((v) => typeof v === 'number');
949
+ if (!vals.length)
950
+ return `${k}: no data`;
951
+ const win = Math.max(1, Math.floor(vals.length / 4));
952
+ const avg = (xs) => xs.reduce((a, b) => a + b, 0) / xs.length;
953
+ const first = avg(vals.slice(0, win)), last = avg(vals.slice(-win));
954
+ const dir = last > first * 1.1 ? 'rising' : last < first * 0.9 ? 'falling' : 'flat';
955
+ return `${k}: ${dir} (${Math.round(first)}→${Math.round(last)}, peak ${Math.max(...vals)})`;
956
+ };
957
+ const txt = r.keywords.map(summarise).join('\n');
958
+ return {
959
+ content: [{ type: 'text', text: `Trend, ${r.series.length} points${r.cached ? ' (cached)' : ` (live, $${r.cost.toFixed(4)})`}:\n${txt}` }],
960
+ structuredContent: { keywords: r.keywords, series: r.series, cached: r.cached, cost: r.cost },
961
+ };
962
+ });
848
963
  server.registerTool('suggest_pages', {
849
964
  title: 'Suggest new pages from real demand (gaps)',
850
965
  description: 'Propose NEW pages grounded in real Search Console demand: queries you already get impressions for but rank 11+ (no winning page), after subtracting demand you already satisfy (you rank ≤10, or a page already covers the query, or one URL owns it). Survivors are clustered into one proposed page per intent and scored by impressions Ɨ intent. Returns evidence + the nearest existing page to link the new page from. Computable from synced GSC + crawl — no paid calls (run search_intent siteUrl:<property> first to weight by intent).',
@@ -1871,7 +1986,7 @@ export function createServer() {
1871
1986
  inputSchema: { siteUrl: z.string() },
1872
1987
  _meta: { ui: { visibility: ['app'] } },
1873
1988
  }, async ({ siteUrl }) => {
1874
- const data = getDashboardData(dataDir(), siteUrl);
1989
+ const data = { ...getDashboardData(dataDir(), siteUrl), apiKeys: apiKeysStatus() };
1875
1990
  return { content: [{ type: 'text', text: 'ok' }], structuredContent: data };
1876
1991
  });
1877
1992
  // export_report — the dependable deliverable: a self-contained interactive dashboard
@@ -1882,7 +1997,7 @@ export function createServer() {
1882
1997
  description: 'Write a self-contained, interactive dashboard HTML for a property (all data + charts inlined) to the reports folder, and return the file path. Open it in any browser or send it to a client — no server, no MCP-App host support needed. Run refresh_property (+ run_audit for findings) first.',
1883
1998
  inputSchema: { siteUrl: z.string(), theme: z.enum(['light', 'dark']).optional() },
1884
1999
  }, async ({ siteUrl, theme }) => {
1885
- const data = getDashboardData(dataDir(), siteUrl);
2000
+ const data = { ...getDashboardData(dataDir(), siteUrl), apiKeys: apiKeysStatus() };
1886
2001
  if (data.empty) {
1887
2002
  return { content: [{ type: 'text', text: `No synced data for ${siteUrl} — run refresh_property first.` }], structuredContent: { error: 'empty', siteUrl } };
1888
2003
  }
@@ -1926,12 +2041,46 @@ export function createServer() {
1926
2041
  // otherwise CREATE an empty DB file for any bogus siteUrl posted at the API.
1927
2042
  if (!listLocalProperties(dataDir()).some(p => p.siteUrl === want))
1928
2043
  throw new Error(`unknown property ${want}`);
1929
- return getDashboardData(dataDir(), want);
2044
+ return { ...getDashboardData(dataDir(), want), apiKeys: apiKeysStatus() };
1930
2045
  },
1931
2046
  related_terms: async (a) => {
1932
2047
  const r = await requireDfs(dfs).relatedTerms(String(a.keyword ?? ''), a.location, a.languageCode);
1933
2048
  return r;
1934
2049
  },
2050
+ // Content research (on-demand) — News (free Google News + paid DFS) / Videos / Trends.
2051
+ news_discovery: async (a) => {
2052
+ const kw = String(a.keyword ?? '');
2053
+ const seen = new Set();
2054
+ const articles = [];
2055
+ const add = (x) => { const k = String(x.url || x.title || '').toLowerCase(); if (k && !seen.has(k)) {
2056
+ seen.add(k);
2057
+ articles.push(x);
2058
+ } };
2059
+ try {
2060
+ const g = await fetchGoogleNews(kw, { limit: 20 });
2061
+ for (const x of g.articles)
2062
+ add({ ...x, via: 'google-news' });
2063
+ }
2064
+ catch { /* free source optional */ }
2065
+ if (dfs) {
2066
+ try {
2067
+ const r = await dfs.serpNews(kw, a.location, a.languageCode);
2068
+ for (const x of r.articles)
2069
+ add({ ...x, via: 'dataforseo' });
2070
+ }
2071
+ catch { /* paid optional */ }
2072
+ }
2073
+ return { articles };
2074
+ },
2075
+ youtube_discovery: async (a) => {
2076
+ const r = await requireDfs(dfs).serpYoutube(String(a.keyword ?? ''), a.location, a.languageCode, a.blockDepth);
2077
+ return r;
2078
+ },
2079
+ topic_trend: async (a) => {
2080
+ const kws = Array.isArray(a.keywords) ? a.keywords : [String(a.keyword ?? '')].filter(Boolean);
2081
+ const r = await requireDfs(dfs).googleTrends(kws, a.location, a.languageCode, {});
2082
+ return r;
2083
+ },
1935
2084
  keyword_volume: async (a) => {
1936
2085
  const r = await requireDfs(dfs).searchVolume(a.keywords ?? [], a.location, a.languageCode);
1937
2086
  const items = (r.tasks[0]?.result ?? []).map((k) => ({