@houtini/seo-audit-console 0.11.1 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/dist/audit/contentArchitecture.d.ts +41 -0
- package/dist/audit/contentArchitecture.d.ts.map +1 -0
- package/dist/audit/contentArchitecture.js +109 -0
- package/dist/audit/contentArchitecture.js.map +1 -0
- package/dist/audit/lexical.d.ts +4 -0
- package/dist/audit/lexical.d.ts.map +1 -0
- package/dist/audit/lexical.js +12 -0
- package/dist/audit/lexical.js.map +1 -0
- package/dist/audit/opportunities.d.ts.map +1 -1
- package/dist/audit/opportunities.js +4 -14
- package/dist/audit/opportunities.js.map +1 -1
- package/dist/core/DataForSeoClient.d.ts +3 -1
- package/dist/core/DataForSeoClient.d.ts.map +1 -1
- package/dist/core/DataForSeoClient.js +9 -2
- package/dist/core/DataForSeoClient.js.map +1 -1
- package/dist/core/dashboardData.d.ts +5 -0
- package/dist/core/dashboardData.d.ts.map +1 -1
- package/dist/core/dashboardData.js +105 -94
- package/dist/core/dashboardData.js.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +157 -152
- package/dist/server.js.map +1 -1
- package/dist/src/ui/dashboard.html +325 -318
- package/package.json +2 -1
- package/server.json +2 -2
package/dist/server.js
CHANGED
|
@@ -44,7 +44,7 @@ import { GscSync, FULL_DIMENSIONS } from './core/GscSync.js';
|
|
|
44
44
|
import { UrlInspector } from './core/UrlInspector.js';
|
|
45
45
|
import { Crawler } from './core/Crawler.js';
|
|
46
46
|
import { Refresh } from './core/Refresh.js';
|
|
47
|
-
import { DataForSeoClient } from './core/DataForSeoClient.js';
|
|
47
|
+
import { DataForSeoClient, historicalDateFrom } from './core/DataForSeoClient.js';
|
|
48
48
|
import { RankTracker } from './core/RankTracker.js';
|
|
49
49
|
import { Backlinks } from './core/Backlinks.js';
|
|
50
50
|
import { LinkIntersect } from './core/LinkIntersect.js';
|
|
@@ -82,93 +82,93 @@ const CHECK_CATEGORIES = [
|
|
|
82
82
|
function isoDaysAgo(days) {
|
|
83
83
|
return new Date(Date.now() - days * 86400000).toISOString().slice(0, 10);
|
|
84
84
|
}
|
|
85
|
-
const SERVER_INSTRUCTIONS = `SEO Audit Console fuses four data sources into one SQLite database per property: Google Search Console history, a first-party site crawl, GSC URL Inspection, and on-demand DataForSEO (SERP/Labs/Backlinks). Its real power is COMPOSITION — joining sources to answer questions no single tool answers.
|
|
86
|
-
|
|
87
|
-
JOIN KEYS (memorise these):
|
|
88
|
-
- url_key — one normalised URL form (https, unified www/apex, sorted params, no tracking params/fragments). Joins the crawl (pages, links) ↔ GSC (search_analytics.page_key) ↔ url_inspection ↔ page_backlinks ↔ page_cwv ↔ page_entity. normalize_url shows the key for any URL.
|
|
89
|
-
- query — the literal search term. Joins GSC search_analytics ↔ DataForSEO keyword tools (keyword_volume, search_intent, ranked_keywords rows) ↔ keyword_intent.
|
|
90
|
-
- domain — a bare host (no scheme, no www). Joins the Labs tools (ranked_keywords, domain_visibility, top_pages, competitors_domain, topic_gaps) ↔ backlinks summary.
|
|
91
|
-
|
|
92
|
-
GRAIN (one line per source):
|
|
93
|
-
- search_analytics: date × query × page (plus device/country when synced with segments). Rows are additive; position must be impression-weighted when aggregated.
|
|
94
|
-
- pages / links: the LATEST crawl only, one row per url_key (history lives in page_snapshots, diffed by detect_changes). Carries title/H1/meta, canonical, schema (json_ld), body_chunks, iPR, click_depth, inlink_count.
|
|
95
|
-
- url_inspection: one row per inspected URL — Google's own view (coverage_state, google_canonical vs user_canonical, last_crawl_time, crawled_as, rich_results).
|
|
96
|
-
- rank_history: month × domain (DataForSEO rank distribution + ETV).
|
|
97
|
-
- page_backlinks: one row per backlinked URL (counts + live HTTP status).
|
|
98
|
-
- Labs tools: keyword × target, or month × target; cached 20 days; each call costs money — never loop them in bulk.
|
|
99
|
-
|
|
100
|
-
ARCHETYPE CHAINS (compose along these lines):
|
|
101
|
-
1. Demand → reality: a GSC query with impressions but weak rank → the ranking page's crawl fields (title/H1/body_chunks) → does the page actually say what the query asks? → fix on-page or draft content.
|
|
102
|
-
2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
|
|
103
|
-
3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
|
|
104
|
-
|
|
105
|
-
COST DISCIPLINE (behave like a strategist who knows the margins):
|
|
106
|
-
- Free and instant, use liberally: trend_categories, and everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
|
|
107
|
-
- Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, topic_trend, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
|
|
108
|
-
- Paid per-keyword (SERP): related_terms, youtube_discovery (ranking videos for a topic — pair with a transcript tool), news_discovery (recent coverage / freshness), and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
|
|
109
|
-
- Content research (what to write / what changed): topic_trend (is a topic rising/seasonal; with a categoryCode from the free trend_categories, is a whole market growing and what is breaking out in it) → youtube_discovery + news_discovery (what the winning videos/articles cover) → draft_content.
|
|
110
|
-
- Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
|
|
111
|
-
|
|
112
|
-
AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
|
|
113
|
-
1. Baseline: refresh_property → run_audit → serve_dashboard (share the URL).
|
|
114
|
-
2. Market: serp_features (feature exposure) + domain_visibility + competitors_domain → topic_gaps vs the named rivals.
|
|
115
|
-
3. Content plan: suggest_pages (demand you already have) + topic_gaps (demand rivals own) → draft_content briefs.
|
|
116
|
-
4. Fix cycle: fix_finding per top finding → re-crawl → detect_changes to prove the fix landed.
|
|
117
|
-
|
|
85
|
+
const SERVER_INSTRUCTIONS = `SEO Audit Console fuses four data sources into one SQLite database per property: Google Search Console history, a first-party site crawl, GSC URL Inspection, and on-demand DataForSEO (SERP/Labs/Backlinks). Its real power is COMPOSITION — joining sources to answer questions no single tool answers.
|
|
86
|
+
|
|
87
|
+
JOIN KEYS (memorise these):
|
|
88
|
+
- url_key — one normalised URL form (https, unified www/apex, sorted params, no tracking params/fragments). Joins the crawl (pages, links) ↔ GSC (search_analytics.page_key) ↔ url_inspection ↔ page_backlinks ↔ page_cwv ↔ page_entity. normalize_url shows the key for any URL.
|
|
89
|
+
- query — the literal search term. Joins GSC search_analytics ↔ DataForSEO keyword tools (keyword_volume, search_intent, ranked_keywords rows) ↔ keyword_intent.
|
|
90
|
+
- domain — a bare host (no scheme, no www). Joins the Labs tools (ranked_keywords, domain_visibility, top_pages, competitors_domain, topic_gaps) ↔ backlinks summary.
|
|
91
|
+
|
|
92
|
+
GRAIN (one line per source):
|
|
93
|
+
- search_analytics: date × query × page (plus device/country when synced with segments). Rows are additive; position must be impression-weighted when aggregated.
|
|
94
|
+
- pages / links: the LATEST crawl only, one row per url_key (history lives in page_snapshots, diffed by detect_changes). Carries title/H1/meta, canonical, schema (json_ld), body_chunks, iPR, click_depth, inlink_count.
|
|
95
|
+
- url_inspection: one row per inspected URL — Google's own view (coverage_state, google_canonical vs user_canonical, last_crawl_time, crawled_as, rich_results).
|
|
96
|
+
- rank_history: month × domain (DataForSEO rank distribution + ETV).
|
|
97
|
+
- page_backlinks: one row per backlinked URL (counts + live HTTP status).
|
|
98
|
+
- Labs tools: keyword × target, or month × target; cached 20 days; each call costs money — never loop them in bulk.
|
|
99
|
+
|
|
100
|
+
ARCHETYPE CHAINS (compose along these lines):
|
|
101
|
+
1. Demand → reality: a GSC query with impressions but weak rank → the ranking page's crawl fields (title/H1/body_chunks) → does the page actually say what the query asks? → fix on-page or draft content.
|
|
102
|
+
2. Authority → waste: iPR / backlinks flowing into non-200, redirected, or orphaned URLs → recover the equity with 301s or internal links (fix_finding generates them).
|
|
103
|
+
3. Competitor → gap: competitor keyword footprints (ranked_keywords / topic_gaps) minus our GSC + crawled-page footprint → topics to cover, each tied to the nearest existing page.
|
|
104
|
+
|
|
105
|
+
COST DISCIPLINE (behave like a strategist who knows the margins):
|
|
106
|
+
- Free and instant, use liberally: trend_categories, and everything on synced data — query_data, run_audit, query_audit, suggest_pages, list_templates, detect_changes, get_dashboard, serve_dashboard, export_report.
|
|
107
|
+
- Paid but CHEAP and 20-day cached (Labs/Keywords, ~$0.01–0.13 a call): keyword_volume, topic_trend, search_intent, ranked_keywords, serp_features, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps. Top-down pulls only — ONE ranked_keywords call answers "what does this domain rank for"; NEVER loop keywords through SERP endpoints to reconstruct what a Labs call returns.
|
|
108
|
+
- Paid per-keyword (SERP): related_terms, youtube_discovery (ranking videos for a topic — pair with a transcript tool), news_discovery (recent coverage / freshness), and AI-Overview CITATION checks. On-demand for a handful of clicked/explicit keywords, never a list.
|
|
109
|
+
- Content research (what to write / what changed): topic_trend (is a topic rising/seasonal; with a categoryCode from the free trend_categories, is a whole market growing and what is breaking out in it) → youtube_discovery + news_discovery (what the winning videos/articles cover) → draft_content.
|
|
110
|
+
- Separate subscription: pull_backlinks (DataForSEO Backlinks — a 40204 error means it isn't activated).
|
|
111
|
+
|
|
112
|
+
AGENCY MACRO-WORKFLOWS (the engagement arc — each stage feeds the next):
|
|
113
|
+
1. Baseline: refresh_property → run_audit → serve_dashboard (share the URL).
|
|
114
|
+
2. Market: serp_features (feature exposure) + domain_visibility + competitors_domain → topic_gaps vs the named rivals.
|
|
115
|
+
3. Content plan: suggest_pages (demand you already have) + topic_gaps (demand rivals own) → draft_content briefs.
|
|
116
|
+
4. Fix cycle: fix_finding per top finding → re-crawl → detect_changes to prove the fix landed.
|
|
117
|
+
|
|
118
118
|
Before planning ANY complex multi-source question, call composition_cookbook — it returns the full data-surface map and worked recipes using these exact tool and table names.`;
|
|
119
|
-
const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
|
|
120
|
-
|
|
121
|
-
This server's value is composition: joining Search Console, the crawl, URL Inspection and DataForSEO to answer questions no preset check covers. This page is static (no API calls) — use it to PLAN, then run the tools.
|
|
122
|
-
|
|
123
|
-
## The data surface
|
|
124
|
-
|
|
125
|
-
| Source (table) | Grain | Key dimensions | Join keys | Freshness | Cost |
|
|
126
|
-
|---|---|---|---|---|---|
|
|
127
|
-
| search_analytics (GSC) | date × query × page (+device/country with segments) | clicks, impressions, ctr, position | page_key (url_key), query, date | sync_gsc / refresh_property — incremental, GSC lags ~2–3 days | free |
|
|
128
|
-
| pages + links (crawl) | one row per url_key, LATEST crawl | status, title, H1, meta, canonical_key, robots, json_ld, hreflang, redirects, body_chunks, word_count, ipr, click_depth, inlink_count, conditional_304 | url_key | start_crawl / refresh_property (on demand) | free |
|
|
129
|
-
| page_snapshots (drift) | url_key × crawl | field-level SEO snapshot per crawl | url_key, captured_at | every crawl, automatically | free |
|
|
130
|
-
| url_inspection | one row per inspected URL (top pages by clicks, quota-limited) | coverage_state, page_fetch_state, google_canonical, user_canonical, last_crawl_time, crawled_as, rich_results | url_key | inspect_urls | free (GSC quota) |
|
|
131
|
-
| sitemap_urls | one row per sitemap URL | lastmod | url_key | captured at crawl time | free |
|
|
132
|
-
| rank_history | month × domain | rank distribution (1–3/4–10/11–20/21–100), ETV | period | track_ranks | paid, 20-day cache |
|
|
133
|
-
| page_backlinks | one row per backlinked URL | backlinks, referring_domains, live status_code | url_key, domain | pull_backlinks (needs the DataForSEO Backlinks subscription) | paid, 20-day cache |
|
|
134
|
-
| link_prospects | one row per prospect domain (links to competitors, not you) | intersections, domain_trust, spam_score, dofollow, trust_flow, topical_trust_flow | domain | link_intersect (needs the DataForSEO Backlinks subscription; Majestic optional) | paid, 20-day cache |
|
|
135
|
-
| keyword_intent | one row per keyword | intent + probability | query | search_intent (pass siteUrl to persist) | paid (cheap), cached |
|
|
136
|
-
| page_cwv | one row per audited URL | performance, LCP, CLS, TBT | url_key | page_lighthouse (pass siteUrl to persist) | paid, cached |
|
|
137
|
-
| page_entity + entity_edge | one row per page / edge per relation | QID, label, subclass-of / part-of | url_key, qid | resolve_entities (free Wikidata) | free |
|
|
138
|
-
| Labs (ranked_keywords, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps) | keyword × target, or month × target | volume, ETV, position, KD, intent, SERP features, AIO citations | domain, query | on demand | paid, 20-day cache |
|
|
139
|
-
| findings (audit_runs) | finding per check × URL | priority, evidence JSON | url_key, check_id | run_audit | free |
|
|
140
|
-
|
|
141
|
-
Raw access: query_audit runs any single check with full evidence; every table above lives in one SQLite file per property (path via data_location) if you need direct SQL.
|
|
142
|
-
|
|
143
|
-
## Worked recipes
|
|
144
|
-
|
|
145
|
-
1. **AI Overview exposure & citation loss.** ranked_keywords target:<your domain> aioOnly:true → keywords you rank for where the SERP shows an AI Overview (exposure). For citation (are YOU a source?) check specific keywords with related_terms/serpOrganic — the ai_overview item's references[] carries domain/url/quoted text. For each exposed keyword's ranking page, pull its per-day clicks from search_analytics (page_key × date). Pages whose clicks fell while the AIO citation appeared = you are feeding the answer without earning the visit.
|
|
146
|
-
2. **Striking distance without body coverage.** query_audit check:striking-distance (GSC rank 11–20) → for each page, check pages.body_chunks for the query's terms. The automated versions: body-missing-top-query, rag-answer-gap, and score_passages for the dense-answer test. Pages ranking 11–20 that never answer the query in one passage are the highest-yield rewrites.
|
|
147
|
-
3. **Not indexed + no equity.** url_inspection.coverage_state ~ 'not indexed' joined to pages.ipr + inlink_count. Low-iPR unindexed pages need internal links, not resubmission; high-iPR unindexed pages are the real anomalies. (Checks: coverage-not-indexed, underlinked-high-demand.)
|
|
148
|
-
4. **Cannibalisation with semantic overlap.** run_audit → keyword-cannibalisation evidence lists the competing URLs per query → compare those pages' body_chunks: heavy chunk overlap = consolidate (301 the loser); light overlap = differentiate the titles/H1s and interlink with distinct anchors.
|
|
149
|
-
5. **Stable rank, falling CTR → SERP feature shift.** In search_analytics find queries where weekly position is flat but ctr declines → related_terms / ranked_keywords serpFeatures for that keyword shows what now sits above you (AIO, featured snippet, shopping). ctr-below-expected is the deterministic starting list.
|
|
150
|
-
6. **Schema vs rich-result reality.** pages.json_ld (declared @types) joined to url_inspection.rich_results (what Google actually detected + issues). The rich-result-issues check automates the per-URL diff; the composition question is per-TEMPLATE (list_templates): which template's schema never earns its rich result?
|
|
151
|
-
7. **Migration signal transfer.** pages.redirects (recorded chains) → url_inspection.google_canonical of the target (has Google accepted the move?) → search_analytics clicks by page_key before/after the migration date. Equity that didn't follow the 301 shows up as a target with no canonical adoption and no click recovery.
|
|
152
|
-
8. **404s with backlinks.** pages.status_code = 404 joined to page_backlinks.backlinks (run pull_backlinks first) → run_audit surfaces backlinks-to-404; fix_finding generates the 301 that recovers the equity.
|
|
153
|
-
9. **Competitor topic gap.** topic_gaps (bounded + cached: competitor ranked_keywords minus your GSC queries and page titles/H1s, clustered and scored) — or do it manually with ranked_keywords per competitor when you want the raw rows.
|
|
154
|
-
10. **Link gap (what links do rivals have that we don't).** link_intersect over the competitor set (or a single company) → link_prospects: domains linking to them but not you, followed-first and sorted by domain trust. DataForSEO domain rank surfaces spam directories at the top; MAJESTIC_API_KEY re-sorts by Trust Flow (a rank-227 domain is often TF 0) and Topical Trust Flow shows whether that authority is on-topic. data_storage flags when a property's prospect set is going stale (competitors keep earning links).
|
|
155
|
-
|
|
156
|
-
## Novel combinations (nothing else surfaces these)
|
|
157
|
-
|
|
158
|
-
- **Crawl budget vs equity:** url_inspection.last_crawl_time × pages.ipr — your highest-iPR pages should be recrawled often; a high-iPR page Google rarely revisits is a freshness/priority problem (and vice versa: junk crawled daily = wasted budget).
|
|
159
|
-
- **AIO text vs your content:** the ai_overview items in a SERP call (related_terms' underlying serpOrganic) × pages.body_chunks — is the text Google quotes actually on your page, and in one extractable chunk?
|
|
160
|
-
- **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
|
|
161
|
-
- **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
|
|
162
|
-
|
|
163
|
-
## Agency engagement recipes (the deliverable arc)
|
|
164
|
-
|
|
165
|
-
- **Week-one baseline:** refresh_property → run_audit → serve_dashboard. Share the dashboard URL; the ranked findings ARE the technical workstream.
|
|
166
|
-
- **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
|
|
167
|
-
- **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
|
|
168
|
-
- **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
|
|
169
|
-
- **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
|
|
170
|
-
- **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
|
|
171
|
-
|
|
119
|
+
const COOKBOOK_TEXT = `# Composition cookbook — the data surface and how to join it
|
|
120
|
+
|
|
121
|
+
This server's value is composition: joining Search Console, the crawl, URL Inspection and DataForSEO to answer questions no preset check covers. This page is static (no API calls) — use it to PLAN, then run the tools.
|
|
122
|
+
|
|
123
|
+
## The data surface
|
|
124
|
+
|
|
125
|
+
| Source (table) | Grain | Key dimensions | Join keys | Freshness | Cost |
|
|
126
|
+
|---|---|---|---|---|---|
|
|
127
|
+
| search_analytics (GSC) | date × query × page (+device/country with segments) | clicks, impressions, ctr, position | page_key (url_key), query, date | sync_gsc / refresh_property — incremental, GSC lags ~2–3 days | free |
|
|
128
|
+
| pages + links (crawl) | one row per url_key, LATEST crawl | status, title, H1, meta, canonical_key, robots, json_ld, hreflang, redirects, body_chunks, word_count, ipr, click_depth, inlink_count, conditional_304 | url_key | start_crawl / refresh_property (on demand) | free |
|
|
129
|
+
| page_snapshots (drift) | url_key × crawl | field-level SEO snapshot per crawl | url_key, captured_at | every crawl, automatically | free |
|
|
130
|
+
| url_inspection | one row per inspected URL (top pages by clicks, quota-limited) | coverage_state, page_fetch_state, google_canonical, user_canonical, last_crawl_time, crawled_as, rich_results | url_key | inspect_urls | free (GSC quota) |
|
|
131
|
+
| sitemap_urls | one row per sitemap URL | lastmod | url_key | captured at crawl time | free |
|
|
132
|
+
| rank_history | month × domain | rank distribution (1–3/4–10/11–20/21–100), ETV | period | track_ranks | paid, 20-day cache |
|
|
133
|
+
| page_backlinks | one row per backlinked URL | backlinks, referring_domains, live status_code | url_key, domain | pull_backlinks (needs the DataForSEO Backlinks subscription) | paid, 20-day cache |
|
|
134
|
+
| link_prospects | one row per prospect domain (links to competitors, not you) | intersections, domain_trust, spam_score, dofollow, trust_flow, topical_trust_flow | domain | link_intersect (needs the DataForSEO Backlinks subscription; Majestic optional) | paid, 20-day cache |
|
|
135
|
+
| keyword_intent | one row per keyword | intent + probability | query | search_intent (pass siteUrl to persist) | paid (cheap), cached |
|
|
136
|
+
| page_cwv | one row per audited URL | performance, LCP, CLS, TBT | url_key | page_lighthouse (pass siteUrl to persist) | paid, cached |
|
|
137
|
+
| page_entity + entity_edge | one row per page / edge per relation | QID, label, subclass-of / part-of | url_key, qid | resolve_entities (free Wikidata) | free |
|
|
138
|
+
| Labs (ranked_keywords, domain_visibility, top_pages, competitors_domain, page_intersection, topic_gaps) | keyword × target, or month × target | volume, ETV, position, KD, intent, SERP features, AIO citations | domain, query | on demand | paid, 20-day cache |
|
|
139
|
+
| findings (audit_runs) | finding per check × URL | priority, evidence JSON | url_key, check_id | run_audit | free |
|
|
140
|
+
|
|
141
|
+
Raw access: query_audit runs any single check with full evidence; every table above lives in one SQLite file per property (path via data_location) if you need direct SQL.
|
|
142
|
+
|
|
143
|
+
## Worked recipes
|
|
144
|
+
|
|
145
|
+
1. **AI Overview exposure & citation loss.** ranked_keywords target:<your domain> aioOnly:true → keywords you rank for where the SERP shows an AI Overview (exposure). For citation (are YOU a source?) check specific keywords with related_terms/serpOrganic — the ai_overview item's references[] carries domain/url/quoted text. For each exposed keyword's ranking page, pull its per-day clicks from search_analytics (page_key × date). Pages whose clicks fell while the AIO citation appeared = you are feeding the answer without earning the visit.
|
|
146
|
+
2. **Striking distance without body coverage.** query_audit check:striking-distance (GSC rank 11–20) → for each page, check pages.body_chunks for the query's terms. The automated versions: body-missing-top-query, rag-answer-gap, and score_passages for the dense-answer test. Pages ranking 11–20 that never answer the query in one passage are the highest-yield rewrites.
|
|
147
|
+
3. **Not indexed + no equity.** url_inspection.coverage_state ~ 'not indexed' joined to pages.ipr + inlink_count. Low-iPR unindexed pages need internal links, not resubmission; high-iPR unindexed pages are the real anomalies. (Checks: coverage-not-indexed, underlinked-high-demand.)
|
|
148
|
+
4. **Cannibalisation with semantic overlap.** run_audit → keyword-cannibalisation evidence lists the competing URLs per query → compare those pages' body_chunks: heavy chunk overlap = consolidate (301 the loser); light overlap = differentiate the titles/H1s and interlink with distinct anchors.
|
|
149
|
+
5. **Stable rank, falling CTR → SERP feature shift.** In search_analytics find queries where weekly position is flat but ctr declines → related_terms / ranked_keywords serpFeatures for that keyword shows what now sits above you (AIO, featured snippet, shopping). ctr-below-expected is the deterministic starting list.
|
|
150
|
+
6. **Schema vs rich-result reality.** pages.json_ld (declared @types) joined to url_inspection.rich_results (what Google actually detected + issues). The rich-result-issues check automates the per-URL diff; the composition question is per-TEMPLATE (list_templates): which template's schema never earns its rich result?
|
|
151
|
+
7. **Migration signal transfer.** pages.redirects (recorded chains) → url_inspection.google_canonical of the target (has Google accepted the move?) → search_analytics clicks by page_key before/after the migration date. Equity that didn't follow the 301 shows up as a target with no canonical adoption and no click recovery.
|
|
152
|
+
8. **404s with backlinks.** pages.status_code = 404 joined to page_backlinks.backlinks (run pull_backlinks first) → run_audit surfaces backlinks-to-404; fix_finding generates the 301 that recovers the equity.
|
|
153
|
+
9. **Competitor topic gap.** topic_gaps (bounded + cached: competitor ranked_keywords minus your GSC queries and page titles/H1s, clustered and scored) — or do it manually with ranked_keywords per competitor when you want the raw rows.
|
|
154
|
+
10. **Link gap (what links do rivals have that we don't).** link_intersect over the competitor set (or a single company) → link_prospects: domains linking to them but not you, followed-first and sorted by domain trust. DataForSEO domain rank surfaces spam directories at the top; MAJESTIC_API_KEY re-sorts by Trust Flow (a rank-227 domain is often TF 0) and Topical Trust Flow shows whether that authority is on-topic. data_storage flags when a property's prospect set is going stale (competitors keep earning links).
|
|
155
|
+
|
|
156
|
+
## Novel combinations (nothing else surfaces these)
|
|
157
|
+
|
|
158
|
+
- **Crawl budget vs equity:** url_inspection.last_crawl_time × pages.ipr — your highest-iPR pages should be recrawled often; a high-iPR page Google rarely revisits is a freshness/priority problem (and vice versa: junk crawled daily = wasted budget).
|
|
159
|
+
- **AIO text vs your content:** the ai_overview items in a SERP call (related_terms' underlying serpOrganic) × pages.body_chunks — is the text Google quotes actually on your page, and in one extractable chunk?
|
|
160
|
+
- **Crawl-to-first-impression latency:** url_inspection.last_crawl_time vs the first date a page appears in search_analytics — how fast does Google turn a crawl into impressions, per template? Slow templates have an indexing-pipeline problem.
|
|
161
|
+
- **Crawled-as vs response times:** url_inspection.crawled_as (mobile/desktop agent) × pages.response_time_ms — slow responses specifically on the agent Google uses against you.
|
|
162
|
+
|
|
163
|
+
## Agency engagement recipes (the deliverable arc)
|
|
164
|
+
|
|
165
|
+
- **Week-one baseline:** refresh_property → run_audit → serve_dashboard. Share the dashboard URL; the ranked findings ARE the technical workstream.
|
|
166
|
+
- **Market read:** serp_features (feature/AIO exposure, volume-weighted) + domain_visibility for the client and each named rival (one cached call each) → who is structurally winning, and how much of the market SERP features already absorb.
|
|
167
|
+
- **Content plan:** suggest_pages (demand you already earn impressions for) + topic_gaps (demand rivals own that you don't) → draft_content for the winners. Every proposal traces to real impressions or a rival's real footprint - no invented "keyword ideas".
|
|
168
|
+
- **Fix-and-prove cycle:** fix_finding on the top finding → ship → start_crawl → detect_changes shows the fix landed → re-run run_audit and watch the finding drop off. That screenshot is the client update.
|
|
169
|
+
- **Content recon (why a page is losing):** recon_targets picks the worst declining/striking pages, fetches our live page + the Google SERP (organic rank + AI-Overview citations + video), and classifies WHY — the sharpest class is "we rank but the AI Overview won't quote us" = a data-accuracy/freshness/markup problem. Then research the competitor set it returns (firecrawl for pages, supadata for the ranking videos), write the gaps back with save_recon_todo, and track the fixes with recon_todos (which can re-measure whether you moved from uncited→cited). Pass location to match where your impressions come from — organic rank is location-sensitive.
|
|
170
|
+
- **Cost rule of thumb:** an entire competitive read (visibility + footprint + gaps for 4 domains) is a handful of cached Labs calls - under a dollar. If a plan involves looping SERP calls over a keyword list, it is the wrong plan; a Labs endpoint already has that answer top-down.
|
|
171
|
+
|
|
172
172
|
Plan the join first (url_key / query / domain), state the grain of each side, then run the fewest paid calls that answer it.`;
|
|
173
173
|
function buildChecksMarkdown(checks) {
|
|
174
174
|
const byCategory = new Map();
|
|
@@ -183,54 +183,54 @@ function buildChecksMarkdown(checks) {
|
|
|
183
183
|
list.map(c => `| \`${c.id}\` | ${c.severity} | ${c.labels.join(',')} | ${c.certainty} | ${c.fixType} | ${esc(c.title)} | ${esc(c.fix)} |`).join('\n'));
|
|
184
184
|
return `# Check registry — ${checks.length} checks\n\nLabels: D = deterministic (cites bytes), G = evidence from Google's own data (Search Console / URL Inspection), N = judgement (heuristic, gated behind includeJudgement). Certainty < 1 discounts a finding's priority.\n\n${sections.join('\n\n')}\n`;
|
|
185
185
|
}
|
|
186
|
-
const HELP_TEXT = `# SEO Audit Console — what it can do
|
|
187
|
-
A technical-SEO audit that fuses **Search Console + a site crawl + DataForSEO**, joined on a normalised URL, with evidence on every finding. Typical flow: **refresh → audit → fix → report**.
|
|
188
|
-
|
|
189
|
-
## 1. Sync the data
|
|
190
|
-
- **refresh_property** — sync everything (GSC → crawl → URL inspection → rank history). _"Refresh sc-domain:example.com"_ · add \`segments:true\` for device/country, \`maxPages\`, \`startDate\`.
|
|
191
|
-
- **sync_gsc / start_crawl / inspect_urls / track_ranks** — run just one part. _"Crawl example.com, 500 pages"_
|
|
192
|
-
- **check_sync_status / check_crawl_status** — poll a job. _"Check sync status"_
|
|
193
|
-
- **list_properties** — _"List my Search Console properties"_
|
|
194
|
-
|
|
195
|
-
## 2. Audit
|
|
196
|
-
- **run_audit** — score all checks; returns a prioritised markdown report. _"Run an SEO audit on sc-domain:example.com"_ · \`scope:full\`, \`categories\`, \`includeJudgement:true\`.
|
|
197
|
-
- **query_audit** — one named check with evidence (\`columns\`/\`offset\` for big sets). _"Show striking-distance for example.com"_
|
|
198
|
-
- **query_data** — read-only queries over the raw tables; aggregates in the database (counts/percentages/sums), answers not rows. _"How do status codes break down on example.com?"_
|
|
199
|
-
- **list_checks** — _"What does the audit check for?"_
|
|
200
|
-
|
|
201
|
-
## 3. Fix (the moat)
|
|
202
|
-
- **fix_finding** — paste-ready remediation from your own data: JSON-LD for missing/invalid schema, a 301 rule for broken links, iPR-ranked internal-link suggestions. _"Generate the fix for finding 12"_ or _"fix_finding check:missing-required-fields url:https://example.com/x"_
|
|
203
|
-
- **detect_changes** — what changed since the last crawl (status, canonical, noindex, title, schema), severity-ranked — the regression monitor. _"Detect changes on example.com"_
|
|
204
|
-
- **check_agent_readiness** — is your site ready for AI agents? Scores llms.txt / agents.md / AI-bot rules / Content Signals / MCP server card / Agent Skills / API Catalog / OAuth signals (0–100 + level) with copy-paste fixes. _"Check agent readiness for example.com"_
|
|
205
|
-
|
|
206
|
-
## 4. Backlinks, keywords & competitive (DataForSEO, on-demand, cached 20 days)
|
|
207
|
-
- **pull_backlinks** — backlink profile + per-page counts + live status → unlocks **backlinks-to-404** (recover lost equity), top-linked pages, true orphans. _"Pull backlinks for example.com"_
|
|
208
|
-
- **link_intersect** — the links your competitors have that you don't - a prioritised outreach prospect list (followed-first, then domain trust, spam filtered). Also answers "what links does company X have that we don't?" for a single company. Set MAJESTIC_API_KEY to re-sort by Trust Flow + Topical Trust Flow (kills directory noise). _"Link intersect for example.com vs rival1.com, rival2.com"_
|
|
209
|
-
- **keyword_volume / related_terms** — volume/CPC, and People-Also-Ask + related searches. _"Search volume for [\\"best widgets\\"]"_
|
|
210
|
-
- **topic_trend / trend_categories** — Google Trends interest over time for keywords, a whole category (no keyword needed), or a keyword inside a category; \`related:true\` adds rising queries and topics. trend_categories (free) finds the category code. _"Is the Software category rising in the UK? What's breaking out in it?"_
|
|
211
|
-
- **search_intent** — informational/navigational/commercial/transactional per keyword → spot intent mismatch behind low CTR. _"Classify intent for [\\"buy running shoes\\", \\"how to clean shoes\\"]"_
|
|
212
|
-
- **page_lighthouse** — lab Core Web Vitals + opportunities for one URL (~20–120s). _"Run Lighthouse on https://example.com/slow-page"_
|
|
213
|
-
- **competitors_domain** — domains competing for your organic keywords. _"Find competitors for example.com in the UK"_
|
|
214
|
-
- **page_intersection** — keywords competitor pages rank for but yours doesn’t (content gap). _"Content gap: competitorUrls [\\"https://rival.com/guide\\"], excludePages [\\"https://example.com/guide\\"]"_
|
|
215
|
-
- **domain_visibility** — monthly ranking-keyword distribution + ETV trend for ANY domain/subdomain (Semrush-style organic overview). _"Show visibility over time for competitor.com"_
|
|
216
|
-
- **top_pages** — a domain's top organic pages by estimated traffic. _"Top pages on competitor.com"_
|
|
217
|
-
- **ranked_keywords** — keywords a domain / subdomain / URL / subfolder ranks for (+ difficulty, intent, SERP features). _"What does competitor.com/blog/ rank for?"_ · \`scope:url|folder\` · \`aioOnly:true\` = keywords where the target is cited in AI Overviews
|
|
218
|
-
- **topic_gaps** — what related topics should this site cover to be expert in its space: competitor keyword footprints minus everything you already rank or have a page for, clustered into ranked topics with volumes, the owning competitor and your nearest existing page. _"What topics should example.com cover? Compare against rival.com"_
|
|
219
|
-
|
|
220
|
-
## 5. Templates & opportunities
|
|
221
|
-
- **list_templates** — cluster pages into templates (one fix → N pages) with a representative exemplar. _"List page templates for example.com"_
|
|
222
|
-
- **suggest_pages** — new-page ideas grounded in real GSC demand, minus what you already cover. _"Suggest new pages for example.com"_
|
|
223
|
-
- **resolve_entities** — map pages to Wikidata entities → unlocks entity-internal-link-gap (judgement). _"Resolve entities for example.com"_
|
|
224
|
-
|
|
225
|
-
## 6. Reports & dashboard
|
|
226
|
-
- **get_dashboard** — interactive dashboard (renders in chat). _"Show the dashboard for example.com"_
|
|
227
|
-
- **export_report** — self-contained shareable HTML to send a client. _"Export the report for example.com"_
|
|
228
|
-
|
|
229
|
-
## 7. Utilities
|
|
230
|
-
- **data_location** — where DBs are stored (set with a path). · **normalize_url** — the join key for a URL.
|
|
231
|
-
- **data_storage** — per-property disk usage + row counts; prune (vacuum / clear-crawl-history / delete-property, destructive ones need \`confirm:true\`). _"How much disk is my audit data using?"_
|
|
232
|
-
|
|
233
|
-
_Tip: first time on a property → \`refresh_property\` then \`run_audit\`._
|
|
186
|
+
const HELP_TEXT = `# SEO Audit Console — what it can do
|
|
187
|
+
A technical-SEO audit that fuses **Search Console + a site crawl + DataForSEO**, joined on a normalised URL, with evidence on every finding. Typical flow: **refresh → audit → fix → report**.
|
|
188
|
+
|
|
189
|
+
## 1. Sync the data
|
|
190
|
+
- **refresh_property** — sync everything (GSC → crawl → URL inspection → rank history). _"Refresh sc-domain:example.com"_ · add \`segments:true\` for device/country, \`maxPages\`, \`startDate\`.
|
|
191
|
+
- **sync_gsc / start_crawl / inspect_urls / track_ranks** — run just one part. _"Crawl example.com, 500 pages"_
|
|
192
|
+
- **check_sync_status / check_crawl_status** — poll a job. _"Check sync status"_
|
|
193
|
+
- **list_properties** — _"List my Search Console properties"_
|
|
194
|
+
|
|
195
|
+
## 2. Audit
|
|
196
|
+
- **run_audit** — score all checks; returns a prioritised markdown report. _"Run an SEO audit on sc-domain:example.com"_ · \`scope:full\`, \`categories\`, \`includeJudgement:true\`.
|
|
197
|
+
- **query_audit** — one named check with evidence (\`columns\`/\`offset\` for big sets). _"Show striking-distance for example.com"_
|
|
198
|
+
- **query_data** — read-only queries over the raw tables; aggregates in the database (counts/percentages/sums), answers not rows. _"How do status codes break down on example.com?"_
|
|
199
|
+
- **list_checks** — _"What does the audit check for?"_
|
|
200
|
+
|
|
201
|
+
## 3. Fix (the moat)
|
|
202
|
+
- **fix_finding** — paste-ready remediation from your own data: JSON-LD for missing/invalid schema, a 301 rule for broken links, iPR-ranked internal-link suggestions. _"Generate the fix for finding 12"_ or _"fix_finding check:missing-required-fields url:https://example.com/x"_
|
|
203
|
+
- **detect_changes** — what changed since the last crawl (status, canonical, noindex, title, schema), severity-ranked — the regression monitor. _"Detect changes on example.com"_
|
|
204
|
+
- **check_agent_readiness** — is your site ready for AI agents? Scores llms.txt / agents.md / AI-bot rules / Content Signals / MCP server card / Agent Skills / API Catalog / OAuth signals (0–100 + level) with copy-paste fixes. _"Check agent readiness for example.com"_
|
|
205
|
+
|
|
206
|
+
## 4. Backlinks, keywords & competitive (DataForSEO, on-demand, cached 20 days)
|
|
207
|
+
- **pull_backlinks** — backlink profile + per-page counts + live status → unlocks **backlinks-to-404** (recover lost equity), top-linked pages, true orphans. _"Pull backlinks for example.com"_
|
|
208
|
+
- **link_intersect** — the links your competitors have that you don't - a prioritised outreach prospect list (followed-first, then domain trust, spam filtered). Also answers "what links does company X have that we don't?" for a single company. Set MAJESTIC_API_KEY to re-sort by Trust Flow + Topical Trust Flow (kills directory noise). _"Link intersect for example.com vs rival1.com, rival2.com"_
|
|
209
|
+
- **keyword_volume / related_terms** — volume/CPC, and People-Also-Ask + related searches. _"Search volume for [\\"best widgets\\"]"_
|
|
210
|
+
- **topic_trend / trend_categories** — Google Trends interest over time for keywords, a whole category (no keyword needed), or a keyword inside a category; \`related:true\` adds rising queries and topics. trend_categories (free) finds the category code. _"Is the Software category rising in the UK? What's breaking out in it?"_
|
|
211
|
+
- **search_intent** — informational/navigational/commercial/transactional per keyword → spot intent mismatch behind low CTR. _"Classify intent for [\\"buy running shoes\\", \\"how to clean shoes\\"]"_
|
|
212
|
+
- **page_lighthouse** — lab Core Web Vitals + opportunities for one URL (~20–120s). _"Run Lighthouse on https://example.com/slow-page"_
|
|
213
|
+
- **competitors_domain** — domains competing for your organic keywords. _"Find competitors for example.com in the UK"_
|
|
214
|
+
- **page_intersection** — keywords competitor pages rank for but yours doesn’t (content gap). _"Content gap: competitorUrls [\\"https://rival.com/guide\\"], excludePages [\\"https://example.com/guide\\"]"_
|
|
215
|
+
- **domain_visibility** — monthly ranking-keyword distribution + ETV trend for ANY domain/subdomain (Semrush-style organic overview). _"Show visibility over time for competitor.com"_
|
|
216
|
+
- **top_pages** — a domain's top organic pages by estimated traffic. _"Top pages on competitor.com"_
|
|
217
|
+
- **ranked_keywords** — keywords a domain / subdomain / URL / subfolder ranks for (+ difficulty, intent, SERP features). _"What does competitor.com/blog/ rank for?"_ · \`scope:url|folder\` · \`aioOnly:true\` = keywords where the target is cited in AI Overviews
|
|
218
|
+
- **topic_gaps** — what related topics should this site cover to be expert in its space: competitor keyword footprints minus everything you already rank or have a page for, clustered into ranked topics with volumes, the owning competitor and your nearest existing page. _"What topics should example.com cover? Compare against rival.com"_
|
|
219
|
+
|
|
220
|
+
## 5. Templates & opportunities
|
|
221
|
+
- **list_templates** — cluster pages into templates (one fix → N pages) with a representative exemplar. _"List page templates for example.com"_
|
|
222
|
+
- **suggest_pages** — new-page ideas grounded in real GSC demand, minus what you already cover. _"Suggest new pages for example.com"_
|
|
223
|
+
- **resolve_entities** — map pages to Wikidata entities → unlocks entity-internal-link-gap (judgement). _"Resolve entities for example.com"_
|
|
224
|
+
|
|
225
|
+
## 6. Reports & dashboard
|
|
226
|
+
- **get_dashboard** — interactive dashboard (renders in chat). _"Show the dashboard for example.com"_
|
|
227
|
+
- **export_report** — self-contained shareable HTML to send a client. _"Export the report for example.com"_
|
|
228
|
+
|
|
229
|
+
## 7. Utilities
|
|
230
|
+
- **data_location** — where DBs are stored (set with a path). · **normalize_url** — the join key for a URL.
|
|
231
|
+
- **data_storage** — per-property disk usage + row counts; prune (vacuum / clear-crawl-history / delete-property, destructive ones need \`confirm:true\`). _"How much disk is my audit data using?"_
|
|
232
|
+
|
|
233
|
+
_Tip: first time on a property → \`refresh_property\` then \`run_audit\`._
|
|
234
234
|
_Composing your own analysis? Call \`composition_cookbook\` first — the data-surface map (grain + join keys per source) and worked multi-source recipes._`;
|
|
235
235
|
export function createServer() {
|
|
236
236
|
const server = new McpServer({ name: SERVER_NAME, version: SERVER_VERSION }, { instructions: SERVER_INSTRUCTIONS });
|
|
@@ -524,7 +524,7 @@ export function createServer() {
|
|
|
524
524
|
try {
|
|
525
525
|
const adb = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
526
526
|
try {
|
|
527
|
-
adb.db.prepare(`INSERT INTO agent_readiness (origin,score,level,checks,by_category,checked_at) VALUES (?,?,?,?,?,datetime('now'))
|
|
527
|
+
adb.db.prepare(`INSERT INTO agent_readiness (origin,score,level,checks,by_category,checked_at) VALUES (?,?,?,?,?,datetime('now'))
|
|
528
528
|
ON CONFLICT(origin) DO UPDATE SET score=excluded.score, level=excluded.level, checks=excluded.checks, by_category=excluded.by_category, checked_at=excluded.checked_at`)
|
|
529
529
|
.run(r.origin, r.score, r.level, JSON.stringify(r.checks), JSON.stringify(r.byCategory));
|
|
530
530
|
}
|
|
@@ -1049,7 +1049,7 @@ export function createServer() {
|
|
|
1049
1049
|
targets = [];
|
|
1050
1050
|
for (const u of urls) {
|
|
1051
1051
|
const key = urlKey(u, { hostForm });
|
|
1052
|
-
const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
|
|
1052
|
+
const row = db.db.prepare(`SELECT query, SUM(impressions) imp, SUM(clicks) clk, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
|
|
1053
1053
|
FROM search_analytics WHERE page_key=? AND query IS NOT NULL AND date > date(?, '-28 days') AND date <= ? GROUP BY query ORDER BY imp DESC LIMIT 1`)
|
|
1054
1054
|
.get(key, fresh.effectiveMax, fresh.effectiveMax);
|
|
1055
1055
|
if (row?.query)
|
|
@@ -1274,7 +1274,7 @@ export function createServer() {
|
|
|
1274
1274
|
where.push('status=?');
|
|
1275
1275
|
args.push(status);
|
|
1276
1276
|
}
|
|
1277
|
-
const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
|
|
1277
|
+
const rows = db.db.prepare(`SELECT id, url_key, query, action, type, rationale, priority, status, source, notes, baseline, outcome
|
|
1278
1278
|
FROM recon_todo ${where.length ? 'WHERE ' + where.join(' AND ') : ''} ORDER BY url_key, priority DESC`).all(...args);
|
|
1279
1279
|
if (!rows.length)
|
|
1280
1280
|
return { content: [{ type: 'text', text: `No recon to-dos${key ? ` for ${key}` : ''}${status ? ` with status ${status}` : ''}. Run recon_targets to generate some.` }], structuredContent: { todos: [] } };
|
|
@@ -1344,7 +1344,7 @@ export function createServer() {
|
|
|
1344
1344
|
if (siteUrl) {
|
|
1345
1345
|
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
1346
1346
|
try {
|
|
1347
|
-
const up = db.db.prepare(`INSERT INTO keyword_intent (keyword,intent,probability,fetched_at) VALUES (?,?,?,datetime('now'))
|
|
1347
|
+
const up = db.db.prepare(`INSERT INTO keyword_intent (keyword,intent,probability,fetched_at) VALUES (?,?,?,datetime('now'))
|
|
1348
1348
|
ON CONFLICT(keyword) DO UPDATE SET intent=excluded.intent, probability=excluded.probability, fetched_at=datetime('now')`);
|
|
1349
1349
|
db.db.transaction(() => { for (const it of items)
|
|
1350
1350
|
if (it.keyword && it.intent) {
|
|
@@ -1398,9 +1398,9 @@ export function createServer() {
|
|
|
1398
1398
|
if (siteUrl && cats.performance) {
|
|
1399
1399
|
const db = new AuditDatabase(dbPathFor(dataDir(), siteUrl));
|
|
1400
1400
|
try {
|
|
1401
|
-
db.db.prepare(`INSERT INTO page_cwv (url_key,url,for_mobile,performance,lcp_ms,cls,tbt_ms,fetched_at)
|
|
1402
|
-
VALUES (?,?,?,?,?,?,?,datetime('now'))
|
|
1403
|
-
ON CONFLICT(url_key) DO UPDATE SET url=excluded.url, for_mobile=excluded.for_mobile, performance=excluded.performance,
|
|
1401
|
+
db.db.prepare(`INSERT INTO page_cwv (url_key,url,for_mobile,performance,lcp_ms,cls,tbt_ms,fetched_at)
|
|
1402
|
+
VALUES (?,?,?,?,?,?,?,datetime('now'))
|
|
1403
|
+
ON CONFLICT(url_key) DO UPDATE SET url=excluded.url, for_mobile=excluded.for_mobile, performance=excluded.performance,
|
|
1404
1404
|
lcp_ms=excluded.lcp_ms, cls=excluded.cls, tbt_ms=excluded.tbt_ms, fetched_at=datetime('now')`)
|
|
1405
1405
|
.run(urlKey(url, { hostForm: hostFormForProperty(siteUrl) }), url, mobile ? 1 : 0, out.scores.performance, numeric('largest-contentful-paint'), numeric('cumulative-layout-shift'), numeric('total-blocking-time'));
|
|
1406
1406
|
persisted = true;
|
|
@@ -1493,12 +1493,14 @@ export function createServer() {
|
|
|
1493
1493
|
target: z.string().describe('Domain or subdomain, no scheme (e.g. example.com or blog.example.com)'),
|
|
1494
1494
|
location: z.union([z.string(), z.number()]).optional(),
|
|
1495
1495
|
languageName: z.string().optional(),
|
|
1496
|
-
months: z.number().int().min(1).max(24).optional(),
|
|
1496
|
+
months: z.number().int().min(1).max(24).optional().describe('Months of history ending last month (default 12). Sent as date_from; longer windows cost more (billed per monthly item).'),
|
|
1497
1497
|
},
|
|
1498
1498
|
}, async ({ target, location, languageName, months }) => {
|
|
1499
1499
|
const client = requireDfs(dfs);
|
|
1500
1500
|
const cleaned = dfsHost(target);
|
|
1501
|
-
const
|
|
1501
|
+
const windowMonths = months ?? 12;
|
|
1502
|
+
const dateFrom = historicalDateFrom(windowMonths);
|
|
1503
|
+
const r = await client.historicalRankOverview(cleaned, location, 'en', languageName, dateFrom);
|
|
1502
1504
|
const items = r.tasks[0]?.result?.[0]?.items ?? [];
|
|
1503
1505
|
const n = (v) => Number(v) || 0;
|
|
1504
1506
|
const series = items
|
|
@@ -1519,11 +1521,11 @@ export function createServer() {
|
|
|
1519
1521
|
};
|
|
1520
1522
|
})
|
|
1521
1523
|
.sort((a, b) => (a.period < b.period ? -1 : 1))
|
|
1522
|
-
.slice(-
|
|
1524
|
+
.slice(-windowMonths);
|
|
1523
1525
|
if (!series.length) {
|
|
1524
1526
|
return {
|
|
1525
1527
|
content: [{ type: 'text', text: `No historical rank data for ${cleaned} — check the target (bare domain/subdomain) and location.` }],
|
|
1526
|
-
structuredContent: { target: cleaned, series: [], cached: r.cached, cost: r.cost },
|
|
1528
|
+
structuredContent: { target: cleaned, requestedMonths: windowMonths, dateFrom, series: [], cached: r.cached, cost: r.cost },
|
|
1527
1529
|
};
|
|
1528
1530
|
}
|
|
1529
1531
|
const first = series[0];
|
|
@@ -1532,13 +1534,16 @@ export function createServer() {
|
|
|
1532
1534
|
const verdict = Math.abs(delta) < 10
|
|
1533
1535
|
? `flat (ETV ${fmtNum(first.etv)} → ${fmtNum(last.etv)}, ${delta >= 0 ? '+' : ''}${delta.toFixed(0)}% ${first.period} → ${last.period})`
|
|
1534
1536
|
: `${delta > 0 ? 'rising' : 'declining'} (ETV ${fmtNum(first.etv)} → ${fmtNum(last.etv)}, ${delta > 0 ? '+' : ''}${delta.toFixed(0)}% ${first.period} → ${last.period})`;
|
|
1535
|
-
const
|
|
1537
|
+
const shortfall = series.length < windowMonths
|
|
1538
|
+
? ` (requested ${windowMonths}; DataForSEO returned ${series.length} from ${dateFrom})`
|
|
1539
|
+
: '';
|
|
1540
|
+
const header = `**${cleaned}** — organic visibility, last ${series.length} months${shortfall}. Trend: ${verdict}\n\n` +
|
|
1536
1541
|
`| Period | Keywords | Pos 1–3 | Pos 4–10 | Pos 11–20 | Pos 21–100 | New | Lost | ETV |\n|---|---|---|---|---|---|---|---|---|`;
|
|
1537
1542
|
const rows = series.map(s => `| ${s.period} | ${fmtNum(s.keywords)} | ${fmtNum(s.pos_1_3)} | ${fmtNum(s.pos_4_10)} | ${fmtNum(s.pos_11_20)} | ${fmtNum(s.pos_21_100)} | ${fmtNum(s.isNew)} | ${fmtNum(s.isLost)} | ${fmtNum(s.etv)} |`);
|
|
1538
|
-
const md = capMdRows(header, rows, `\n\n${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}).`}`);
|
|
1543
|
+
const md = capMdRows(header, rows, `\n\n${r.cached ? 'Cached.' : `Live ($${r.cost.toFixed(4)}; billed per monthly item, date_from ${dateFrom}).`}`);
|
|
1539
1544
|
return {
|
|
1540
1545
|
content: [{ type: 'text', text: md }],
|
|
1541
|
-
structuredContent: { target: cleaned, months: series.length, verdict, series, cached: r.cached, cost: r.cost },
|
|
1546
|
+
structuredContent: { target: cleaned, months: series.length, requestedMonths: windowMonths, dateFrom, verdict, series, cached: r.cached, cost: r.cost },
|
|
1542
1547
|
};
|
|
1543
1548
|
});
|
|
1544
1549
|
server.registerTool('top_pages', {
|
|
@@ -1720,7 +1725,7 @@ export function createServer() {
|
|
|
1720
1725
|
browserLink(siteUrl);
|
|
1721
1726
|
return {
|
|
1722
1727
|
content: [{ type: 'text', text: md }],
|
|
1723
|
-
structuredContent: fp,
|
|
1728
|
+
structuredContent: { ...fp, cost: r.cost, cached: r.cached },
|
|
1724
1729
|
};
|
|
1725
1730
|
});
|
|
1726
1731
|
server.registerTool('market_sizing', {
|
|
@@ -1774,7 +1779,7 @@ export function createServer() {
|
|
|
1774
1779
|
browserLink(siteUrl);
|
|
1775
1780
|
return {
|
|
1776
1781
|
content: [{ type: 'text', text: md }],
|
|
1777
|
-
structuredContent: m,
|
|
1782
|
+
structuredContent: { ...m, cost, cached: cachedAll },
|
|
1778
1783
|
};
|
|
1779
1784
|
});
|
|
1780
1785
|
server.registerTool('topic_gaps', {
|
|
@@ -2058,7 +2063,7 @@ export function createServer() {
|
|
|
2058
2063
|
`${result.cached ? 'DataForSEO cached.' : `Live cost $${totalCost.toFixed(4)}.`} Persisted to link_prospects.` + browserLink(siteUrl);
|
|
2059
2064
|
return {
|
|
2060
2065
|
content: [{ type: 'text', text: capMdRows(header, rows, footer) }],
|
|
2061
|
-
structuredContent: result,
|
|
2066
|
+
structuredContent: { ...result, cost: totalCost, intersectCost: result.cost, deriveCost },
|
|
2062
2067
|
};
|
|
2063
2068
|
});
|
|
2064
2069
|
server.registerTool('resolve_entities', {
|