@houtini/seo-audit-console 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/LICENSE +92 -0
  2. package/README.md +211 -0
  3. package/dist/audit/checks.d.ts +35 -0
  4. package/dist/audit/checks.d.ts.map +1 -0
  5. package/dist/audit/checks.js +1475 -0
  6. package/dist/audit/checks.js.map +1 -0
  7. package/dist/audit/drift.d.ts +40 -0
  8. package/dist/audit/drift.d.ts.map +1 -0
  9. package/dist/audit/drift.js +148 -0
  10. package/dist/audit/drift.js.map +1 -0
  11. package/dist/audit/engine.d.ts +33 -0
  12. package/dist/audit/engine.d.ts.map +1 -0
  13. package/dist/audit/engine.js +186 -0
  14. package/dist/audit/engine.js.map +1 -0
  15. package/dist/audit/opportunities.d.ts +23 -0
  16. package/dist/audit/opportunities.d.ts.map +1 -0
  17. package/dist/audit/opportunities.js +149 -0
  18. package/dist/audit/opportunities.js.map +1 -0
  19. package/dist/audit/report.d.ts +11 -0
  20. package/dist/audit/report.d.ts.map +1 -0
  21. package/dist/audit/report.js +59 -0
  22. package/dist/audit/report.js.map +1 -0
  23. package/dist/audit/schema-validate.d.ts +35 -0
  24. package/dist/audit/schema-validate.d.ts.map +1 -0
  25. package/dist/audit/schema-validate.js +293 -0
  26. package/dist/audit/schema-validate.js.map +1 -0
  27. package/dist/audit/templates.d.ts +43 -0
  28. package/dist/audit/templates.d.ts.map +1 -0
  29. package/dist/audit/templates.js +129 -0
  30. package/dist/audit/templates.js.map +1 -0
  31. package/dist/audit/topicGaps.d.ts +42 -0
  32. package/dist/audit/topicGaps.d.ts.map +1 -0
  33. package/dist/audit/topicGaps.js +181 -0
  34. package/dist/audit/topicGaps.js.map +1 -0
  35. package/dist/core/AuditDatabase.d.ts +33 -0
  36. package/dist/core/AuditDatabase.d.ts.map +1 -0
  37. package/dist/core/AuditDatabase.js +481 -0
  38. package/dist/core/AuditDatabase.js.map +1 -0
  39. package/dist/core/Backlinks.d.ts +33 -0
  40. package/dist/core/Backlinks.d.ts.map +1 -0
  41. package/dist/core/Backlinks.js +110 -0
  42. package/dist/core/Backlinks.js.map +1 -0
  43. package/dist/core/Crawler.d.ts +23 -0
  44. package/dist/core/Crawler.d.ts.map +1 -0
  45. package/dist/core/Crawler.js +588 -0
  46. package/dist/core/Crawler.js.map +1 -0
  47. package/dist/core/DataForSeoClient.d.ts +86 -0
  48. package/dist/core/DataForSeoClient.d.ts.map +1 -0
  49. package/dist/core/DataForSeoClient.js +232 -0
  50. package/dist/core/DataForSeoClient.js.map +1 -0
  51. package/dist/core/Entities.d.ts +23 -0
  52. package/dist/core/Entities.d.ts.map +1 -0
  53. package/dist/core/Entities.js +62 -0
  54. package/dist/core/Entities.js.map +1 -0
  55. package/dist/core/GscClient.d.ts +22 -0
  56. package/dist/core/GscClient.d.ts.map +1 -0
  57. package/dist/core/GscClient.js +93 -0
  58. package/dist/core/GscClient.js.map +1 -0
  59. package/dist/core/GscSync.d.ts +20 -0
  60. package/dist/core/GscSync.d.ts.map +1 -0
  61. package/dist/core/GscSync.js +133 -0
  62. package/dist/core/GscSync.js.map +1 -0
  63. package/dist/core/JobManager.d.ts +30 -0
  64. package/dist/core/JobManager.d.ts.map +1 -0
  65. package/dist/core/JobManager.js +68 -0
  66. package/dist/core/JobManager.js.map +1 -0
  67. package/dist/core/RankTracker.d.ts +25 -0
  68. package/dist/core/RankTracker.d.ts.map +1 -0
  69. package/dist/core/RankTracker.js +78 -0
  70. package/dist/core/RankTracker.js.map +1 -0
  71. package/dist/core/Refresh.d.ts +32 -0
  72. package/dist/core/Refresh.d.ts.map +1 -0
  73. package/dist/core/Refresh.js +73 -0
  74. package/dist/core/Refresh.js.map +1 -0
  75. package/dist/core/UrlInspector.d.ts +22 -0
  76. package/dist/core/UrlInspector.d.ts.map +1 -0
  77. package/dist/core/UrlInspector.js +92 -0
  78. package/dist/core/UrlInspector.js.map +1 -0
  79. package/dist/core/WikidataClient.d.ts +19 -0
  80. package/dist/core/WikidataClient.d.ts.map +1 -0
  81. package/dist/core/WikidataClient.js +59 -0
  82. package/dist/core/WikidataClient.js.map +1 -0
  83. package/dist/core/agentReadiness.d.ts +34 -0
  84. package/dist/core/agentReadiness.d.ts.map +1 -0
  85. package/dist/core/agentReadiness.js +119 -0
  86. package/dist/core/agentReadiness.js.map +1 -0
  87. package/dist/core/ctrModel.d.ts +2 -0
  88. package/dist/core/ctrModel.d.ts.map +1 -0
  89. package/dist/core/ctrModel.js +6 -0
  90. package/dist/core/ctrModel.js.map +1 -0
  91. package/dist/core/dashboardData.d.ts +312 -0
  92. package/dist/core/dashboardData.d.ts.map +1 -0
  93. package/dist/core/dashboardData.js +550 -0
  94. package/dist/core/dashboardData.js.map +1 -0
  95. package/dist/core/dataStorage.d.ts +42 -0
  96. package/dist/core/dataStorage.d.ts.map +1 -0
  97. package/dist/core/dataStorage.js +193 -0
  98. package/dist/core/dataStorage.js.map +1 -0
  99. package/dist/core/draftBrief.d.ts +27 -0
  100. package/dist/core/draftBrief.d.ts.map +1 -0
  101. package/dist/core/draftBrief.js +69 -0
  102. package/dist/core/draftBrief.js.map +1 -0
  103. package/dist/core/extract.d.ts +67 -0
  104. package/dist/core/extract.d.ts.map +1 -0
  105. package/dist/core/extract.js +262 -0
  106. package/dist/core/extract.js.map +1 -0
  107. package/dist/core/gscFreshness.d.ts +18 -0
  108. package/dist/core/gscFreshness.d.ts.map +1 -0
  109. package/dist/core/gscFreshness.js +32 -0
  110. package/dist/core/gscFreshness.js.map +1 -0
  111. package/dist/core/linkGraph.d.ts +19 -0
  112. package/dist/core/linkGraph.d.ts.map +1 -0
  113. package/dist/core/linkGraph.js +125 -0
  114. package/dist/core/linkGraph.js.map +1 -0
  115. package/dist/core/passageScore.d.ts +19 -0
  116. package/dist/core/passageScore.d.ts.map +1 -0
  117. package/dist/core/passageScore.js +59 -0
  118. package/dist/core/passageScore.js.map +1 -0
  119. package/dist/core/paths.d.ts +5 -0
  120. package/dist/core/paths.d.ts.map +1 -0
  121. package/dist/core/paths.js +15 -0
  122. package/dist/core/paths.js.map +1 -0
  123. package/dist/core/queryData.d.ts +35 -0
  124. package/dist/core/queryData.d.ts.map +1 -0
  125. package/dist/core/queryData.js +200 -0
  126. package/dist/core/queryData.js.map +1 -0
  127. package/dist/core/reranker.d.ts +8 -0
  128. package/dist/core/reranker.d.ts.map +1 -0
  129. package/dist/core/reranker.js +68 -0
  130. package/dist/core/reranker.js.map +1 -0
  131. package/dist/core/robots.d.ts +11 -0
  132. package/dist/core/robots.d.ts.map +1 -0
  133. package/dist/core/robots.js +76 -0
  134. package/dist/core/robots.js.map +1 -0
  135. package/dist/core/sitemap.d.ts +15 -0
  136. package/dist/core/sitemap.d.ts.map +1 -0
  137. package/dist/core/sitemap.js +100 -0
  138. package/dist/core/sitemap.js.map +1 -0
  139. package/dist/core/sql.d.ts +6 -0
  140. package/dist/core/sql.d.ts.map +1 -0
  141. package/dist/core/sql.js +6 -0
  142. package/dist/core/sql.js.map +1 -0
  143. package/dist/core/types.d.ts +22 -0
  144. package/dist/core/types.d.ts.map +1 -0
  145. package/dist/core/types.js +2 -0
  146. package/dist/core/types.js.map +1 -0
  147. package/dist/core/url-key.d.ts +39 -0
  148. package/dist/core/url-key.d.ts.map +1 -0
  149. package/dist/core/url-key.js +106 -0
  150. package/dist/core/url-key.js.map +1 -0
  151. package/dist/generators/index.d.ts +45 -0
  152. package/dist/generators/index.d.ts.map +1 -0
  153. package/dist/generators/index.js +184 -0
  154. package/dist/generators/index.js.map +1 -0
  155. package/dist/index.d.ts +3 -0
  156. package/dist/index.d.ts.map +1 -0
  157. package/dist/index.js +10 -0
  158. package/dist/index.js.map +1 -0
  159. package/dist/server.d.ts +7 -0
  160. package/dist/server.d.ts.map +1 -0
  161. package/dist/server.js +1387 -0
  162. package/dist/server.js.map +1 -0
  163. package/dist/src/ui/dashboard.html +347 -0
  164. package/dist/src/ui/sync-progress.html +104 -0
  165. package/package.json +101 -0
  166. package/server.json +57 -0
package/LICENSE ADDED
@@ -0,0 +1,92 @@
1
+ SEO Audit Console — Source-Available Licence
2
+
3
+ Copyright (c) 2026 Richard Baxter / Houtini (houtini.com)
4
+
5
+ 1. Definitions
6
+
7
+ "Software" means this repository: source code, built outputs,
8
+ documentation, and all associated files.
9
+
10
+ "Non-commercial use" means personal use, private study, evaluation,
11
+ and academic or educational use, where the Software is not used in
12
+ the course of business, for client work, or to generate revenue,
13
+ directly or indirectly.
14
+
15
+ "Commercial use" means any use other than non-commercial use. This
16
+ includes (without limitation) use inside a business, use by an agency
17
+ or consultant in the course of client work, use in or as part of a
18
+ paid product or service, and internal use by a for-profit
19
+ organisation.
20
+
21
+ 2. What you may do without asking
22
+
23
+ You may download, install, build, and run the Software, unmodified,
24
+ for non-commercial use. You may read the source. You may link to this
25
+ repository. You may open issues and submit pull requests.
26
+
27
+ 3. What needs a commercial licence or written permission
28
+
29
+ a. Commercial use of the Software, in any form, requires a
30
+ commercial licence. Terms and how to buy one are published in
31
+ COMMERCIAL.md in this repository (or contact hello@houtini.com).
32
+ A valid commercial licence satisfies this section for the uses it
33
+ covers.
34
+
35
+ The following additionally require prior written permission from the
36
+ copyright holder, whether or not you hold a commercial licence:
37
+
38
+ b. Modifying the Software, or creating derivative works from it,
39
+ beyond changes strictly necessary to build and run it locally;
40
+ c. Redistributing the Software or any derivative of it, in source or
41
+ built form, on any registry, marketplace, or hosting service;
42
+ d. Offering the Software, or a service substantially built on it, to
43
+ third parties (including as a hosted or managed service).
44
+
45
+ Permission is usually straightforward to get. Contact:
46
+ hello@houtini.com
47
+
48
+ 4. Contributions
49
+
50
+ By submitting a pull request or patch you grant the copyright holder
51
+ a perpetual, worldwide, irrevocable, royalty-free licence to use,
52
+ modify, relicense, and distribute your contribution as part of the
53
+ Software.
54
+
55
+ 5. No trademark rights
56
+
57
+ This licence grants no rights to the "Houtini" or "SEO Audit Console"
58
+ names or logos.
59
+
60
+ 6. Termination
61
+
62
+ Any use outside the permissions above ends your rights under this
63
+ licence automatically. Rights are restored if the violation is cured
64
+ within 30 days of notice.
65
+
66
+ 7. No warranty; no liability
67
+
68
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
69
+ EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
70
+ MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
71
+ NON-INFRINGEMENT. IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE
72
+ FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF
73
+ CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
74
+ WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
75
+
76
+ 8. Change licence (automatic conversion to open source)
77
+
78
+ Each released version of the Software automatically converts to the
79
+ Apache License, Version 2.0, on the third anniversary of that
80
+ version's first publication (its "Change Date"). From a version's
81
+ Change Date, that version - and only that version - may be used,
82
+ modified, and redistributed under the Apache License 2.0, and the
83
+ restrictions in this licence no longer apply to it. Later versions
84
+ keep their own Change Dates.
85
+
86
+ 9. Governing law
87
+
88
+ This licence is governed by the laws of England and Wales.
89
+
90
+ Note: releases of this Software prior to this licence change were
91
+ published under the Apache License 2.0; that licence continues to apply
92
+ only to those earlier versions.
package/README.md ADDED
@@ -0,0 +1,211 @@
1
+ # SEO Audit Console
2
+
3
+ **A technical SEO audit you can hold a conversation with - built from your own Search Console data and a live crawl of your site, run inside Claude.**
4
+
5
+ [![License: Source-Available](https://img.shields.io/badge/License-Source--Available-orange.svg?style=flat-square)](./LICENSE)
6
+ [![MCP](https://img.shields.io/badge/Model_Context_Protocol-server-purple?style=flat-square)](https://modelcontextprotocol.io)
7
+ [![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?style=flat-square&logo=node.js&logoColor=white)](https://nodejs.org)
8
+
9
+ **The complete technical SEO audit, at conversation speed.** SEO Audit Console merges your **Google Search Console** history, a **first-party crawl** of your site, and on-demand **DataForSEO** market data into one prioritised audit inside Claude - from crawlability, indexation, canonicalisation, structured data, Core Web Vitals and hreflang right through to keyword cannibalisation, striking-distance queries, content gaps, competitor analysis and AI-search readiness. Ninety-three checks, every finding ranked by the clicks it could recover, every fix written for you: paste-ready redirects, JSON-LD, internal links and grounded content briefs. What used to be a fortnight of crawling, exporting and cross-referencing spreadsheets is twenty minutes and a prompt - and your data never leaves your machine.
10
+
11
+ **Built by [Houtini](https://houtini.com).** We build automation for the grunt work of digital marketing - the data collection, the crawling, the merging, the checking - so your team's time goes on the thinking, the strategy and the client work that needs a human. This plugin is that idea applied to the technical SEO audit.
12
+
13
+ ```console
14
+ you › run an SEO audit on simracingcockpit.gg
15
+
16
+ ⣾ search console 1.8M rows synced (19s - incremental)
17
+ ⣾ crawl 868 pages · HTTP/2 · robots-polite · 8 parallel
18
+ ⣾ link graph internal PageRank · click depth · in-degree
19
+ ✓ 93 checks · 220 findings · ranked by expected clicks per dev-hour
20
+
21
+ #1 CTR far below position-expected /how-to-install-mods XL
22
+ #2 Page losing clicks (trend) site-wide XL
23
+ #3 Keyword cannibalisation "beamng drive mods" L
24
+ #4 Robots-blocked page earning traffic /category/wheels L
25
+
26
+ you › generate the fix for #1 ▍
27
+ ```
28
+
29
+ ![The dashboard overview - executive summary, critical issues, recoverable clicks](assets/dashboard-overview.png)
30
+
31
+ ## The manual
32
+
33
+ This README is the story and the quick start. The detail lives in the manual:
34
+
35
+ | Page | What's in it |
36
+ |---|---|
37
+ | [Getting started](manual/getting-started.md) | Install, the GSC service-account setup (and the step everyone misses), Claude Desktop and Claude Code config, your first audit, troubleshooting |
38
+ | [Tool reference](manual/tools.md) | Every tool: what it does, inputs, joins, an example prompt |
39
+ | [The check registry](manual/checks.md) | All 93 checks with what each catches, its D/N label, and the fix |
40
+ | [Composition](manual/composition.md) | The join keys, the grains, and thirteen worked recipes for asking your own questions across the data |
41
+ | [Competitive analysis](manual/competitive.md) | The Semrush-replacement workflows, DataForSEO setup, and the real costs |
42
+ | [Dashboard & reports](manual/dashboard.md) | The six tabs, what each chart shows, and the shareable export |
43
+
44
+ ## Surprisingly little has changed in twenty years
45
+
46
+ The technical audit I was writing for clients in 2006 is, structurally, the audit most agencies still sell today. A crawler runs, a template fills, a 60-page PDF lands. Everything a crawler could find, in severity order, with no idea which findings are worth money and which are cosmetic.
47
+
48
+ What *has* changed is what's possible. Google gives every site owner a complete record of its search reality - which queries, which pages, how many impressions, where you ranked. Your crawl tells you what your site says. Search Console tells you what Google did about it. And in my experience, the gap between those two datasets is where nearly all of the recoverable traffic hides.
49
+
50
+ So that's what I built. SEO Audit Console is a [Model Context Protocol](https://modelcontextprotocol.io) server that merges your **Search Console history** with a **first-party crawl of your site** (and, when you want it, **DataForSEO**) into one thing: a prioritised, evidence-backed audit you can interrogate inside Claude Desktop. It hands you paste-ready fixes. Every finding traces back to a real datapoint.
51
+
52
+ One idea underneath all of it:
53
+
54
+ > **Your crawl is intent. Search Console is reality. The money is where they diverge.**
55
+
56
+ A flat crawler tells you a page 404s. Useful, but only just. This tells you the 404 is draining 15% of your homepage's internal PageRank, that the page used to earn 10,000 clicks a month, and it writes the 301 rule to fix it. It finds the page at position #3 on 150,000 impressions with a 0.2% click-through rate - a title rewrite probably worth thousands of clicks - and ranks that *above* the cosmetic findings. Severity is what crawlers sell you. Yield is what moves the numbers.
57
+
58
+ ### Who is this for?
59
+
60
+ The SEO consultant who wants the collection and checking automated so the thinking time survives. The in-house marketer who's been quoted four figures for a commodity audit. And anyone newer to this who wants to learn what a good audit looks at - because every finding shows its evidence, the tool doubles as a teacher.
61
+
62
+ A note on where to run it. Claude Desktop is the easy start, but in my view **Claude Code is the best home for this tool** - because it closes the loop. In a chat client the audit hands you a 301 rule to paste somewhere. In Claude Code, the same session has your site's repo, a terminal and git: the audit finds the issue, writes the fix, applies it to the codebase, commits it, and re-crawls to verify. Finding to deployed fix, one conversation.
63
+
64
+ You don't need to learn an interface. You type *"run an SEO audit on mysite.com"* into Claude and it happens. Forget what's possible? Ask *"run seo_audit_help"* and you get the full menu with example prompts.
65
+
66
+ ### Does the approach work?
67
+
68
+ Yes. The crawl-plus-GSC merge is not a novelty; it's the method. On one property, seeding the crawl from Search Console URLs took coverage of GSC-known pages from 29% to 70% - every one of those extra pages is a page a conventional crawl silently missed, and several were earning traffic with no internal links pointing at them at all. On the same property the incremental sync turned a 33-minute data refresh into 19 seconds, which is the difference between "audit quarterly" and "audit whenever you're curious".
69
+
70
+ ---
71
+
72
+ ## What the audit checks
73
+
74
+ `run_audit` executes **93 checks** over the joined data and returns a ranked list - not a wall of everything, a priority order with the traffic at stake attached to each finding. The families, briefly:
75
+
76
+ | Family | What it catches |
77
+ |---|---|
78
+ | **Crawlability & indexation** | Broken links, redirect chains, orphans, index bloat, spider-traps, robots-blocked pages still earning traffic, and the *reason* every URL isn't indexable |
79
+ | **On-page & structured data** | Titles, metas, H1s, alt text - plus a local validator covering ~30 rich-result types, required fields only, so it never nags about properties Google ignores |
80
+ | **Trends (GSC over time)** | Pages losing clicks, rankings slipping, vanished queries, rising pages worth doubling down on, stale content decaying year-on-year |
81
+ | **The merged questions** | Cannibalisation, striking distance, ghost pages, traffic to dead URLs, internal authority wasted on no-click pages, titles missing the query you already rank for |
82
+ | **AI-search readiness** | Phrases you rank for but never say, queries your copy never answers in one passage, content that doesn't chunk cleanly for retrieval |
83
+
84
+ Every check is labelled **D** (deterministic - here are the bytes) or **N** (judgement - off by default, ask for *"the judgement findings"* to see them). In my view a wrong finding is worse than no finding at all, so the heuristic checks have to ask permission. The full registry, check by check, is in [the manual](manual/checks.md).
85
+
86
+ And if you grew up on desktop crawlers, the dashboard's Site health tab will feel like home - response codes, indexability reasons, crawl depth, the heaviest images, server errors and slow pages, all as clean stat bars. The tab-by-tab tour is in [dashboard.md](manual/dashboard.md).
87
+
88
+ ---
89
+
90
+ ## The crawl, properly explained
91
+
92
+ The crawl is where audits usually go wrong, so it's worth understanding what this one does differently. I've spent enough of my career cleaning up after crawlers that fooled themselves.
93
+
94
+ **It discovers pages three ways.** Following links, reading your XML sitemaps, and - the important one - starting from every URL Google is already sending traffic to, straight out of your GSC data. Coverage stops depending on your sitemap being honest. It's also exactly how ghost pages get caught: if Google ranks a URL your own site structure can't reach, that URL still gets crawled, and the mismatch becomes a finding.
95
+
96
+ **It records *why*, not just *what*.** For every URL that isn't indexable it stores the reason - 404, noindex, X-Robots header, canonicalised elsewhere, robots-blocked, non-HTML. "This page won't rank" is a fact; "this page won't rank because a plugin set an X-Robots header nobody remembers" is a fix.
97
+
98
+ **It refuses to be fooled.** A redirect that leaves your site (Shopify OAuth flows, I'm looking at you) is recorded as a redirect-out, never stored as a page. It always uses GET rather than HEAD, because a HEAD request can return a different status than the real request would - but it abandons the body for images, PDFs and assets, so it records status and size without downloading the bytes.
99
+
100
+ **It's quick without being rude.** HTTP/2 where your origin supports it, gzip and brotli negotiated, keep-alive connections reused. The speed comes from efficiency, not from hammering your server. It respects robots.txt properly (a bot-specific group replaces `*`, per the spec, which plenty of commercial crawlers get wrong), backs off when your host rate-limits, and skips the junk: internal search, faceted filter combinations, login flows. This is a crawler for sites you own. Being a good guest is the point.
101
+
102
+ After the crawl it computes a real link graph: internal PageRank with nav and footer links down-weighted, click depth from the homepage counting body links only, in-degree per page. That graph powers the orphan, equity-leak and underlinked-page checks - and the donor rankings when the tool suggests internal links.
103
+
104
+ ---
105
+
106
+ ## The workflows, briefly
107
+
108
+ Each of these is a real procedure I use, and each is one prompt. The expanded versions, with what happens underneath, live in the manual pages linked.
109
+
110
+ 1. **Your first audit.** *"Refresh sc-domain:mysite.com"*, then *"Run an SEO audit on mysite.com"*. Twenty minutes on a mid-size site, and the top five findings are usually worth more than the other eighty-five combined. Then *"generate the fix for #1"*. → [getting-started.md](manual/getting-started.md)
111
+ 2. **Sitewide keyword optimisation.** The question isn't "what keywords should I target?" - it's "where does my copy fail to say what I already rank for?" *"Score the passages on mysite.com"* runs a small local relevance model over every ranking page; *"draft the missing content for /page"* writes the fix in your site's own voice, grounded so it invents nothing. → [tools.md](manual/tools.md#ai-search-readiness)
112
+ 3. **Cannibalisation.** *"Show me the cannibalisation findings with evidence"* - thresholds tuned so incidental long-tail overlap doesn't count, so the consolidate-or-differentiate call is made on numbers. → [checks.md](manual/checks.md#the-merged-checks-23)
113
+ 4. **The content plan.** *"Suggest new pages for mysite.com"* mines demand Google already shows you; *"what topics should mysite.com cover?"* maps the demand it doesn't. Together: a quarter's plan. → [competitive.md](manual/competitive.md)
114
+ 5. **The template play.** *"List the page templates"* - big sites aren't 50,000 pages, they're a dozen templates repeated, and one template fix corrects the whole cluster. → [tools.md](manual/tools.md#list_templates)
115
+ 6. **Monitoring and migrations.** *"Detect changes on mysite.com"* diffs your two most recent crawls by severity. During a migration this is the difference between catching a stray noindex on Tuesday and explaining a traffic graph in a board meeting three weeks later. I've been on the wrong end of that one. → [tools.md](manual/tools.md#monitoring)
116
+ 7. **AI-search readiness.** *"Check agent readiness for mysite.com"* - the web is quietly growing a second audience, and almost no SEO tool checks any of it. → [tools.md](manual/tools.md#check_agent_readiness)
117
+ 8. **Your own questions.** The four datasets share three join keys, and the most valuable analyses are the ones you compose yourself - *"which pages lost clicks after being cited in AI Overviews?"* is one prompt here and a feature nowhere else. → [composition.md](manual/composition.md)
118
+ 9. **Reporting.** *"Show me the dashboard"* in the chat, or *"export the report"* as one self-contained HTML file you can send a client. → [dashboard.md](manual/dashboard.md)
119
+
120
+ ![Ranking distribution over time - impressions by position bucket](assets/search-performance.png)
121
+
122
+ ---
123
+
124
+ ## What DataForSEO adds (and what it costs)
125
+
126
+ Everything above works with just your Search Console data. But GSC can only describe searches where you already appear. The moment your question is "how big is this market?" or "what do competitors rank for that I don't?", you need third-party data - and that's [DataForSEO](https://dataforseo.com/?aff=213701): a pay-as-you-go API for volumes, live rankings, competitor data and Lighthouse runs. No subscription; calls cost fractions of a cent to a few cents, cached for 20 days, and only ever run when you ask. My own usage runs to a few dollars a month.
127
+
128
+ It unlocks the Semrush-replacement layer: the organic visibility overview for any domain, any site's top pages and ranked keywords (including which keywords cite a site in **AI Overviews**), the content gap, topic gaps, search intent, lab Core Web Vitals and backlinks. The full workflows and setup: [competitive.md](manual/competitive.md).
129
+
130
+ ---
131
+
132
+ ## Installation, in brief
133
+
134
+ Three steps - the full walkthrough with the gotchas is [getting-started.md](manual/getting-started.md):
135
+
136
+ 1. **Get it:** the quick route is npx - point your MCP config at `npx -y @houtini/seo-audit-console` and there's nothing to build. Or clone this repo, `npm install`, `npm run build` if you want the source. Either way, Node ≥ 20.
137
+ 2. **Connect Search Console:** create a Google Cloud service account, download its JSON key, and - the step everyone misses - **add the service account's email as a user on your property** in Search Console.
138
+ 3. **Point your MCP client at `dist/index.js`** with `GOOGLE_APPLICATION_CREDENTIALS` set. Works in Claude Desktop and Claude Code; only the one env var is required.
139
+
140
+ Then: *"list properties"* to check it's connected, *"refresh"*, *"run an SEO audit"*.
141
+
142
+ ---
143
+
144
+ ## The tools, at a glance
145
+
146
+ The one-line version - full descriptions, inputs and example prompts for every tool are in [the tool reference](manual/tools.md).
147
+
148
+ | Tool | What it does |
149
+ |---|---|
150
+ | `refresh_property` | Sync GSC + crawl + inspect + rank history, one job |
151
+ | `sync_gsc` · `start_crawl` · `inspect_urls` · `track_ranks` | Run a single part on its own |
152
+ | `check_sync_status` · `check_crawl_status` | Watch a long job's progress |
153
+ | `run_audit` · `query_audit` · `list_checks` | The scored audit · one check with evidence · the catalogue |
154
+ | `query_data` | Read-only queries over the raw tables - aggregates in the database, answers not rows |
155
+ | `fix_finding` | Paste-ready remediation (JSON-LD / 301 / internal links) |
156
+ | `detect_changes` | What changed between the two most recent crawls |
157
+ | `check_agent_readiness` | 0-100 AI-agent readiness score with copy-paste fixes |
158
+ | `list_templates` · `suggest_pages` | Template clusters · new-page ideas from real demand |
159
+ | `score_passages` · `draft_content` | Local relevance scoring · grounded writing briefs |
160
+ | `resolve_entities` | Wikidata entities and the link gaps between them |
161
+ | `keyword_volume` · `related_terms` · `search_intent` | DataForSEO keyword data |
162
+ | `competitors_domain` · `page_intersection` · `topic_gaps` | Competitors · content gap · topic gaps |
163
+ | `domain_visibility` · `top_pages` · `ranked_keywords` | The Semrush-style views, any domain (+ `aioOnly` for AI Overview citations) |
164
+ | `page_lighthouse` · `pull_backlinks` | Lab CWV · backlink profile with live status |
165
+ | `get_dashboard` · `export_report` | In-chat dashboard · shareable HTML |
166
+ | `composition_cookbook` | The data-surface map and recipes for bespoke analyses |
167
+ | `data_storage` | Per-property disk usage and row counts, with confirm-gated pruning |
168
+ | `normalize_url` · `data_location` · `list_properties` · `seo_audit_help` | Utilities and the help menu |
169
+
170
+ ---
171
+
172
+ ## How it works under the hood
173
+
174
+ - **The join key (`url_key`).** GSC `page` and crawl `url` both normalise down to the same key - force HTTPS, unify www and apex, strip tracking params, and so on. Everything joins on that. It's the whole trick, really.
175
+ - **One SQLite database per property** (WAL, prepared statements). Your data stays on your machine.
176
+ - **Scored once, sorted by yield.** `(expected clicks × yield × certainty) / effort`. Covering-indexed, so the audit stays fast even when the GSC table runs to millions of rows.
177
+ - **Careful with your history.** Crawls and syncs never destroy the previous snapshot until new data has started arriving - a site outage mid-crawl doesn't cost you your data.
178
+ - **Owned-site only, dry-run fixes.** It crawls sites you control, and the generators return artifacts. They never write to your site. That's a line I won't cross.
179
+
180
+ ## Privacy and data
181
+
182
+ Your Search Console data and the crawl live in local SQLite files under `SAC_DATA_DIR`. The passage-scoring model runs locally too. Nothing leaves your machine except the API calls *you* trigger - Google (your own GSC) and, if you've set it up, DataForSEO. No telemetry. None.
183
+
184
+ ---
185
+
186
+ ## What's coming
187
+
188
+ A few things I'm building next, in rough order:
189
+
190
+ - **List mode.** Paste any list of URLs - a migration map, old ranking pages, a PPC export - and have it status-checked and audited. The migration-verification workflow, basically.
191
+ - **Structured-data opportunities, by template.** Not "you have no schema" (most modern stores have plenty), but "this template could earn review stars or an FAQ rich result and doesn't."
192
+ - **A per-page content scorecard.** A dedicated table of *every* ranking phrase your copy misses, not just the top query.
193
+ - **More agent readiness.** A WebMCP advisory (which tool actions your site could expose to agents) and the agent-commerce protocols.
194
+ - **A printable report.** A proper A4 document you can hand a client, not a slide deck.
195
+ - **Source-level parser checks.** The audit spec is written - ~94 checks covering the layer most tools never touch: elements that silently break `<head>` parsing, directives hoisted into the body and ignored, raw-vs-rendered divergence.
196
+
197
+ Got a weird edge case you wish a tool caught? Tell me - that's exactly how the merged GSC×crawl checks got built.
198
+
199
+ ## About Houtini
200
+
201
+ [Houtini](https://houtini.com) exists for one reason: the hours your team loses to grunt work. Pulling Search Console exports, running crawls, cross-referencing spreadsheets, re-checking what changed since last month - none of it needs a person, and all of it eats the time your people should be spending on strategy, on clients, on the work that moves numbers. So we automate exactly that layer. SEO Audit Console is one of a family of tools built on the same principle - if a machine can collect it, merge it and check it, a machine should.
202
+
203
+ Questions, licensing, or something you'd like automated: **hello@houtini.com**
204
+
205
+ ## Contributing
206
+
207
+ Issues and PRs welcome. The check registry (`src/audit/checks.ts`) is built to be extended - each check is a pure read over the joined data that returns findings with evidence, so adding one is fairly self-contained. There's an end-to-end smoke test (`npm run smoke`) and per-feature probes (`npm run probe:*`) to keep you honest. By submitting a PR you grant the licence set out in the [LICENSE](./LICENSE) contributions clause.
208
+
209
+ ## License
210
+
211
+ **Source-available, converting to open source.** Free to download, build, and run unmodified for personal, evaluation, and educational use. Commercial use - including agency and client work - needs a [commercial licence](./COMMERCIAL.md), which is a short email away. Modifications and redistribution need written permission. And every released version automatically becomes **Apache 2.0** three years after its release, so nothing stays locked up forever. Full terms in [LICENSE](./LICENSE).
@@ -0,0 +1,35 @@
1
+ import type Database from 'better-sqlite3';
2
+ import type { FindingLabel, Severity } from '../core/AuditDatabase.js';
3
+ import { expectedCtr } from '../core/ctrModel.js';
4
+ /**
5
+ * Check registry. Each check is a pure read over the AuditDatabase (crawl + GSC +
6
+ * url_inspection, joined on url_key/page_key) returning affected items with evidence.
7
+ * Scoring + persistence happen in engine.ts. Grounded in research/01 (master checklist),
8
+ * research/05 (merged GSC×crawl) and research/02 (war-stories); taxonomy cross-checked
9
+ * against SEOmator's MIT 251-rule set.
10
+ */
11
+ export interface CheckContext {
12
+ db: Database.Database;
13
+ gscMaxDate: string | null;
14
+ brand?: string | null;
15
+ }
16
+ export interface RawFinding {
17
+ urlKey?: string | null;
18
+ evidence: Record<string, unknown>;
19
+ }
20
+ export interface CheckDef {
21
+ id: string;
22
+ category: 'crawlability' | 'indexation' | 'onpage' | 'content' | 'schema' | 'security' | 'performance' | 'war-stories' | 'merged';
23
+ severity: Severity;
24
+ labels: FindingLabel[];
25
+ certainty: number;
26
+ effortBase: number;
27
+ fixType: 'global' | 'per-page' | 'automated';
28
+ yieldCoef?: number;
29
+ title: string;
30
+ fix: string;
31
+ run(ctx: CheckContext): RawFinding[];
32
+ }
33
+ export { expectedCtr };
34
+ export declare const CHECKS: CheckDef[];
35
+ //# sourceMappingURL=checks.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"checks.d.ts","sourceRoot":"","sources":["../../src/audit/checks.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,QAAQ,MAAM,gBAAgB,CAAC;AAC3C,OAAO,KAAK,EAAE,YAAY,EAAE,QAAQ,EAAE,MAAM,0BAA0B,CAAC;AAIvE,OAAO,EAAE,WAAW,EAAE,MAAM,qBAAqB,CAAC;AAIlD;;;;;;GAMG;AACH,MAAM,WAAW,YAAY;IAC3B,EAAE,EAAE,QAAQ,CAAC,QAAQ,CAAC;IACtB,UAAU,EAAE,MAAM,GAAG,IAAI,CAAC;IAC1B,KAAK,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;CACvB;AACD,MAAM,WAAW,UAAU;IACzB,MAAM,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IACvB,QAAQ,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACnC;AACD,MAAM,WAAW,QAAQ;IACvB,EAAE,EAAE,MAAM,CAAC;IACX,QAAQ,EAAE,cAAc,GAAG,YAAY,GAAG,QAAQ,GAAG,SAAS,GAAG,QAAQ,GAAG,UAAU,GAAG,aAAa,GAAG,aAAa,GAAG,QAAQ,CAAC;IAClI,QAAQ,EAAE,QAAQ,CAAC;IACnB,MAAM,EAAE,YAAY,EAAE,CAAC;IACvB,SAAS,EAAE,MAAM,CAAC;IAClB,UAAU,EAAE,MAAM,CAAC;IACnB,OAAO,EAAE,QAAQ,GAAG,UAAU,GAAG,WAAW,CAAC;IAC7C,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;IACZ,GAAG,CAAC,GAAG,EAAE,YAAY,GAAG,UAAU,EAAE,CAAC;CACtC;AAiFD,OAAO,EAAE,WAAW,EAAE,CAAC;AAEvB,eAAO,MAAM,MAAM,EAAE,QAAQ,EA0qC5B,CAAC"}